Sources
Create sources
Creates one or more sources. Accepts an array.
- URL sources are crawled and ingested automatically.
- File sources return a
presigned_urlandcontent_typein the response — upload the file bytes to that URL, then mark the source ready with Update a source (file_upload_status: COMPLETE).
Attach sources to a knowledge base with Add sources to a knowledge base. Requires a token with write access.
POST
/
knowledge-base
/
sources
Create sources
curl --request POST \
--url https://api.anyreach.ai/knowledge-base/sources \
--header 'Content-Type: application/json' \
--data '
[
{
"name": "<string>",
"chunking_strategy": {
"chunk_size": 1000
},
"id": "<string>",
"domain": "<string>",
"file_processor": "pymupdf",
"url_crawl_options": {
"limit": 10,
"max_depth": 3,
"max_discovery_depth": 3,
"include_paths": [
"<string>"
],
"exclude_paths": [
"<string>"
],
"add_to_datasets": [
"<string>"
],
"crawler_provider": "firecrawl"
},
"file_size": 123
}
]
'import requests
url = "https://api.anyreach.ai/knowledge-base/sources"
payload = [
{
"name": "<string>",
"chunking_strategy": { "chunk_size": 1000 },
"id": "<string>",
"domain": "<string>",
"file_processor": "pymupdf",
"url_crawl_options": {
"limit": 10,
"max_depth": 3,
"max_discovery_depth": 3,
"include_paths": ["<string>"],
"exclude_paths": ["<string>"],
"add_to_datasets": ["<string>"],
"crawler_provider": "firecrawl"
},
"file_size": 123
}
]
headers = {"Content-Type": "application/json"}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'Content-Type': 'application/json'},
body: JSON.stringify([
{
name: '<string>',
chunking_strategy: {chunk_size: 1000},
id: '<string>',
domain: '<string>',
file_processor: 'pymupdf',
url_crawl_options: {
limit: 10,
max_depth: 3,
max_discovery_depth: 3,
include_paths: ['<string>'],
exclude_paths: ['<string>'],
add_to_datasets: ['<string>'],
crawler_provider: 'firecrawl'
},
file_size: 123
}
])
};
fetch('https://api.anyreach.ai/knowledge-base/sources', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.anyreach.ai/knowledge-base/sources",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
[
'name' => '<string>',
'chunking_strategy' => [
'chunk_size' => 1000
],
'id' => '<string>',
'domain' => '<string>',
'file_processor' => 'pymupdf',
'url_crawl_options' => [
'limit' => 10,
'max_depth' => 3,
'max_discovery_depth' => 3,
'include_paths' => [
'<string>'
],
'exclude_paths' => [
'<string>'
],
'add_to_datasets' => [
'<string>'
],
'crawler_provider' => 'firecrawl'
],
'file_size' => 123
]
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.anyreach.ai/knowledge-base/sources"
payload := strings.NewReader("[\n {\n \"name\": \"<string>\",\n \"chunking_strategy\": {\n \"chunk_size\": 1000\n },\n \"id\": \"<string>\",\n \"domain\": \"<string>\",\n \"file_processor\": \"pymupdf\",\n \"url_crawl_options\": {\n \"limit\": 10,\n \"max_depth\": 3,\n \"max_discovery_depth\": 3,\n \"include_paths\": [\n \"<string>\"\n ],\n \"exclude_paths\": [\n \"<string>\"\n ],\n \"add_to_datasets\": [\n \"<string>\"\n ],\n \"crawler_provider\": \"firecrawl\"\n },\n \"file_size\": 123\n }\n]")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.anyreach.ai/knowledge-base/sources")
.header("Content-Type", "application/json")
.body("[\n {\n \"name\": \"<string>\",\n \"chunking_strategy\": {\n \"chunk_size\": 1000\n },\n \"id\": \"<string>\",\n \"domain\": \"<string>\",\n \"file_processor\": \"pymupdf\",\n \"url_crawl_options\": {\n \"limit\": 10,\n \"max_depth\": 3,\n \"max_discovery_depth\": 3,\n \"include_paths\": [\n \"<string>\"\n ],\n \"exclude_paths\": [\n \"<string>\"\n ],\n \"add_to_datasets\": [\n \"<string>\"\n ],\n \"crawler_provider\": \"firecrawl\"\n },\n \"file_size\": 123\n }\n]")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.anyreach.ai/knowledge-base/sources")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Content-Type"] = 'application/json'
request.body = "[\n {\n \"name\": \"<string>\",\n \"chunking_strategy\": {\n \"chunk_size\": 1000\n },\n \"id\": \"<string>\",\n \"domain\": \"<string>\",\n \"file_processor\": \"pymupdf\",\n \"url_crawl_options\": {\n \"limit\": 10,\n \"max_depth\": 3,\n \"max_discovery_depth\": 3,\n \"include_paths\": [\n \"<string>\"\n ],\n \"exclude_paths\": [\n \"<string>\"\n ],\n \"add_to_datasets\": [\n \"<string>\"\n ],\n \"crawler_provider\": \"firecrawl\"\n },\n \"file_size\": 123\n }\n]"
response = http.request(request)
puts response.read_body[
{
"id": "3fa85f64-5717-4562-b3fc-2c963f66afa6",
"organization_id": "3fa85f64-5717-4562-b3fc-2c963f66afa6",
"created_at": "2026-01-15T09:30:00Z",
"updated_at": "2026-01-15T09:30:00Z",
"created_by": "user:3fa85f64-5717-4562-b3fc-2c963f66afa6",
"modified_by": "user:3fa85f64-5717-4562-b3fc-2c963f66afa6",
"type": "FILE",
"name": "Example",
"domain": "string",
"file_upload_status": "PENDING",
"chunking_strategy": {
"method": "fixed",
"chunk_size": 1000
},
"file_processor": "pymupdf",
"file_size": 1,
"description": "string",
"presigned_url": "https://example.com/resource",
"content_type": "string"
}
]{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>"
}
]
}Headers
Organization ID for user PATs (pat_ prefix tokens)
Body
application/json
Available options:
FILE, URL Show child attributes
Show child attributes
Available options:
pymupdf Show child attributes
Show child attributes
Response
Successful Response
Available options:
FILE, URL Available options:
PENDING, IN_PROGRESS, COMPLETE, FAILED Show child attributes
Show child attributes
Available options:
pymupdf ⌘I
Create sources
curl --request POST \
--url https://api.anyreach.ai/knowledge-base/sources \
--header 'Content-Type: application/json' \
--data '
[
{
"name": "<string>",
"chunking_strategy": {
"chunk_size": 1000
},
"id": "<string>",
"domain": "<string>",
"file_processor": "pymupdf",
"url_crawl_options": {
"limit": 10,
"max_depth": 3,
"max_discovery_depth": 3,
"include_paths": [
"<string>"
],
"exclude_paths": [
"<string>"
],
"add_to_datasets": [
"<string>"
],
"crawler_provider": "firecrawl"
},
"file_size": 123
}
]
'import requests
url = "https://api.anyreach.ai/knowledge-base/sources"
payload = [
{
"name": "<string>",
"chunking_strategy": { "chunk_size": 1000 },
"id": "<string>",
"domain": "<string>",
"file_processor": "pymupdf",
"url_crawl_options": {
"limit": 10,
"max_depth": 3,
"max_discovery_depth": 3,
"include_paths": ["<string>"],
"exclude_paths": ["<string>"],
"add_to_datasets": ["<string>"],
"crawler_provider": "firecrawl"
},
"file_size": 123
}
]
headers = {"Content-Type": "application/json"}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'Content-Type': 'application/json'},
body: JSON.stringify([
{
name: '<string>',
chunking_strategy: {chunk_size: 1000},
id: '<string>',
domain: '<string>',
file_processor: 'pymupdf',
url_crawl_options: {
limit: 10,
max_depth: 3,
max_discovery_depth: 3,
include_paths: ['<string>'],
exclude_paths: ['<string>'],
add_to_datasets: ['<string>'],
crawler_provider: 'firecrawl'
},
file_size: 123
}
])
};
fetch('https://api.anyreach.ai/knowledge-base/sources', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.anyreach.ai/knowledge-base/sources",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
[
'name' => '<string>',
'chunking_strategy' => [
'chunk_size' => 1000
],
'id' => '<string>',
'domain' => '<string>',
'file_processor' => 'pymupdf',
'url_crawl_options' => [
'limit' => 10,
'max_depth' => 3,
'max_discovery_depth' => 3,
'include_paths' => [
'<string>'
],
'exclude_paths' => [
'<string>'
],
'add_to_datasets' => [
'<string>'
],
'crawler_provider' => 'firecrawl'
],
'file_size' => 123
]
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.anyreach.ai/knowledge-base/sources"
payload := strings.NewReader("[\n {\n \"name\": \"<string>\",\n \"chunking_strategy\": {\n \"chunk_size\": 1000\n },\n \"id\": \"<string>\",\n \"domain\": \"<string>\",\n \"file_processor\": \"pymupdf\",\n \"url_crawl_options\": {\n \"limit\": 10,\n \"max_depth\": 3,\n \"max_discovery_depth\": 3,\n \"include_paths\": [\n \"<string>\"\n ],\n \"exclude_paths\": [\n \"<string>\"\n ],\n \"add_to_datasets\": [\n \"<string>\"\n ],\n \"crawler_provider\": \"firecrawl\"\n },\n \"file_size\": 123\n }\n]")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.anyreach.ai/knowledge-base/sources")
.header("Content-Type", "application/json")
.body("[\n {\n \"name\": \"<string>\",\n \"chunking_strategy\": {\n \"chunk_size\": 1000\n },\n \"id\": \"<string>\",\n \"domain\": \"<string>\",\n \"file_processor\": \"pymupdf\",\n \"url_crawl_options\": {\n \"limit\": 10,\n \"max_depth\": 3,\n \"max_discovery_depth\": 3,\n \"include_paths\": [\n \"<string>\"\n ],\n \"exclude_paths\": [\n \"<string>\"\n ],\n \"add_to_datasets\": [\n \"<string>\"\n ],\n \"crawler_provider\": \"firecrawl\"\n },\n \"file_size\": 123\n }\n]")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.anyreach.ai/knowledge-base/sources")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Content-Type"] = 'application/json'
request.body = "[\n {\n \"name\": \"<string>\",\n \"chunking_strategy\": {\n \"chunk_size\": 1000\n },\n \"id\": \"<string>\",\n \"domain\": \"<string>\",\n \"file_processor\": \"pymupdf\",\n \"url_crawl_options\": {\n \"limit\": 10,\n \"max_depth\": 3,\n \"max_discovery_depth\": 3,\n \"include_paths\": [\n \"<string>\"\n ],\n \"exclude_paths\": [\n \"<string>\"\n ],\n \"add_to_datasets\": [\n \"<string>\"\n ],\n \"crawler_provider\": \"firecrawl\"\n },\n \"file_size\": 123\n }\n]"
response = http.request(request)
puts response.read_body[
{
"id": "3fa85f64-5717-4562-b3fc-2c963f66afa6",
"organization_id": "3fa85f64-5717-4562-b3fc-2c963f66afa6",
"created_at": "2026-01-15T09:30:00Z",
"updated_at": "2026-01-15T09:30:00Z",
"created_by": "user:3fa85f64-5717-4562-b3fc-2c963f66afa6",
"modified_by": "user:3fa85f64-5717-4562-b3fc-2c963f66afa6",
"type": "FILE",
"name": "Example",
"domain": "string",
"file_upload_status": "PENDING",
"chunking_strategy": {
"method": "fixed",
"chunk_size": 1000
},
"file_processor": "pymupdf",
"file_size": 1,
"description": "string",
"presigned_url": "https://example.com/resource",
"content_type": "string"
}
]{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>"
}
]
}
