curl --request POST \
--url https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads \
--header 'Content-Type: application/json' \
--header 'x-api-key: <api-key>' \
--data '
{
"data_source_config": {
"source": "S3",
"s3_bucket": "<string>",
"aws_region": "<string>",
"aws_account_id": "<string>",
"s3_prefix": ""
},
"data_source_auth_config": {
"source": "SharePoint",
"client_secret": "<string>",
"encrypted": false
},
"chunking_strategy_config": {
"strategy": "character",
"separator": "\n\n",
"chunk_size": 1000,
"chunk_overlap": 200
},
"force_reupload": false,
"tagging_information": {
"type": "per_file",
"tags_to_apply": {}
}
}
'import requests
url = "https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads"
payload = {
"data_source_config": {
"source": "S3",
"s3_bucket": "<string>",
"aws_region": "<string>",
"aws_account_id": "<string>",
"s3_prefix": ""
},
"data_source_auth_config": {
"source": "SharePoint",
"client_secret": "<string>",
"encrypted": False
},
"chunking_strategy_config": {
"strategy": "character",
"separator": "
",
"chunk_size": 1000,
"chunk_overlap": 200
},
"force_reupload": False,
"tagging_information": {
"type": "per_file",
"tags_to_apply": {}
}
}
headers = {
"x-api-key": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'x-api-key': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
data_source_config: {
source: 'S3',
s3_bucket: '<string>',
aws_region: '<string>',
aws_account_id: '<string>',
s3_prefix: ''
},
data_source_auth_config: {source: 'SharePoint', client_secret: '<string>', encrypted: false},
chunking_strategy_config: {strategy: 'character', separator: '\n\n', chunk_size: 1000, chunk_overlap: 200},
force_reupload: false,
tagging_information: {type: 'per_file', tags_to_apply: {}}
})
};
fetch('https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'data_source_config' => [
'source' => 'S3',
's3_bucket' => '<string>',
'aws_region' => '<string>',
'aws_account_id' => '<string>',
's3_prefix' => ''
],
'data_source_auth_config' => [
'source' => 'SharePoint',
'client_secret' => '<string>',
'encrypted' => false
],
'chunking_strategy_config' => [
'strategy' => 'character',
'separator' => '
',
'chunk_size' => 1000,
'chunk_overlap' => 200
],
'force_reupload' => false,
'tagging_information' => [
'type' => 'per_file',
'tags_to_apply' => [
]
]
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"x-api-key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads"
payload := strings.NewReader("{\n \"data_source_config\": {\n \"source\": \"S3\",\n \"s3_bucket\": \"<string>\",\n \"aws_region\": \"<string>\",\n \"aws_account_id\": \"<string>\",\n \"s3_prefix\": \"\"\n },\n \"data_source_auth_config\": {\n \"source\": \"SharePoint\",\n \"client_secret\": \"<string>\",\n \"encrypted\": false\n },\n \"chunking_strategy_config\": {\n \"strategy\": \"character\",\n \"separator\": \"\\n\\n\",\n \"chunk_size\": 1000,\n \"chunk_overlap\": 200\n },\n \"force_reupload\": false,\n \"tagging_information\": {\n \"type\": \"per_file\",\n \"tags_to_apply\": {}\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("x-api-key", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads")
.header("x-api-key", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"data_source_config\": {\n \"source\": \"S3\",\n \"s3_bucket\": \"<string>\",\n \"aws_region\": \"<string>\",\n \"aws_account_id\": \"<string>\",\n \"s3_prefix\": \"\"\n },\n \"data_source_auth_config\": {\n \"source\": \"SharePoint\",\n \"client_secret\": \"<string>\",\n \"encrypted\": false\n },\n \"chunking_strategy_config\": {\n \"strategy\": \"character\",\n \"separator\": \"\\n\\n\",\n \"chunk_size\": 1000,\n \"chunk_overlap\": 200\n },\n \"force_reupload\": false,\n \"tagging_information\": {\n \"type\": \"per_file\",\n \"tags_to_apply\": {}\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["x-api-key"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"data_source_config\": {\n \"source\": \"S3\",\n \"s3_bucket\": \"<string>\",\n \"aws_region\": \"<string>\",\n \"aws_account_id\": \"<string>\",\n \"s3_prefix\": \"\"\n },\n \"data_source_auth_config\": {\n \"source\": \"SharePoint\",\n \"client_secret\": \"<string>\",\n \"encrypted\": false\n },\n \"chunking_strategy_config\": {\n \"strategy\": \"character\",\n \"separator\": \"\\n\\n\",\n \"chunk_size\": 1000,\n \"chunk_overlap\": 200\n },\n \"force_reupload\": false,\n \"tagging_information\": {\n \"type\": \"per_file\",\n \"tags_to_apply\": {}\n }\n}"
response = http.request(request)
puts response.read_body{
"upload_id": "<string>"
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>",
"input": "<unknown>",
"ctx": {}
}
]
}Submit Upload Job
Starts an asynchronous ingestion job that syncs the knowledge base with a data source, and returns immediately with an upload_id to poll via the Get Upload endpoint. The request body is a union that supports three modes: an inline data source config, a reference to a previously created data source by data_source_id, or a set of locally provided chunks. On each run the job detects new, updated, and deleted artifacts since the last upload, then extracts, chunks (per the chunking strategy config), and embeds new/updated content while removing deleted artifacts, so uploads are eventually consistent and idempotent — retrying with the same data source and chunking config resumes without duplication, while changing the chunking config forces re-chunking. Before starting, it validates data source access and the chunking config, and it rejects the request if another upload for the same data source is already in progress or if the knowledge base is mid-migration. PER_FILE tagging information is not accepted here (use the file-upload endpoint for that); it is rejected with a validation error. If the backend fails to launch the ingestion workflow the upload is marked failed and an error is returned.
curl --request POST \
--url https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads \
--header 'Content-Type: application/json' \
--header 'x-api-key: <api-key>' \
--data '
{
"data_source_config": {
"source": "S3",
"s3_bucket": "<string>",
"aws_region": "<string>",
"aws_account_id": "<string>",
"s3_prefix": ""
},
"data_source_auth_config": {
"source": "SharePoint",
"client_secret": "<string>",
"encrypted": false
},
"chunking_strategy_config": {
"strategy": "character",
"separator": "\n\n",
"chunk_size": 1000,
"chunk_overlap": 200
},
"force_reupload": false,
"tagging_information": {
"type": "per_file",
"tags_to_apply": {}
}
}
'import requests
url = "https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads"
payload = {
"data_source_config": {
"source": "S3",
"s3_bucket": "<string>",
"aws_region": "<string>",
"aws_account_id": "<string>",
"s3_prefix": ""
},
"data_source_auth_config": {
"source": "SharePoint",
"client_secret": "<string>",
"encrypted": False
},
"chunking_strategy_config": {
"strategy": "character",
"separator": "
",
"chunk_size": 1000,
"chunk_overlap": 200
},
"force_reupload": False,
"tagging_information": {
"type": "per_file",
"tags_to_apply": {}
}
}
headers = {
"x-api-key": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'x-api-key': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
data_source_config: {
source: 'S3',
s3_bucket: '<string>',
aws_region: '<string>',
aws_account_id: '<string>',
s3_prefix: ''
},
data_source_auth_config: {source: 'SharePoint', client_secret: '<string>', encrypted: false},
chunking_strategy_config: {strategy: 'character', separator: '\n\n', chunk_size: 1000, chunk_overlap: 200},
force_reupload: false,
tagging_information: {type: 'per_file', tags_to_apply: {}}
})
};
fetch('https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'data_source_config' => [
'source' => 'S3',
's3_bucket' => '<string>',
'aws_region' => '<string>',
'aws_account_id' => '<string>',
's3_prefix' => ''
],
'data_source_auth_config' => [
'source' => 'SharePoint',
'client_secret' => '<string>',
'encrypted' => false
],
'chunking_strategy_config' => [
'strategy' => 'character',
'separator' => '
',
'chunk_size' => 1000,
'chunk_overlap' => 200
],
'force_reupload' => false,
'tagging_information' => [
'type' => 'per_file',
'tags_to_apply' => [
]
]
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"x-api-key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads"
payload := strings.NewReader("{\n \"data_source_config\": {\n \"source\": \"S3\",\n \"s3_bucket\": \"<string>\",\n \"aws_region\": \"<string>\",\n \"aws_account_id\": \"<string>\",\n \"s3_prefix\": \"\"\n },\n \"data_source_auth_config\": {\n \"source\": \"SharePoint\",\n \"client_secret\": \"<string>\",\n \"encrypted\": false\n },\n \"chunking_strategy_config\": {\n \"strategy\": \"character\",\n \"separator\": \"\\n\\n\",\n \"chunk_size\": 1000,\n \"chunk_overlap\": 200\n },\n \"force_reupload\": false,\n \"tagging_information\": {\n \"type\": \"per_file\",\n \"tags_to_apply\": {}\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("x-api-key", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads")
.header("x-api-key", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"data_source_config\": {\n \"source\": \"S3\",\n \"s3_bucket\": \"<string>\",\n \"aws_region\": \"<string>\",\n \"aws_account_id\": \"<string>\",\n \"s3_prefix\": \"\"\n },\n \"data_source_auth_config\": {\n \"source\": \"SharePoint\",\n \"client_secret\": \"<string>\",\n \"encrypted\": false\n },\n \"chunking_strategy_config\": {\n \"strategy\": \"character\",\n \"separator\": \"\\n\\n\",\n \"chunk_size\": 1000,\n \"chunk_overlap\": 200\n },\n \"force_reupload\": false,\n \"tagging_information\": {\n \"type\": \"per_file\",\n \"tags_to_apply\": {}\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.egp.scale.com/v4/knowledge-bases/{knowledge_base_id}/uploads")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["x-api-key"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"data_source_config\": {\n \"source\": \"S3\",\n \"s3_bucket\": \"<string>\",\n \"aws_region\": \"<string>\",\n \"aws_account_id\": \"<string>\",\n \"s3_prefix\": \"\"\n },\n \"data_source_auth_config\": {\n \"source\": \"SharePoint\",\n \"client_secret\": \"<string>\",\n \"encrypted\": false\n },\n \"chunking_strategy_config\": {\n \"strategy\": \"character\",\n \"separator\": \"\\n\\n\",\n \"chunk_size\": 1000,\n \"chunk_overlap\": 200\n },\n \"force_reupload\": false,\n \"tagging_information\": {\n \"type\": \"per_file\",\n \"tags_to_apply\": {}\n }\n}"
response = http.request(request)
puts response.read_body{
"upload_id": "<string>"
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>",
"input": "<unknown>",
"ctx": {}
}
]
}Authorizations
Headers
Path Parameters
Body
- DataSource Upload Request
- Local Chunks Upload Request
- DataSource Upload Request
Configuration for the data source which describes where to find the data.
- S3 DataSource Config
- SharePoint DataSource Config
- SharePoint Page DataSource Config
- Google Drive DataSource Config
- Azure Blob Storage DataSource Config
- Google Cloud Storage DataSource Config
- Confluence DataSource Config
- Slack DataSource Config
- Snowflake DataSource Config
- Databricks DataSource Config
- SQL Database DataSource Config
- MongoDB DataSource Config
- BigQuery DataSource Config
Show child attributes
Show child attributes
Configuration for the data source which describes how to authenticate to the data source.
- SharePoint DataSource Auth Config
- SharePoint Page DataSource Auth Config
- Azure DataSource Auth Config
- Google Cloud Storage DataSource Auth Config
- Google Drive DataSource Auth Config
- S3 DataSource Auth Config
- Confluence DataSource Auth Config
- Slack DataSource Auth Config
- Snowflake DataSource Auth Config
- Databricks DataSource Auth Config
- SQL Database DataSource Auth Config
- MongoDB DataSource Auth Config
- BigQuery DataSource Auth Config
Show child attributes
Show child attributes
Configuration for the chunking strategy which describes how to chunk the data.
- CharacterChunkingStrategyConfig
- TokenChunkingStrategyConfig
- CustomChunkingStrategyConfig
- PreChunkedStrategyConfig
- EnhancedChunkingStrategyConfig
Show child attributes
Show child attributes
Force reingest, regardless the change of the source file.
A dictionary of tags to apply to all artifacts added from the data source.
- TaggingInformationPerFile
- TaggingInformationAll
Show child attributes
Show child attributes
Response
Successful Response
ID of the created knowledge base upload job.

