curl --request POST \
--url https://{host}/v2/workspaces/slug:storage/webhooks/v1/knowledge_bases/{knowledgeBaseId}/documents \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"source_type": "uploaded_file",
"native_file_id": "nf_8GkX2p9QwLmZ4rTv1bCdE",
"native_file_workspace_id": "wsp_acme",
"tags": {
"conversation_id": "conv_42"
}
}
'const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
source_type: 'uploaded_file',
native_file_id: 'nf_8GkX2p9QwLmZ4rTv1bCdE',
native_file_workspace_id: 'wsp_acme',
tags: {conversation_id: 'conv_42'}
})
};
fetch('https://{host}/v2/workspaces/slug:storage/webhooks/v1/knowledge_bases/{knowledgeBaseId}/documents', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));import requests
url = "https://{host}/v2/workspaces/slug:storage/webhooks/v1/knowledge_bases/{knowledgeBaseId}/documents"
payload = {
"source_type": "uploaded_file",
"native_file_id": "nf_8GkX2p9QwLmZ4rTv1bCdE",
"native_file_workspace_id": "wsp_acme",
"tags": { "conversation_id": "conv_42" }
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text){
"id": "vsf_550e8400-e29b-41d4-a716-446655440000",
"object": "knowledge_base.document",
"knowledge_base_id": "vs_7c9e6679-7425-40de-944b-e07fc1f90ae7",
"source_type": "connector_document",
"source_key": "web://https://contoso.sharepoint.com/sites/x/Shared%20Documents/report.pdf",
"source_url": "https://contoso.sharepoint.com/sites/x/Shared%20Documents/report.pdf",
"native_file_id": null,
"native_file_workspace_id": null,
"origin": "user",
"web_source_id": null,
"crawler_document_id": "crawl_doc_8842",
"status": "completed",
"index_generation": 3,
"filename": "report.pdf",
"mime_type": "application/pdf",
"bytes": 234567,
"parser": "tika",
"parsed_at": 1717000000,
"page_count": 12,
"char_count": 48213,
"chunking_strategy": null,
"scope": "knowledge",
"conversation_id": null,
"expires_at": null,
"tags": {
"project": "contracts-2026"
},
"metadata": {
"connector": "sharepoint",
"site_id": "site_123",
"drive_item_id": "item_456",
"version": "42"
},
"chunks_count": 37,
"vectors_count": 37,
"last_error": null,
"indexed_at": 1717000050,
"created_at": 1714972900,
"updated_at": 1717000050
}Attach a source (idempotent upsert)
The single ingestion endpoint. The server computes the
canonical source_key and upserts by
(knowledge_base_id, source_key):
- no existing row → insert, schedule indexing, 201;
- existing row → refresh mutable fields (
fetch_url,tags,metadata,filename,mime_type,bytes), schedule a fresh indexing run, 200.
For native_file_id sources no fetch URL is stored; storage
mints short-lived ones at fetch time.
Connector sync = repeat this call per document per cycle with
the same source_url and a fresh fetch_url.
Requires editor+ on the knowledge base.
curl --request POST \
--url https://{host}/v2/workspaces/slug:storage/webhooks/v1/knowledge_bases/{knowledgeBaseId}/documents \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"source_type": "uploaded_file",
"native_file_id": "nf_8GkX2p9QwLmZ4rTv1bCdE",
"native_file_workspace_id": "wsp_acme",
"tags": {
"conversation_id": "conv_42"
}
}
'const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
source_type: 'uploaded_file',
native_file_id: 'nf_8GkX2p9QwLmZ4rTv1bCdE',
native_file_workspace_id: 'wsp_acme',
tags: {conversation_id: 'conv_42'}
})
};
fetch('https://{host}/v2/workspaces/slug:storage/webhooks/v1/knowledge_bases/{knowledgeBaseId}/documents', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));import requests
url = "https://{host}/v2/workspaces/slug:storage/webhooks/v1/knowledge_bases/{knowledgeBaseId}/documents"
payload = {
"source_type": "uploaded_file",
"native_file_id": "nf_8GkX2p9QwLmZ4rTv1bCdE",
"native_file_workspace_id": "wsp_acme",
"tags": { "conversation_id": "conv_42" }
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text){
"id": "vsf_550e8400-e29b-41d4-a716-446655440000",
"object": "knowledge_base.document",
"knowledge_base_id": "vs_7c9e6679-7425-40de-944b-e07fc1f90ae7",
"source_type": "connector_document",
"source_key": "web://https://contoso.sharepoint.com/sites/x/Shared%20Documents/report.pdf",
"source_url": "https://contoso.sharepoint.com/sites/x/Shared%20Documents/report.pdf",
"native_file_id": null,
"native_file_workspace_id": null,
"origin": "user",
"web_source_id": null,
"crawler_document_id": "crawl_doc_8842",
"status": "completed",
"index_generation": 3,
"filename": "report.pdf",
"mime_type": "application/pdf",
"bytes": 234567,
"parser": "tika",
"parsed_at": 1717000000,
"page_count": 12,
"char_count": 48213,
"chunking_strategy": null,
"scope": "knowledge",
"conversation_id": null,
"expires_at": null,
"tags": {
"project": "contracts-2026"
},
"metadata": {
"connector": "sharepoint",
"site_id": "site_123",
"drive_item_id": "item_456",
"version": "42"
},
"chunks_count": 37,
"vectors_count": 37,
"last_error": null,
"indexed_at": 1717000050,
"created_at": 1714972900,
"updated_at": 1717000050
}Authorizations
User session JWT or instance API key (iak_*). Send as Authorization: Bearer <token>.
Path Parameters
Knowledge base id. Legacy physical prefix vs_ (the kb_ rename is deferred).
128^vs_[A-Za-z0-9-]+$Query Parameters
Agent context for hook resolution and agent-owned knowledge bases. May also be supplied in the request body.
128Body
Body of POST /documents. Exactly one source must be
identifiable: native_file_id, or source_url, or fetch_url
(in dedup-key priority order) - otherwise 400 MISSING_SOURCE.
source_type is optional: inferred uploaded_file when
native_file_id is present (a conflicting explicit value is a
400); defaults to web_page for a bare URL. Pass
remote_file explicitly for non-HTML document bytes at a URL,
and connector_document from connectors.
uploaded_file, remote_file, web_page, connector_document Platform-native file id. 404 FILE_NOT_FOUND if the
referenced native file cannot be resolved. Content access is
governed by the knowledge base's ACL. Storage resolves
filename, mime_type, bytes from the native API unless
supplied inline.
Workspace owning the native file. Defaults to the caller's workspace.
Stable provenance URL - the dedup key for non-native
sources. Defaults to fetch_url when omitted. For
uploaded files, an optional display override (e.g. a doc
viewer page URL).
Operational URL used to retrieve bytes at index time. May
be temporary, signed, tokenized. Write-only - never
returned by any response. Ignored for
source_type: uploaded_file (storage mints its own).
Refreshed on every upsert (200 path); connectors send a
fresh one on each sync.
255x >= 0Optional embedding model override threaded into the indexing pipeline for this document.
256Agent context for hook resolution and agent-owned knowledge
bases. May also be supplied as the agent_id query parameter.
128knowledge, conversation Unix seconds.
Optional external-hook pass-through (e.g. from
agent-factory). Merged with the server-resolved
platform/org/agent hook config. A before_file_ingest or
after_file_parse hook may transform or block the content;
a block yields 403 HOOK_REJECTED (sync) or a document in
status: rejected (async), with rejection_reason /
rejected_by set. Write-only; never returned.
Response
Existing document refreshed; indexing rescheduled.
Association between a source and a knowledge base - the only
membership model (no files registry). The server computes
source_key and dedups on (knowledge_base_id, source_key).
fetch_url is intentionally ABSENT from this schema: it is
write-only at the API boundary (may embed credentials). For
source_type: uploaded_file no fetch URL is stored at all -
storage mints a fresh short-lived one (as a privileged
workspace) whenever the crawler needs bytes, so expired-token
failures are structurally impossible for uploaded files.
128^vsf_[A-Za-z0-9-]+$knowledge_base.document ^vs_[A-Za-z0-9-]+$Discriminator. uploaded_file for platform-native files
(attached by native_file_id); remote_file for document
bytes at an arbitrary URL; web_page for HTML pages
(single URL or crawl-discovered); connector_document for
sources owned by a third-party connector.
uploaded_file, remote_file, web_page, connector_document Server-computed canonical source URI; the dedup key,
UNIQUE with knowledge_base_id.
prisme-file://{workspace_id}/{native_file_id} for
uploaded files (stable across renames - never the rebuilt
native URL); web://{normalized_url} otherwise.
Normalization is conservative (strip fragment, lowercase
scheme + host, normalize percent-encoding, strip default
ports - query params untouched) and shared with the
crawler's canonicalization.
2048User-facing provenance URL - citations, UI links, audit
logs. Stable by construction for connector documents
(e.g. SharePoint webUrl) and web pages (the page URL). For
uploaded files, a display URL (caller-overridable at
attach time); downloadable links are minted lazily via
GET /documents/{document_id}/source_url.
Provenance, set at row creation and never overwritten by
seed adoption. Seed deletion cascades only over
origin: web_source rows - a manually attached page later
discovered by a seed keeps origin: user and survives the
seed's deletion (losing only its web_source_id).
user, web_source Authoritative indexing status. Transitions are guarded by
index_generation - stale or duplicate callbacks are
ignored. fetch_expired is connector-facing: the stored
ephemeral fetch_url no longer works and the owning
connector must refresh it via the upsert (POST /documents with the same source_url); it is never a
user-actionable error and never occurs for
uploaded_file sources. rejected means an external hook
(before_file_ingest / after_file_parse) blocked the
content - see rejection_reason / rejected_by.
queued, in_progress, completed, failed, fetch_expired, rejected Platform-native file id when source_type: uploaded_file.
Workspace owning the native file. Defaults to the caller's workspace.
Primary producing/adopting seed (origin-of-record), when
any. Multi-attribution (a page reachable from several
seeds) is carried by parent_web_source_ids.
^seed_[A-Za-z0-9-]+$All seeds whose crawl scope covers this page
(multi-attribution, §F-5). Backs per-seed page counts and
seed cascade-delete: deleting a seed removes it from this
array and drops the row only when the array empties AND
origin: web_source. Empty/absent ⇒ unattributed (counted
in the /web_sources summary residual).
^seed_[A-Za-z0-9-]+$Crawler-side document id, set once the crawler has seen this source.
Monotonically increasing per document; incremented on every (re)index. Search reads only the latest committed generation; a reindex swaps the new generation in only on success, so the previous good vectors remain searchable on failure.
x >= 1255x >= 0Parser used at last successful parse.
x >= 0x >= 0Per-document override of the store default.
knowledge, conversation Unix seconds. TTL for conversation-scope documents.
Caller-supplied tags. Editable via PATCH. Filterable on list.
Connector-supplied structured context (e.g. connector,
site_id, drive_item_id, version). Editable via PATCH.
x >= 0x >= 0Most recent failure context. Cleared when a new indexing
attempt transitions to in_progress. code: SOURCE_UNAVAILABLE signals a dangling reference (e.g. the
native file was deleted out-of-band - native files emit no
lifecycle events, so this surfaces at the next fetch).
Human-readable narrative status, distinct from last_error
(e.g. a partial-success or skipped note).
Unix seconds of the most recent failure.
Set when status: rejected. Why an external hook blocked
the content. Surfaced to the user only when the resolved
hook config sets expose_reason: true.
Identifier of the hook that produced the rejection.
Unix seconds. Set when the document reaches completed.
Was this page helpful?