Update Document Version Metadata Handler
Merge metadata fields into an existing document version’s metadata.
Only non-null fields in the request body are merged; existing metadata fields not present in the request are preserved.
curl --request PATCH \
--url http://localhost:8000/v1/document_versions/{version_id}/metadata \
--header 'Content-Type: application/json' \
--data '
{
"source_s3": "<string>",
"cleaned_source_s3": "<string>",
"standard_pipeline_json_s3": "<string>",
"fast_plaintext_s3": "<string>",
"high_accuracy_content_list_s3": "<string>",
"high_accuracy_middle_s3": "<string>",
"hash": "<string>",
"pipeline_state": {
"last_run_timestamp": "2023-11-07T05:31:56Z",
"last_state_update_timestamp": "2023-11-07T05:31:56Z",
"last_activity": "<string>",
"error": "<string>",
"temporal_workflow_id": "<string>",
"chunks_processed": 123,
"page_dpi": 123
},
"total_pages": 123,
"total_sections": 123,
"total_chunks": 123,
"total_formulas": 123,
"xlsx_parse_result_s3": "<string>",
"xlsx_named_ranges": [
{}
],
"xlsx_kpi_catalog": [
{}
],
"information_statistics": {
"num_chunks_by_type": {},
"total_tokens": 0,
"num_direct_children": 0,
"children_depth": 0
}
}
'import requests
url = "http://localhost:8000/v1/document_versions/{version_id}/metadata"
payload = {
"source_s3": "<string>",
"cleaned_source_s3": "<string>",
"standard_pipeline_json_s3": "<string>",
"fast_plaintext_s3": "<string>",
"high_accuracy_content_list_s3": "<string>",
"high_accuracy_middle_s3": "<string>",
"hash": "<string>",
"pipeline_state": {
"last_run_timestamp": "2023-11-07T05:31:56Z",
"last_state_update_timestamp": "2023-11-07T05:31:56Z",
"last_activity": "<string>",
"error": "<string>",
"temporal_workflow_id": "<string>",
"chunks_processed": 123,
"page_dpi": 123
},
"total_pages": 123,
"total_sections": 123,
"total_chunks": 123,
"total_formulas": 123,
"xlsx_parse_result_s3": "<string>",
"xlsx_named_ranges": [{}],
"xlsx_kpi_catalog": [{}],
"information_statistics": {
"num_chunks_by_type": {},
"total_tokens": 0,
"num_direct_children": 0,
"children_depth": 0
}
}
headers = {"Content-Type": "application/json"}
response = requests.patch(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'PATCH',
headers: {'Content-Type': 'application/json'},
body: JSON.stringify({
source_s3: '<string>',
cleaned_source_s3: '<string>',
standard_pipeline_json_s3: '<string>',
fast_plaintext_s3: '<string>',
high_accuracy_content_list_s3: '<string>',
high_accuracy_middle_s3: '<string>',
hash: '<string>',
pipeline_state: {
last_run_timestamp: '2023-11-07T05:31:56Z',
last_state_update_timestamp: '2023-11-07T05:31:56Z',
last_activity: '<string>',
error: '<string>',
temporal_workflow_id: '<string>',
chunks_processed: 123,
page_dpi: 123
},
total_pages: 123,
total_sections: 123,
total_chunks: 123,
total_formulas: 123,
xlsx_parse_result_s3: '<string>',
xlsx_named_ranges: [{}],
xlsx_kpi_catalog: [{}],
information_statistics: {
num_chunks_by_type: {},
total_tokens: 0,
num_direct_children: 0,
children_depth: 0
}
})
};
fetch('http://localhost:8000/v1/document_versions/{version_id}/metadata', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_PORT => "8000",
CURLOPT_URL => "http://localhost:8000/v1/document_versions/{version_id}/metadata",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "PATCH",
CURLOPT_POSTFIELDS => json_encode([
'source_s3' => '<string>',
'cleaned_source_s3' => '<string>',
'standard_pipeline_json_s3' => '<string>',
'fast_plaintext_s3' => '<string>',
'high_accuracy_content_list_s3' => '<string>',
'high_accuracy_middle_s3' => '<string>',
'hash' => '<string>',
'pipeline_state' => [
'last_run_timestamp' => '2023-11-07T05:31:56Z',
'last_state_update_timestamp' => '2023-11-07T05:31:56Z',
'last_activity' => '<string>',
'error' => '<string>',
'temporal_workflow_id' => '<string>',
'chunks_processed' => 123,
'page_dpi' => 123
],
'total_pages' => 123,
'total_sections' => 123,
'total_chunks' => 123,
'total_formulas' => 123,
'xlsx_parse_result_s3' => '<string>',
'xlsx_named_ranges' => [
[
]
],
'xlsx_kpi_catalog' => [
[
]
],
'information_statistics' => [
'num_chunks_by_type' => [
],
'total_tokens' => 0,
'num_direct_children' => 0,
'children_depth' => 0
]
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "http://localhost:8000/v1/document_versions/{version_id}/metadata"
payload := strings.NewReader("{\n \"source_s3\": \"<string>\",\n \"cleaned_source_s3\": \"<string>\",\n \"standard_pipeline_json_s3\": \"<string>\",\n \"fast_plaintext_s3\": \"<string>\",\n \"high_accuracy_content_list_s3\": \"<string>\",\n \"high_accuracy_middle_s3\": \"<string>\",\n \"hash\": \"<string>\",\n \"pipeline_state\": {\n \"last_run_timestamp\": \"2023-11-07T05:31:56Z\",\n \"last_state_update_timestamp\": \"2023-11-07T05:31:56Z\",\n \"last_activity\": \"<string>\",\n \"error\": \"<string>\",\n \"temporal_workflow_id\": \"<string>\",\n \"chunks_processed\": 123,\n \"page_dpi\": 123\n },\n \"total_pages\": 123,\n \"total_sections\": 123,\n \"total_chunks\": 123,\n \"total_formulas\": 123,\n \"xlsx_parse_result_s3\": \"<string>\",\n \"xlsx_named_ranges\": [\n {}\n ],\n \"xlsx_kpi_catalog\": [\n {}\n ],\n \"information_statistics\": {\n \"num_chunks_by_type\": {},\n \"total_tokens\": 0,\n \"num_direct_children\": 0,\n \"children_depth\": 0\n }\n}")
req, _ := http.NewRequest("PATCH", url, payload)
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.patch("http://localhost:8000/v1/document_versions/{version_id}/metadata")
.header("Content-Type", "application/json")
.body("{\n \"source_s3\": \"<string>\",\n \"cleaned_source_s3\": \"<string>\",\n \"standard_pipeline_json_s3\": \"<string>\",\n \"fast_plaintext_s3\": \"<string>\",\n \"high_accuracy_content_list_s3\": \"<string>\",\n \"high_accuracy_middle_s3\": \"<string>\",\n \"hash\": \"<string>\",\n \"pipeline_state\": {\n \"last_run_timestamp\": \"2023-11-07T05:31:56Z\",\n \"last_state_update_timestamp\": \"2023-11-07T05:31:56Z\",\n \"last_activity\": \"<string>\",\n \"error\": \"<string>\",\n \"temporal_workflow_id\": \"<string>\",\n \"chunks_processed\": 123,\n \"page_dpi\": 123\n },\n \"total_pages\": 123,\n \"total_sections\": 123,\n \"total_chunks\": 123,\n \"total_formulas\": 123,\n \"xlsx_parse_result_s3\": \"<string>\",\n \"xlsx_named_ranges\": [\n {}\n ],\n \"xlsx_kpi_catalog\": [\n {}\n ],\n \"information_statistics\": {\n \"num_chunks_by_type\": {},\n \"total_tokens\": 0,\n \"num_direct_children\": 0,\n \"children_depth\": 0\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("http://localhost:8000/v1/document_versions/{version_id}/metadata")
http = Net::HTTP.new(url.host, url.port)
request = Net::HTTP::Patch.new(url)
request["Content-Type"] = 'application/json'
request.body = "{\n \"source_s3\": \"<string>\",\n \"cleaned_source_s3\": \"<string>\",\n \"standard_pipeline_json_s3\": \"<string>\",\n \"fast_plaintext_s3\": \"<string>\",\n \"high_accuracy_content_list_s3\": \"<string>\",\n \"high_accuracy_middle_s3\": \"<string>\",\n \"hash\": \"<string>\",\n \"pipeline_state\": {\n \"last_run_timestamp\": \"2023-11-07T05:31:56Z\",\n \"last_state_update_timestamp\": \"2023-11-07T05:31:56Z\",\n \"last_activity\": \"<string>\",\n \"error\": \"<string>\",\n \"temporal_workflow_id\": \"<string>\",\n \"chunks_processed\": 123,\n \"page_dpi\": 123\n },\n \"total_pages\": 123,\n \"total_sections\": 123,\n \"total_chunks\": 123,\n \"total_formulas\": 123,\n \"xlsx_parse_result_s3\": \"<string>\",\n \"xlsx_named_ranges\": [\n {}\n ],\n \"xlsx_kpi_catalog\": [\n {}\n ],\n \"information_statistics\": {\n \"num_chunks_by_type\": {},\n \"total_tokens\": 0,\n \"num_direct_children\": 0,\n \"children_depth\": 0\n }\n}"
response = http.request(request)
puts response.read_body{
"id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"path_part_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"version": 123,
"name": "<string>",
"parent_path_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"materialized_path": "<string>",
"system_managed": true,
"tenant_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"created_at": "2023-11-07T05:31:56Z",
"updated_at": "2023-11-07T05:31:56Z",
"asset_s3_url": "<string>",
"fast_plaintext_url": "<string>",
"system_metadata": {
"source_s3": "<string>",
"cleaned_source_s3": "<string>",
"fast_plaintext_s3": "<string>",
"hash": "<string>",
"pipeline_state": {
"last_run_timestamp": "2023-11-07T05:31:56Z",
"last_state_update_timestamp": "2023-11-07T05:31:56Z",
"last_activity": "<string>",
"error": "<string>",
"temporal_workflow_id": "<string>",
"chunks_processed": 123,
"page_dpi": 123
},
"total_pages": 123,
"total_sections": 123,
"total_chunks": 123,
"total_formulas": 123,
"xlsx_parse_result_s3": "<string>",
"xlsx_named_ranges": [
{}
],
"xlsx_kpi_catalog": [
{}
],
"information_statistics": {
"num_chunks_by_type": {},
"total_tokens": 0,
"num_direct_children": 0,
"children_depth": 0
}
}
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>",
"input": "<unknown>",
"ctx": {}
}
]
}Headers
Path Parameters
DocumentVersion ID
Cookies
Body
Partial update schema for document version metadata.
All fields are optional. Only non-None fields are merged into
the existing metadata dict.
Pipeline execution state tracking.
Show child attributes
Show child attributes
Aggregate statistics for a section subtree or document version.
Show child attributes
Show child attributes
Response
Successful Response
DocumentVersion response model.
Shared schema for DocumentVersion responses, used by Document endpoints and DocumentVersion endpoints.
DocumentVersion ID
PathPart ID
Version number (0, 1, 2...)
Auto-generated name from path_part (v0, v1, ...)
Document's PathPart ID
Full materialized path from root
Whether this version is system-managed
Tenant ID
Creation timestamp
Last update timestamp
Presigned URL to download the source document (6-hour validity)
Presigned URL to download the fast plaintext export (6-hour validity)
Version metadata (S3 artifacts, pipeline state, statistics)
Show child attributes
Show child attributes
curl --request PATCH \
--url http://localhost:8000/v1/document_versions/{version_id}/metadata \
--header 'Content-Type: application/json' \
--data '
{
"source_s3": "<string>",
"cleaned_source_s3": "<string>",
"standard_pipeline_json_s3": "<string>",
"fast_plaintext_s3": "<string>",
"high_accuracy_content_list_s3": "<string>",
"high_accuracy_middle_s3": "<string>",
"hash": "<string>",
"pipeline_state": {
"last_run_timestamp": "2023-11-07T05:31:56Z",
"last_state_update_timestamp": "2023-11-07T05:31:56Z",
"last_activity": "<string>",
"error": "<string>",
"temporal_workflow_id": "<string>",
"chunks_processed": 123,
"page_dpi": 123
},
"total_pages": 123,
"total_sections": 123,
"total_chunks": 123,
"total_formulas": 123,
"xlsx_parse_result_s3": "<string>",
"xlsx_named_ranges": [
{}
],
"xlsx_kpi_catalog": [
{}
],
"information_statistics": {
"num_chunks_by_type": {},
"total_tokens": 0,
"num_direct_children": 0,
"children_depth": 0
}
}
'import requests
url = "http://localhost:8000/v1/document_versions/{version_id}/metadata"
payload = {
"source_s3": "<string>",
"cleaned_source_s3": "<string>",
"standard_pipeline_json_s3": "<string>",
"fast_plaintext_s3": "<string>",
"high_accuracy_content_list_s3": "<string>",
"high_accuracy_middle_s3": "<string>",
"hash": "<string>",
"pipeline_state": {
"last_run_timestamp": "2023-11-07T05:31:56Z",
"last_state_update_timestamp": "2023-11-07T05:31:56Z",
"last_activity": "<string>",
"error": "<string>",
"temporal_workflow_id": "<string>",
"chunks_processed": 123,
"page_dpi": 123
},
"total_pages": 123,
"total_sections": 123,
"total_chunks": 123,
"total_formulas": 123,
"xlsx_parse_result_s3": "<string>",
"xlsx_named_ranges": [{}],
"xlsx_kpi_catalog": [{}],
"information_statistics": {
"num_chunks_by_type": {},
"total_tokens": 0,
"num_direct_children": 0,
"children_depth": 0
}
}
headers = {"Content-Type": "application/json"}
response = requests.patch(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'PATCH',
headers: {'Content-Type': 'application/json'},
body: JSON.stringify({
source_s3: '<string>',
cleaned_source_s3: '<string>',
standard_pipeline_json_s3: '<string>',
fast_plaintext_s3: '<string>',
high_accuracy_content_list_s3: '<string>',
high_accuracy_middle_s3: '<string>',
hash: '<string>',
pipeline_state: {
last_run_timestamp: '2023-11-07T05:31:56Z',
last_state_update_timestamp: '2023-11-07T05:31:56Z',
last_activity: '<string>',
error: '<string>',
temporal_workflow_id: '<string>',
chunks_processed: 123,
page_dpi: 123
},
total_pages: 123,
total_sections: 123,
total_chunks: 123,
total_formulas: 123,
xlsx_parse_result_s3: '<string>',
xlsx_named_ranges: [{}],
xlsx_kpi_catalog: [{}],
information_statistics: {
num_chunks_by_type: {},
total_tokens: 0,
num_direct_children: 0,
children_depth: 0
}
})
};
fetch('http://localhost:8000/v1/document_versions/{version_id}/metadata', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_PORT => "8000",
CURLOPT_URL => "http://localhost:8000/v1/document_versions/{version_id}/metadata",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "PATCH",
CURLOPT_POSTFIELDS => json_encode([
'source_s3' => '<string>',
'cleaned_source_s3' => '<string>',
'standard_pipeline_json_s3' => '<string>',
'fast_plaintext_s3' => '<string>',
'high_accuracy_content_list_s3' => '<string>',
'high_accuracy_middle_s3' => '<string>',
'hash' => '<string>',
'pipeline_state' => [
'last_run_timestamp' => '2023-11-07T05:31:56Z',
'last_state_update_timestamp' => '2023-11-07T05:31:56Z',
'last_activity' => '<string>',
'error' => '<string>',
'temporal_workflow_id' => '<string>',
'chunks_processed' => 123,
'page_dpi' => 123
],
'total_pages' => 123,
'total_sections' => 123,
'total_chunks' => 123,
'total_formulas' => 123,
'xlsx_parse_result_s3' => '<string>',
'xlsx_named_ranges' => [
[
]
],
'xlsx_kpi_catalog' => [
[
]
],
'information_statistics' => [
'num_chunks_by_type' => [
],
'total_tokens' => 0,
'num_direct_children' => 0,
'children_depth' => 0
]
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "http://localhost:8000/v1/document_versions/{version_id}/metadata"
payload := strings.NewReader("{\n \"source_s3\": \"<string>\",\n \"cleaned_source_s3\": \"<string>\",\n \"standard_pipeline_json_s3\": \"<string>\",\n \"fast_plaintext_s3\": \"<string>\",\n \"high_accuracy_content_list_s3\": \"<string>\",\n \"high_accuracy_middle_s3\": \"<string>\",\n \"hash\": \"<string>\",\n \"pipeline_state\": {\n \"last_run_timestamp\": \"2023-11-07T05:31:56Z\",\n \"last_state_update_timestamp\": \"2023-11-07T05:31:56Z\",\n \"last_activity\": \"<string>\",\n \"error\": \"<string>\",\n \"temporal_workflow_id\": \"<string>\",\n \"chunks_processed\": 123,\n \"page_dpi\": 123\n },\n \"total_pages\": 123,\n \"total_sections\": 123,\n \"total_chunks\": 123,\n \"total_formulas\": 123,\n \"xlsx_parse_result_s3\": \"<string>\",\n \"xlsx_named_ranges\": [\n {}\n ],\n \"xlsx_kpi_catalog\": [\n {}\n ],\n \"information_statistics\": {\n \"num_chunks_by_type\": {},\n \"total_tokens\": 0,\n \"num_direct_children\": 0,\n \"children_depth\": 0\n }\n}")
req, _ := http.NewRequest("PATCH", url, payload)
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.patch("http://localhost:8000/v1/document_versions/{version_id}/metadata")
.header("Content-Type", "application/json")
.body("{\n \"source_s3\": \"<string>\",\n \"cleaned_source_s3\": \"<string>\",\n \"standard_pipeline_json_s3\": \"<string>\",\n \"fast_plaintext_s3\": \"<string>\",\n \"high_accuracy_content_list_s3\": \"<string>\",\n \"high_accuracy_middle_s3\": \"<string>\",\n \"hash\": \"<string>\",\n \"pipeline_state\": {\n \"last_run_timestamp\": \"2023-11-07T05:31:56Z\",\n \"last_state_update_timestamp\": \"2023-11-07T05:31:56Z\",\n \"last_activity\": \"<string>\",\n \"error\": \"<string>\",\n \"temporal_workflow_id\": \"<string>\",\n \"chunks_processed\": 123,\n \"page_dpi\": 123\n },\n \"total_pages\": 123,\n \"total_sections\": 123,\n \"total_chunks\": 123,\n \"total_formulas\": 123,\n \"xlsx_parse_result_s3\": \"<string>\",\n \"xlsx_named_ranges\": [\n {}\n ],\n \"xlsx_kpi_catalog\": [\n {}\n ],\n \"information_statistics\": {\n \"num_chunks_by_type\": {},\n \"total_tokens\": 0,\n \"num_direct_children\": 0,\n \"children_depth\": 0\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("http://localhost:8000/v1/document_versions/{version_id}/metadata")
http = Net::HTTP.new(url.host, url.port)
request = Net::HTTP::Patch.new(url)
request["Content-Type"] = 'application/json'
request.body = "{\n \"source_s3\": \"<string>\",\n \"cleaned_source_s3\": \"<string>\",\n \"standard_pipeline_json_s3\": \"<string>\",\n \"fast_plaintext_s3\": \"<string>\",\n \"high_accuracy_content_list_s3\": \"<string>\",\n \"high_accuracy_middle_s3\": \"<string>\",\n \"hash\": \"<string>\",\n \"pipeline_state\": {\n \"last_run_timestamp\": \"2023-11-07T05:31:56Z\",\n \"last_state_update_timestamp\": \"2023-11-07T05:31:56Z\",\n \"last_activity\": \"<string>\",\n \"error\": \"<string>\",\n \"temporal_workflow_id\": \"<string>\",\n \"chunks_processed\": 123,\n \"page_dpi\": 123\n },\n \"total_pages\": 123,\n \"total_sections\": 123,\n \"total_chunks\": 123,\n \"total_formulas\": 123,\n \"xlsx_parse_result_s3\": \"<string>\",\n \"xlsx_named_ranges\": [\n {}\n ],\n \"xlsx_kpi_catalog\": [\n {}\n ],\n \"information_statistics\": {\n \"num_chunks_by_type\": {},\n \"total_tokens\": 0,\n \"num_direct_children\": 0,\n \"children_depth\": 0\n }\n}"
response = http.request(request)
puts response.read_body{
"id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"path_part_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"version": 123,
"name": "<string>",
"parent_path_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"materialized_path": "<string>",
"system_managed": true,
"tenant_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"created_at": "2023-11-07T05:31:56Z",
"updated_at": "2023-11-07T05:31:56Z",
"asset_s3_url": "<string>",
"fast_plaintext_url": "<string>",
"system_metadata": {
"source_s3": "<string>",
"cleaned_source_s3": "<string>",
"fast_plaintext_s3": "<string>",
"hash": "<string>",
"pipeline_state": {
"last_run_timestamp": "2023-11-07T05:31:56Z",
"last_state_update_timestamp": "2023-11-07T05:31:56Z",
"last_activity": "<string>",
"error": "<string>",
"temporal_workflow_id": "<string>",
"chunks_processed": 123,
"page_dpi": 123
},
"total_pages": 123,
"total_sections": 123,
"total_chunks": 123,
"total_formulas": 123,
"xlsx_parse_result_s3": "<string>",
"xlsx_named_ranges": [
{}
],
"xlsx_kpi_catalog": [
{}
],
"information_statistics": {
"num_chunks_by_type": {},
"total_tokens": 0,
"num_direct_children": 0,
"children_depth": 0
}
}
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>",
"input": "<unknown>",
"ctx": {}
}
]
}