curl --request POST \
--url https://api.mixpeek.com/v1/collections/{collection_identifier}/export \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--header 'X-Namespace: <api-key>' \
--data '
{
"format": "parquet",
"include_vectors": false,
"include_lineage": false,
"select_fields": [
"document_id",
"metadata.title",
"metadata.category"
],
"filters": {
"AND": [
{
"field": "name",
"operator": "eq",
"value": "John"
},
{
"field": "age",
"operator": "gte",
"value": 30
}
],
"OR": [
{
"field": "status",
"operator": "eq",
"value": "active"
},
{
"field": "role",
"operator": "eq",
"value": "admin"
}
],
"NOT": [
{
"field": "department",
"operator": "eq",
"value": "HR"
},
{
"field": "location",
"operator": "eq",
"value": "remote"
}
],
"case_sensitive": true
},
"sample_size": 500000,
"include_media": false,
"samples_per_shard": 1000,
"destination": {
"connection_id": "<string>",
"prefix": ""
}
}
'import requests
url = "https://api.mixpeek.com/v1/collections/{collection_identifier}/export"
payload = {
"format": "parquet",
"include_vectors": False,
"include_lineage": False,
"select_fields": ["document_id", "metadata.title", "metadata.category"],
"filters": {
"AND": [
{
"field": "name",
"operator": "eq",
"value": "John"
},
{
"field": "age",
"operator": "gte",
"value": 30
}
],
"OR": [
{
"field": "status",
"operator": "eq",
"value": "active"
},
{
"field": "role",
"operator": "eq",
"value": "admin"
}
],
"NOT": [
{
"field": "department",
"operator": "eq",
"value": "HR"
},
{
"field": "location",
"operator": "eq",
"value": "remote"
}
],
"case_sensitive": True
},
"sample_size": 500000,
"include_media": False,
"samples_per_shard": 1000,
"destination": {
"connection_id": "<string>",
"prefix": ""
}
}
headers = {
"Authorization": "Bearer <token>",
"X-Namespace": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {
Authorization: 'Bearer <token>',
'X-Namespace': '<api-key>',
'Content-Type': 'application/json'
},
body: JSON.stringify({
format: 'parquet',
include_vectors: false,
include_lineage: false,
select_fields: ['document_id', 'metadata.title', 'metadata.category'],
filters: {
AND: [
{field: 'name', operator: 'eq', value: 'John'},
{field: 'age', operator: 'gte', value: 30}
],
OR: [
{field: 'status', operator: 'eq', value: 'active'},
{field: 'role', operator: 'eq', value: 'admin'}
],
NOT: [
{field: 'department', operator: 'eq', value: 'HR'},
{field: 'location', operator: 'eq', value: 'remote'}
],
case_sensitive: true
},
sample_size: 500000,
include_media: false,
samples_per_shard: 1000,
destination: {connection_id: '<string>', prefix: ''}
})
};
fetch('https://api.mixpeek.com/v1/collections/{collection_identifier}/export', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.mixpeek.com/v1/collections/{collection_identifier}/export",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'format' => 'parquet',
'include_vectors' => false,
'include_lineage' => false,
'select_fields' => [
'document_id',
'metadata.title',
'metadata.category'
],
'filters' => [
'AND' => [
[
'field' => 'name',
'operator' => 'eq',
'value' => 'John'
],
[
'field' => 'age',
'operator' => 'gte',
'value' => 30
]
],
'OR' => [
[
'field' => 'status',
'operator' => 'eq',
'value' => 'active'
],
[
'field' => 'role',
'operator' => 'eq',
'value' => 'admin'
]
],
'NOT' => [
[
'field' => 'department',
'operator' => 'eq',
'value' => 'HR'
],
[
'field' => 'location',
'operator' => 'eq',
'value' => 'remote'
]
],
'case_sensitive' => true
],
'sample_size' => 500000,
'include_media' => false,
'samples_per_shard' => 1000,
'destination' => [
'connection_id' => '<string>',
'prefix' => ''
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json",
"X-Namespace: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.mixpeek.com/v1/collections/{collection_identifier}/export"
payload := strings.NewReader("{\n \"format\": \"parquet\",\n \"include_vectors\": false,\n \"include_lineage\": false,\n \"select_fields\": [\n \"document_id\",\n \"metadata.title\",\n \"metadata.category\"\n ],\n \"filters\": {\n \"AND\": [\n {\n \"field\": \"name\",\n \"operator\": \"eq\",\n \"value\": \"John\"\n },\n {\n \"field\": \"age\",\n \"operator\": \"gte\",\n \"value\": 30\n }\n ],\n \"OR\": [\n {\n \"field\": \"status\",\n \"operator\": \"eq\",\n \"value\": \"active\"\n },\n {\n \"field\": \"role\",\n \"operator\": \"eq\",\n \"value\": \"admin\"\n }\n ],\n \"NOT\": [\n {\n \"field\": \"department\",\n \"operator\": \"eq\",\n \"value\": \"HR\"\n },\n {\n \"field\": \"location\",\n \"operator\": \"eq\",\n \"value\": \"remote\"\n }\n ],\n \"case_sensitive\": true\n },\n \"sample_size\": 500000,\n \"include_media\": false,\n \"samples_per_shard\": 1000,\n \"destination\": {\n \"connection_id\": \"<string>\",\n \"prefix\": \"\"\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("X-Namespace", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.mixpeek.com/v1/collections/{collection_identifier}/export")
.header("Authorization", "Bearer <token>")
.header("X-Namespace", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"format\": \"parquet\",\n \"include_vectors\": false,\n \"include_lineage\": false,\n \"select_fields\": [\n \"document_id\",\n \"metadata.title\",\n \"metadata.category\"\n ],\n \"filters\": {\n \"AND\": [\n {\n \"field\": \"name\",\n \"operator\": \"eq\",\n \"value\": \"John\"\n },\n {\n \"field\": \"age\",\n \"operator\": \"gte\",\n \"value\": 30\n }\n ],\n \"OR\": [\n {\n \"field\": \"status\",\n \"operator\": \"eq\",\n \"value\": \"active\"\n },\n {\n \"field\": \"role\",\n \"operator\": \"eq\",\n \"value\": \"admin\"\n }\n ],\n \"NOT\": [\n {\n \"field\": \"department\",\n \"operator\": \"eq\",\n \"value\": \"HR\"\n },\n {\n \"field\": \"location\",\n \"operator\": \"eq\",\n \"value\": \"remote\"\n }\n ],\n \"case_sensitive\": true\n },\n \"sample_size\": 500000,\n \"include_media\": false,\n \"samples_per_shard\": 1000,\n \"destination\": {\n \"connection_id\": \"<string>\",\n \"prefix\": \"\"\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.mixpeek.com/v1/collections/{collection_identifier}/export")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["X-Namespace"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"format\": \"parquet\",\n \"include_vectors\": false,\n \"include_lineage\": false,\n \"select_fields\": [\n \"document_id\",\n \"metadata.title\",\n \"metadata.category\"\n ],\n \"filters\": {\n \"AND\": [\n {\n \"field\": \"name\",\n \"operator\": \"eq\",\n \"value\": \"John\"\n },\n {\n \"field\": \"age\",\n \"operator\": \"gte\",\n \"value\": 30\n }\n ],\n \"OR\": [\n {\n \"field\": \"status\",\n \"operator\": \"eq\",\n \"value\": \"active\"\n },\n {\n \"field\": \"role\",\n \"operator\": \"eq\",\n \"value\": \"admin\"\n }\n ],\n \"NOT\": [\n {\n \"field\": \"department\",\n \"operator\": \"eq\",\n \"value\": \"HR\"\n },\n {\n \"field\": \"location\",\n \"operator\": \"eq\",\n \"value\": \"remote\"\n }\n ],\n \"case_sensitive\": true\n },\n \"sample_size\": 500000,\n \"include_media\": false,\n \"samples_per_shard\": 1000,\n \"destination\": {\n \"connection_id\": \"<string>\",\n \"prefix\": \"\"\n }\n}"
response = http.request(request)
puts response.read_body{
"document_count": 10000,
"download_url": "https://s3.amazonaws.com/bucket/org_xxx/ns_xxx/api_collections_export/col_xxx/v1/documents.parquet?...",
"exported_at": "2024-01-15T10:30:00Z",
"file_size_bytes": 5242880,
"format": "parquet",
"s3_path": "s3://bucket/org_xxx/ns_xxx/api_collections_export/col_xxx/v1/documents.parquet"
}{
"error": {
"details": {
"id": "ns_123",
"resource": "namespace"
},
"message": "Namespace not found",
"type": "NotFoundError"
},
"status": 404,
"success": false
}{
"error": {
"details": {
"id": "ns_123",
"resource": "namespace"
},
"message": "Namespace not found",
"type": "NotFoundError"
},
"status": 404,
"success": false
}{
"error": {
"details": {
"id": "ns_123",
"resource": "namespace"
},
"message": "Namespace not found",
"type": "NotFoundError"
},
"status": 404,
"success": false
}{
"error": {
"details": {
"id": "ns_123",
"resource": "namespace"
},
"message": "Namespace not found",
"type": "NotFoundError"
},
"status": 404,
"success": false
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>",
"input": "<unknown>",
"ctx": {}
}
]
}{
"error": {
"details": {
"id": "ns_123",
"resource": "namespace"
},
"message": "Namespace not found",
"type": "NotFoundError"
},
"status": 404,
"success": false
}Export Collection
Export collection documents to JSON, CSV, Parquet or WebDataset format.
Export Formats:
- JSON: Line-delimited JSON (JSONL) format. Good for streaming.
- CSV: Comma-separated values. Best for spreadsheets.
- PARQUET: Columnar format (default). Best for data pipelines.
- WEBDATASET: Tar shards for training loaders. One sample per document,
carrying
<key>.jsonand, withinclude_media=True, the source object’s bytes. The response returns every shard plus the brace pattern loaders take.
Vector Export:
Vectors are large and exported separately. When include_vectors=True,
a separate file is created for vectors with document_id mapping.
Field Selection:
Use select_fields to export only specific fields, reducing file size.
Lineage:
Set include_lineage=True to add lineage_chain, source_content_hash and
document_created_at to every row, so each exported document traces back to its
source object and processing steps.
Filtering: Apply filters to export a subset of documents.
Destination (your own storage):
Set destination to a storage connection you own plus a prefix, and the files and a
manifest.json are written there as well as to the download copy. The connection
needs write_enabled set, which is off by default, and its credentials need the
write permission on your side (s3:PutObject, or storage.objects.create for GCS).
A connection without it is refused with the permission named.
Response: Returns presigned download URLs valid for 1 hour.
Limits:
- Large exports may take time. Consider using
sample_sizefor testing. - Vector exports significantly increase processing time.
curl --request POST \
--url https://api.mixpeek.com/v1/collections/{collection_identifier}/export \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--header 'X-Namespace: <api-key>' \
--data '
{
"format": "parquet",
"include_vectors": false,
"include_lineage": false,
"select_fields": [
"document_id",
"metadata.title",
"metadata.category"
],
"filters": {
"AND": [
{
"field": "name",
"operator": "eq",
"value": "John"
},
{
"field": "age",
"operator": "gte",
"value": 30
}
],
"OR": [
{
"field": "status",
"operator": "eq",
"value": "active"
},
{
"field": "role",
"operator": "eq",
"value": "admin"
}
],
"NOT": [
{
"field": "department",
"operator": "eq",
"value": "HR"
},
{
"field": "location",
"operator": "eq",
"value": "remote"
}
],
"case_sensitive": true
},
"sample_size": 500000,
"include_media": false,
"samples_per_shard": 1000,
"destination": {
"connection_id": "<string>",
"prefix": ""
}
}
'import requests
url = "https://api.mixpeek.com/v1/collections/{collection_identifier}/export"
payload = {
"format": "parquet",
"include_vectors": False,
"include_lineage": False,
"select_fields": ["document_id", "metadata.title", "metadata.category"],
"filters": {
"AND": [
{
"field": "name",
"operator": "eq",
"value": "John"
},
{
"field": "age",
"operator": "gte",
"value": 30
}
],
"OR": [
{
"field": "status",
"operator": "eq",
"value": "active"
},
{
"field": "role",
"operator": "eq",
"value": "admin"
}
],
"NOT": [
{
"field": "department",
"operator": "eq",
"value": "HR"
},
{
"field": "location",
"operator": "eq",
"value": "remote"
}
],
"case_sensitive": True
},
"sample_size": 500000,
"include_media": False,
"samples_per_shard": 1000,
"destination": {
"connection_id": "<string>",
"prefix": ""
}
}
headers = {
"Authorization": "Bearer <token>",
"X-Namespace": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {
Authorization: 'Bearer <token>',
'X-Namespace': '<api-key>',
'Content-Type': 'application/json'
},
body: JSON.stringify({
format: 'parquet',
include_vectors: false,
include_lineage: false,
select_fields: ['document_id', 'metadata.title', 'metadata.category'],
filters: {
AND: [
{field: 'name', operator: 'eq', value: 'John'},
{field: 'age', operator: 'gte', value: 30}
],
OR: [
{field: 'status', operator: 'eq', value: 'active'},
{field: 'role', operator: 'eq', value: 'admin'}
],
NOT: [
{field: 'department', operator: 'eq', value: 'HR'},
{field: 'location', operator: 'eq', value: 'remote'}
],
case_sensitive: true
},
sample_size: 500000,
include_media: false,
samples_per_shard: 1000,
destination: {connection_id: '<string>', prefix: ''}
})
};
fetch('https://api.mixpeek.com/v1/collections/{collection_identifier}/export', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.mixpeek.com/v1/collections/{collection_identifier}/export",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'format' => 'parquet',
'include_vectors' => false,
'include_lineage' => false,
'select_fields' => [
'document_id',
'metadata.title',
'metadata.category'
],
'filters' => [
'AND' => [
[
'field' => 'name',
'operator' => 'eq',
'value' => 'John'
],
[
'field' => 'age',
'operator' => 'gte',
'value' => 30
]
],
'OR' => [
[
'field' => 'status',
'operator' => 'eq',
'value' => 'active'
],
[
'field' => 'role',
'operator' => 'eq',
'value' => 'admin'
]
],
'NOT' => [
[
'field' => 'department',
'operator' => 'eq',
'value' => 'HR'
],
[
'field' => 'location',
'operator' => 'eq',
'value' => 'remote'
]
],
'case_sensitive' => true
],
'sample_size' => 500000,
'include_media' => false,
'samples_per_shard' => 1000,
'destination' => [
'connection_id' => '<string>',
'prefix' => ''
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json",
"X-Namespace: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.mixpeek.com/v1/collections/{collection_identifier}/export"
payload := strings.NewReader("{\n \"format\": \"parquet\",\n \"include_vectors\": false,\n \"include_lineage\": false,\n \"select_fields\": [\n \"document_id\",\n \"metadata.title\",\n \"metadata.category\"\n ],\n \"filters\": {\n \"AND\": [\n {\n \"field\": \"name\",\n \"operator\": \"eq\",\n \"value\": \"John\"\n },\n {\n \"field\": \"age\",\n \"operator\": \"gte\",\n \"value\": 30\n }\n ],\n \"OR\": [\n {\n \"field\": \"status\",\n \"operator\": \"eq\",\n \"value\": \"active\"\n },\n {\n \"field\": \"role\",\n \"operator\": \"eq\",\n \"value\": \"admin\"\n }\n ],\n \"NOT\": [\n {\n \"field\": \"department\",\n \"operator\": \"eq\",\n \"value\": \"HR\"\n },\n {\n \"field\": \"location\",\n \"operator\": \"eq\",\n \"value\": \"remote\"\n }\n ],\n \"case_sensitive\": true\n },\n \"sample_size\": 500000,\n \"include_media\": false,\n \"samples_per_shard\": 1000,\n \"destination\": {\n \"connection_id\": \"<string>\",\n \"prefix\": \"\"\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("X-Namespace", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.mixpeek.com/v1/collections/{collection_identifier}/export")
.header("Authorization", "Bearer <token>")
.header("X-Namespace", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"format\": \"parquet\",\n \"include_vectors\": false,\n \"include_lineage\": false,\n \"select_fields\": [\n \"document_id\",\n \"metadata.title\",\n \"metadata.category\"\n ],\n \"filters\": {\n \"AND\": [\n {\n \"field\": \"name\",\n \"operator\": \"eq\",\n \"value\": \"John\"\n },\n {\n \"field\": \"age\",\n \"operator\": \"gte\",\n \"value\": 30\n }\n ],\n \"OR\": [\n {\n \"field\": \"status\",\n \"operator\": \"eq\",\n \"value\": \"active\"\n },\n {\n \"field\": \"role\",\n \"operator\": \"eq\",\n \"value\": \"admin\"\n }\n ],\n \"NOT\": [\n {\n \"field\": \"department\",\n \"operator\": \"eq\",\n \"value\": \"HR\"\n },\n {\n \"field\": \"location\",\n \"operator\": \"eq\",\n \"value\": \"remote\"\n }\n ],\n \"case_sensitive\": true\n },\n \"sample_size\": 500000,\n \"include_media\": false,\n \"samples_per_shard\": 1000,\n \"destination\": {\n \"connection_id\": \"<string>\",\n \"prefix\": \"\"\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.mixpeek.com/v1/collections/{collection_identifier}/export")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["X-Namespace"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"format\": \"parquet\",\n \"include_vectors\": false,\n \"include_lineage\": false,\n \"select_fields\": [\n \"document_id\",\n \"metadata.title\",\n \"metadata.category\"\n ],\n \"filters\": {\n \"AND\": [\n {\n \"field\": \"name\",\n \"operator\": \"eq\",\n \"value\": \"John\"\n },\n {\n \"field\": \"age\",\n \"operator\": \"gte\",\n \"value\": 30\n }\n ],\n \"OR\": [\n {\n \"field\": \"status\",\n \"operator\": \"eq\",\n \"value\": \"active\"\n },\n {\n \"field\": \"role\",\n \"operator\": \"eq\",\n \"value\": \"admin\"\n }\n ],\n \"NOT\": [\n {\n \"field\": \"department\",\n \"operator\": \"eq\",\n \"value\": \"HR\"\n },\n {\n \"field\": \"location\",\n \"operator\": \"eq\",\n \"value\": \"remote\"\n }\n ],\n \"case_sensitive\": true\n },\n \"sample_size\": 500000,\n \"include_media\": false,\n \"samples_per_shard\": 1000,\n \"destination\": {\n \"connection_id\": \"<string>\",\n \"prefix\": \"\"\n }\n}"
response = http.request(request)
puts response.read_body{
"document_count": 10000,
"download_url": "https://s3.amazonaws.com/bucket/org_xxx/ns_xxx/api_collections_export/col_xxx/v1/documents.parquet?...",
"exported_at": "2024-01-15T10:30:00Z",
"file_size_bytes": 5242880,
"format": "parquet",
"s3_path": "s3://bucket/org_xxx/ns_xxx/api_collections_export/col_xxx/v1/documents.parquet"
}{
"error": {
"details": {
"id": "ns_123",
"resource": "namespace"
},
"message": "Namespace not found",
"type": "NotFoundError"
},
"status": 404,
"success": false
}{
"error": {
"details": {
"id": "ns_123",
"resource": "namespace"
},
"message": "Namespace not found",
"type": "NotFoundError"
},
"status": 404,
"success": false
}{
"error": {
"details": {
"id": "ns_123",
"resource": "namespace"
},
"message": "Namespace not found",
"type": "NotFoundError"
},
"status": 404,
"success": false
}{
"error": {
"details": {
"id": "ns_123",
"resource": "namespace"
},
"message": "Namespace not found",
"type": "NotFoundError"
},
"status": 404,
"success": false
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>",
"input": "<unknown>",
"ctx": {}
}
]
}{
"error": {
"details": {
"id": "ns_123",
"resource": "namespace"
},
"message": "Namespace not found",
"type": "NotFoundError"
},
"status": 404,
"success": false
}Authorizations
Mixpeek API key, sent as Authorization: Bearer mxp_sk_.... Create one in Studio under Settings → API Keys, or with an admin key via POST /v1/organizations/users/{user_email}/api-keys. A missing header returns 403; an invalid or revoked key returns 401.
Namespace id (ns_...), not the namespace name. This scopes the request rather than authenticating it, and it is required on every operation marked x-mixpeek-namespace-scoped.
Path Parameters
The ID or name of the collection to export
Body
Request model for exporting collection data.
Export Formats:
- JSON: Line-delimited JSON (JSONL) format, one document per line. Good for streaming and large files.
- CSV: Comma-separated values. Best for tabular data analysis in spreadsheets.
- PARQUET: Columnar format optimized for analytics. Best for large datasets and data pipelines.
- WEBDATASET: Tar shards in the WebDataset layout, one sample per document.
Each sample carries
<key>.json(the same row the other formats emit) and, withinclude_media=True, the source object's bytes as<key>.<ext>. Read it with any WebDataset loader.samples_per_shardsets the shard size.
Vector Export:
Vectors are stored separately from document metadata due to their large size.
When include_vectors=True, vectors are exported to a separate file with the naming convention:
{collection_name}_vectors.{format}
Lineage:
When include_lineage=True, every row also carries lineage_chain (each processing
step from the source object to this document), source_content_hash (SHA-256 of the
source content) and document_created_at (ISO 8601). In JSON the chain is a list; in
CSV and Parquet it is a JSON-encoded string so every row keeps one column type.
Field Selection:
Use select_fields to export only specific fields, reducing file size for large collections.
Supports dot notation for nested fields (e.g., "metadata.title").
Filtering: Apply filters to export a subset of documents. Uses the same LogicalOperator format as the documents list endpoint.
Export format: json (line-delimited), csv, or parquet (default).
json, csv, parquet, webdataset Whether to include vectors in the export. Vectors are exported to a separate file due to their large size. This significantly increases export time and file size.
Add lineage columns to every row: lineage_chain (the processing steps from source object to document), source_content_hash (SHA-256 of the source content) and document_created_at (ISO 8601). Kept even when select_fields is set. In CSV and Parquet, lineage_chain is a JSON-encoded string.
Specific fields to include in the export. If not provided, all fields are exported. Supports dot notation for nested fields (e.g., 'metadata.title', 'metadata.author').
[
"document_id",
"metadata.title",
"metadata.category"
]
Filter conditions to export only matching documents. Uses LogicalOperator format (AND/OR/NOT) same as document listing.
Show child attributes
Show child attributes
Maximum number of documents to export. If not provided, exports all documents. Useful for testing exports or creating sample datasets.
1 <= x <= 1000000WebDataset only. Put the source object's bytes in each sample alongside its row, resolved through the document's root object. A document whose object is gone, or which never had one, ships with its row alone and is counted in the response's media summary. This moves the full media set, so expect a much larger export and a longer run.
WebDataset only. Documents per tar shard. Smaller shards parallelize better across training workers; larger shards mean fewer files to move.
1 <= x <= 100000Write the export into your own storage through a connection you own. The files and a manifest land under the prefix you give. The 7-day download copy is still written too, so the presigned URLs in the response keep working and nothing existing changes.
Show child attributes
Show child attributes
Response
Successful Response
Response model for collection export.
Contains the presigned URL for downloading the exported file. The URL is valid for a limited time (typically 1 hour).
Presigned URL for downloading the exported file. Valid for 1 hour.
Full S3 path where the export is stored (for internal reference). For webdataset this is the prefix holding the shards, not a single file.
The format of the exported file.
json, csv, parquet, webdataset Number of documents included in the export.
x >= 0Size of the exported file in bytes. For webdataset this is every shard added together.
x >= 0Timestamp when the export was completed.
Presigned URL for downloading the vectors file (if include_vectors=True). Vectors are exported separately due to their large size.
Full S3 path for the vectors file (if include_vectors=True).
WebDataset only. Number of tar shards written. Always exact, even when the shards list below is capped.
x >= 0WebDataset only. Brace pattern naming every shard, e.g. documents-{000000..000007}.tar. Most loaders take this directly.
WebDataset only. Per-shard paths and presigned URLs, capped at 200 entries. When shard_count exceeds that, address the rest through s3_path and shard_pattern.
Show child attributes
Show child attributes
WebDataset only, present when include_media was set. Counts what media reached the shards and what did not.
Show child attributes
Show child attributes
Present when the request named a destination. Lists what was written into your storage and where the manifest is.
Show child attributes
Show child attributes
Was this page helpful?

