Get Document Indexing Status
Returns indexing progress for every document in a batch: the current stage and chunk completion counts. Poll until each indexing_status reaches completed or error. Status advances through waiting → parsing → cleaning → splitting → indexing → completed.
curl --request GET \
--url https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status \
--header 'Authorization: Bearer <token>'import requests
url = "https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body{
"data": [
{
"cleaning_completed_at": 1741267200,
"completed_at": 1741267200,
"completed_segments": 5,
"error": null,
"id": "a8e0e5b5-78c6-4130-a5ce-25feb0e0b4ac",
"indexing_status": "completed",
"parsing_completed_at": 1741267200,
"paused_at": null,
"processing_started_at": 1741267200,
"splitting_completed_at": 1741267200,
"stopped_at": null,
"total_segments": 5
}
]
}Authorizations
Every request authenticates with an API key: Authorization: Bearer {API_KEY}. App endpoints take an app API key; knowledge endpoints take a knowledge base API key (Get Started).
Keep keys server-side; never embed them in client code. Requests with a missing or invalid key fail with HTTP 401 (unauthorized).
Path Parameters
Knowledge base ID, from List Knowledge Bases. For a scoped key, copy the ID from the Dify URL.
Batch ID returned when you create or update a document.
Response
Indexing status for documents in the batch.
List of indexing status entries.
Show child attributes
Show child attributes
Was this page helpful?
curl --request GET \
--url https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status \
--header 'Authorization: Bearer <token>'import requests
url = "https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://{api_base_url}/datasets/{dataset_id}/documents/{batch}/indexing-status")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body{
"data": [
{
"cleaning_completed_at": 1741267200,
"completed_at": 1741267200,
"completed_segments": 5,
"error": null,
"id": "a8e0e5b5-78c6-4130-a5ce-25feb0e0b4ac",
"indexing_status": "completed",
"parsing_completed_at": 1741267200,
"paused_at": null,
"processing_started_at": 1741267200,
"splitting_completed_at": 1741267200,
"stopped_at": null,
"total_segments": 5
}
]
}