Create Document by Text
Creates a document in a knowledge base from raw text. Indexing runs asynchronously; track it with the returned batch ID via Get Document Indexing Status.
curl --request POST \
--url https://{api_base_url}/datasets/{dataset_id}/document/create-by-text \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"indexing_technique": "high_quality",
"name": "guide.txt",
"text": "Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents."
}
'import requests
url = "https://{api_base_url}/datasets/{dataset_id}/document/create-by-text"
payload = {
"indexing_technique": "high_quality",
"name": "guide.txt",
"text": "Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents."
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
indexing_technique: 'high_quality',
name: 'guide.txt',
text: 'Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents.'
})
};
fetch('https://{api_base_url}/datasets/{dataset_id}/document/create-by-text', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://{api_base_url}/datasets/{dataset_id}/document/create-by-text",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'indexing_technique' => 'high_quality',
'name' => 'guide.txt',
'text' => 'Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents.'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://{api_base_url}/datasets/{dataset_id}/document/create-by-text"
payload := strings.NewReader("{\n \"indexing_technique\": \"high_quality\",\n \"name\": \"guide.txt\",\n \"text\": \"Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents.\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://{api_base_url}/datasets/{dataset_id}/document/create-by-text")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"indexing_technique\": \"high_quality\",\n \"name\": \"guide.txt\",\n \"text\": \"Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents.\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://{api_base_url}/datasets/{dataset_id}/document/create-by-text")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"indexing_technique\": \"high_quality\",\n \"name\": \"guide.txt\",\n \"text\": \"Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents.\"\n}"
response = http.request(request)
puts response.read_bodyAuthorizations
Every request authenticates with an API key: Authorization: Bearer {API_KEY}. App endpoints take an app API key; knowledge endpoints take a knowledge base API key (Get Started).
Keep keys server-side; never embed them in client code. Requests with a missing or invalid key fail with HTTP 401 (unauthorized).
Path Parameters
Knowledge base ID, from List Knowledge Bases. For a scoped key, copy the ID from the Dify URL.
Body
Document name.
Document text content.
text_model for standard text chunking, hierarchical_model for parent-child chunk structure, qa_model for question-answer pair extraction.
text_model, hierarchical_model, qa_model Language of the document for processing optimization.
Embedding model name. Use the model field from Get Available Models with model_type=text-embedding.
Embedding model provider. Use the provider field from Get Available Models with model_type=text-embedding.
Required when adding the first document to a knowledge base. Subsequent documents inherit the knowledge base's indexing technique if omitted. high_quality uses embedding models for precise search; economy uses keyword-based indexing.
high_quality, economy Original document ID for versioning. Get it from List Documents.
Processing rules for chunking.
Show child attributes
Show child attributes
Controls how chunks are searched and ranked when querying this knowledge base.
Show child attributes
Show child attributes
Was this page helpful?
curl --request POST \
--url https://{api_base_url}/datasets/{dataset_id}/document/create-by-text \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"indexing_technique": "high_quality",
"name": "guide.txt",
"text": "Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents."
}
'import requests
url = "https://{api_base_url}/datasets/{dataset_id}/document/create-by-text"
payload = {
"indexing_technique": "high_quality",
"name": "guide.txt",
"text": "Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents."
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
indexing_technique: 'high_quality',
name: 'guide.txt',
text: 'Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents.'
})
};
fetch('https://{api_base_url}/datasets/{dataset_id}/document/create-by-text', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://{api_base_url}/datasets/{dataset_id}/document/create-by-text",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'indexing_technique' => 'high_quality',
'name' => 'guide.txt',
'text' => 'Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents.'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://{api_base_url}/datasets/{dataset_id}/document/create-by-text"
payload := strings.NewReader("{\n \"indexing_technique\": \"high_quality\",\n \"name\": \"guide.txt\",\n \"text\": \"Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents.\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://{api_base_url}/datasets/{dataset_id}/document/create-by-text")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"indexing_technique\": \"high_quality\",\n \"name\": \"guide.txt\",\n \"text\": \"Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents.\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://{api_base_url}/datasets/{dataset_id}/document/create-by-text")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"indexing_technique\": \"high_quality\",\n \"name\": \"guide.txt\",\n \"text\": \"Dify is an open-source LLM application development platform. This guide covers building AI applications with workflows, RAG pipelines, and agents.\"\n}"
response = http.request(request)
puts response.read_body