general information extration
curl --request POST \
--url https://api.textin.com/ai/service/v3/entity_extraction \
--header 'Content-Type: application/json' \
--header 'x-ti-app-id: <api-key>' \
--header 'x-ti-secret-code: <api-key>' \
--data '
{
"file": {
"file_base64": "/9j/4AAQSk...",
"file_url": "https://example.com/document.pdf",
"file_name": "document.pdf"
},
"schema": {
"type": "object",
"properties": {
"商品": {
"type": "string",
"description": "商品名称"
}
},
"required": [
"商品"
]
},
"parse_options": {
"page_start": 1,
"page_count": 10,
"get_image": "objects",
"crop_dewarp": 0,
"remove_watermark": 0,
"parse_mode": "scan",
"formula_level": 0,
"table_flavor": "html",
"pdf_pwd": "<string>"
},
"extract_options": {
"generate_citations": true,
"stamp": true
}
}
'import requests
url = "https://api.textin.com/ai/service/v3/entity_extraction"
payload = {
"file": {
"file_base64": "/9j/4AAQSk...",
"file_url": "https://example.com/document.pdf",
"file_name": "document.pdf"
},
"schema": {
"type": "object",
"properties": { "商品": {
"type": "string",
"description": "商品名称"
} },
"required": ["商品"]
},
"parse_options": {
"page_start": 1,
"page_count": 10,
"get_image": "objects",
"crop_dewarp": 0,
"remove_watermark": 0,
"parse_mode": "scan",
"formula_level": 0,
"table_flavor": "html",
"pdf_pwd": "<string>"
},
"extract_options": {
"generate_citations": True,
"stamp": True
}
}
headers = {
"x-ti-app-id": "<api-key>",
"x-ti-secret-code": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {
'x-ti-app-id': '<api-key>',
'x-ti-secret-code': '<api-key>',
'Content-Type': 'application/json'
},
body: JSON.stringify({
file: {
file_base64: '/9j/4AAQSk...',
file_url: 'https://example.com/document.pdf',
file_name: 'document.pdf'
},
schema: {
type: 'object',
properties: {'商品': {type: 'string', description: '商品名称'}},
required: ['商品']
},
parse_options: {
page_start: 1,
page_count: 10,
get_image: 'objects',
crop_dewarp: 0,
remove_watermark: 0,
parse_mode: 'scan',
formula_level: 0,
table_flavor: 'html',
pdf_pwd: '<string>'
},
extract_options: {generate_citations: true, stamp: true}
})
};
fetch('https://api.textin.com/ai/service/v3/entity_extraction', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.textin.com/ai/service/v3/entity_extraction",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'file' => [
'file_base64' => '/9j/4AAQSk...',
'file_url' => 'https://example.com/document.pdf',
'file_name' => 'document.pdf'
],
'schema' => [
'type' => 'object',
'properties' => [
'商品' => [
'type' => 'string',
'description' => '商品名称'
]
],
'required' => [
'商品'
]
],
'parse_options' => [
'page_start' => 1,
'page_count' => 10,
'get_image' => 'objects',
'crop_dewarp' => 0,
'remove_watermark' => 0,
'parse_mode' => 'scan',
'formula_level' => 0,
'table_flavor' => 'html',
'pdf_pwd' => '<string>'
],
'extract_options' => [
'generate_citations' => true,
'stamp' => true
]
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"x-ti-app-id: <api-key>",
"x-ti-secret-code: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.textin.com/ai/service/v3/entity_extraction"
payload := strings.NewReader("{\n \"file\": {\n \"file_base64\": \"/9j/4AAQSk...\",\n \"file_url\": \"https://example.com/document.pdf\",\n \"file_name\": \"document.pdf\"\n },\n \"schema\": {\n \"type\": \"object\",\n \"properties\": {\n \"商品\": {\n \"type\": \"string\",\n \"description\": \"商品名称\"\n }\n },\n \"required\": [\n \"商品\"\n ]\n },\n \"parse_options\": {\n \"page_start\": 1,\n \"page_count\": 10,\n \"get_image\": \"objects\",\n \"crop_dewarp\": 0,\n \"remove_watermark\": 0,\n \"parse_mode\": \"scan\",\n \"formula_level\": 0,\n \"table_flavor\": \"html\",\n \"pdf_pwd\": \"<string>\"\n },\n \"extract_options\": {\n \"generate_citations\": true,\n \"stamp\": true\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("x-ti-app-id", "<api-key>")
req.Header.Add("x-ti-secret-code", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.textin.com/ai/service/v3/entity_extraction")
.header("x-ti-app-id", "<api-key>")
.header("x-ti-secret-code", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"file\": {\n \"file_base64\": \"/9j/4AAQSk...\",\n \"file_url\": \"https://example.com/document.pdf\",\n \"file_name\": \"document.pdf\"\n },\n \"schema\": {\n \"type\": \"object\",\n \"properties\": {\n \"商品\": {\n \"type\": \"string\",\n \"description\": \"商品名称\"\n }\n },\n \"required\": [\n \"商品\"\n ]\n },\n \"parse_options\": {\n \"page_start\": 1,\n \"page_count\": 10,\n \"get_image\": \"objects\",\n \"crop_dewarp\": 0,\n \"remove_watermark\": 0,\n \"parse_mode\": \"scan\",\n \"formula_level\": 0,\n \"table_flavor\": \"html\",\n \"pdf_pwd\": \"<string>\"\n },\n \"extract_options\": {\n \"generate_citations\": true,\n \"stamp\": true\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.textin.com/ai/service/v3/entity_extraction")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["x-ti-app-id"] = '<api-key>'
request["x-ti-secret-code"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"file\": {\n \"file_base64\": \"/9j/4AAQSk...\",\n \"file_url\": \"https://example.com/document.pdf\",\n \"file_name\": \"document.pdf\"\n },\n \"schema\": {\n \"type\": \"object\",\n \"properties\": {\n \"商品\": {\n \"type\": \"string\",\n \"description\": \"商品名称\"\n }\n },\n \"required\": [\n \"商品\"\n ]\n },\n \"parse_options\": {\n \"page_start\": 1,\n \"page_count\": 10,\n \"get_image\": \"objects\",\n \"crop_dewarp\": 0,\n \"remove_watermark\": 0,\n \"parse_mode\": \"scan\",\n \"formula_level\": 0,\n \"table_flavor\": \"html\",\n \"pdf_pwd\": \"<string>\"\n },\n \"extract_options\": {\n \"generate_citations\": true,\n \"stamp\": true\n }\n}"
response = http.request(request)
puts response.read_body{
"code": 200,
"message": "Success",
"version": "v3.0.29_20250819",
"duration": 8267,
"x_request_id": "7596b8c9d2ddbc9924b66651e9efc174",
"status": "finished",
"result": {
"success_count": 1,
"extracted_schema": {
"商品": "童装 Looney Tunes UT(短袖T恤)女装SUPIMA COTTON圆领T恤(短袖)"
},
"citations": {
"商品": {
"value": "童装 Looney Tunes UT(短袖T恤)女装SUPIMA COTTON圆领T恤(短袖)",
"bounding_regions": [
{
"page_number": 1,
"position": [
137,
599,
1129,
599,
1129,
625,
182,
625
],
"text": "童装 Looney Tunes UT(短袖T恤)女装SUPIMA COTTON圆领T恤(短袖)"
}
]
}
},
"pages": [
{
"page_number": 1,
"status": "Success",
"durations": 930.178466796875,
"image_id": "62bfe3c3a8e9c9cf.jpg",
"height": 1824,
"width": 600,
"angle": 0
}
],
"stamps": [
{
"color": "红色",
"position": [
1223,
995,
1642,
1007,
1630,
1689,
1621,
1677
],
"stamp_shape": "圆章",
"type": "公章",
"value": "电力公司专用章"
}
]
},
"part_durations": {
"parse_duration": 1080,
"retrieve_duration": 0,
"prompt_duration": 1,
"llm_duration": 7114,
"format_duration": 51
}
}智能文档解析
智能抽取
智能抽取API已更新至v3, 如需查看旧版API请点击
快速调试:请参考Postman调试教程或Apifox调试教程
POST
/
ai
/
service
/
v3
/
entity_extraction
general information extration
curl --request POST \
--url https://api.textin.com/ai/service/v3/entity_extraction \
--header 'Content-Type: application/json' \
--header 'x-ti-app-id: <api-key>' \
--header 'x-ti-secret-code: <api-key>' \
--data '
{
"file": {
"file_base64": "/9j/4AAQSk...",
"file_url": "https://example.com/document.pdf",
"file_name": "document.pdf"
},
"schema": {
"type": "object",
"properties": {
"商品": {
"type": "string",
"description": "商品名称"
}
},
"required": [
"商品"
]
},
"parse_options": {
"page_start": 1,
"page_count": 10,
"get_image": "objects",
"crop_dewarp": 0,
"remove_watermark": 0,
"parse_mode": "scan",
"formula_level": 0,
"table_flavor": "html",
"pdf_pwd": "<string>"
},
"extract_options": {
"generate_citations": true,
"stamp": true
}
}
'import requests
url = "https://api.textin.com/ai/service/v3/entity_extraction"
payload = {
"file": {
"file_base64": "/9j/4AAQSk...",
"file_url": "https://example.com/document.pdf",
"file_name": "document.pdf"
},
"schema": {
"type": "object",
"properties": { "商品": {
"type": "string",
"description": "商品名称"
} },
"required": ["商品"]
},
"parse_options": {
"page_start": 1,
"page_count": 10,
"get_image": "objects",
"crop_dewarp": 0,
"remove_watermark": 0,
"parse_mode": "scan",
"formula_level": 0,
"table_flavor": "html",
"pdf_pwd": "<string>"
},
"extract_options": {
"generate_citations": True,
"stamp": True
}
}
headers = {
"x-ti-app-id": "<api-key>",
"x-ti-secret-code": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {
'x-ti-app-id': '<api-key>',
'x-ti-secret-code': '<api-key>',
'Content-Type': 'application/json'
},
body: JSON.stringify({
file: {
file_base64: '/9j/4AAQSk...',
file_url: 'https://example.com/document.pdf',
file_name: 'document.pdf'
},
schema: {
type: 'object',
properties: {'商品': {type: 'string', description: '商品名称'}},
required: ['商品']
},
parse_options: {
page_start: 1,
page_count: 10,
get_image: 'objects',
crop_dewarp: 0,
remove_watermark: 0,
parse_mode: 'scan',
formula_level: 0,
table_flavor: 'html',
pdf_pwd: '<string>'
},
extract_options: {generate_citations: true, stamp: true}
})
};
fetch('https://api.textin.com/ai/service/v3/entity_extraction', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.textin.com/ai/service/v3/entity_extraction",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'file' => [
'file_base64' => '/9j/4AAQSk...',
'file_url' => 'https://example.com/document.pdf',
'file_name' => 'document.pdf'
],
'schema' => [
'type' => 'object',
'properties' => [
'商品' => [
'type' => 'string',
'description' => '商品名称'
]
],
'required' => [
'商品'
]
],
'parse_options' => [
'page_start' => 1,
'page_count' => 10,
'get_image' => 'objects',
'crop_dewarp' => 0,
'remove_watermark' => 0,
'parse_mode' => 'scan',
'formula_level' => 0,
'table_flavor' => 'html',
'pdf_pwd' => '<string>'
],
'extract_options' => [
'generate_citations' => true,
'stamp' => true
]
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"x-ti-app-id: <api-key>",
"x-ti-secret-code: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.textin.com/ai/service/v3/entity_extraction"
payload := strings.NewReader("{\n \"file\": {\n \"file_base64\": \"/9j/4AAQSk...\",\n \"file_url\": \"https://example.com/document.pdf\",\n \"file_name\": \"document.pdf\"\n },\n \"schema\": {\n \"type\": \"object\",\n \"properties\": {\n \"商品\": {\n \"type\": \"string\",\n \"description\": \"商品名称\"\n }\n },\n \"required\": [\n \"商品\"\n ]\n },\n \"parse_options\": {\n \"page_start\": 1,\n \"page_count\": 10,\n \"get_image\": \"objects\",\n \"crop_dewarp\": 0,\n \"remove_watermark\": 0,\n \"parse_mode\": \"scan\",\n \"formula_level\": 0,\n \"table_flavor\": \"html\",\n \"pdf_pwd\": \"<string>\"\n },\n \"extract_options\": {\n \"generate_citations\": true,\n \"stamp\": true\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("x-ti-app-id", "<api-key>")
req.Header.Add("x-ti-secret-code", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.textin.com/ai/service/v3/entity_extraction")
.header("x-ti-app-id", "<api-key>")
.header("x-ti-secret-code", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"file\": {\n \"file_base64\": \"/9j/4AAQSk...\",\n \"file_url\": \"https://example.com/document.pdf\",\n \"file_name\": \"document.pdf\"\n },\n \"schema\": {\n \"type\": \"object\",\n \"properties\": {\n \"商品\": {\n \"type\": \"string\",\n \"description\": \"商品名称\"\n }\n },\n \"required\": [\n \"商品\"\n ]\n },\n \"parse_options\": {\n \"page_start\": 1,\n \"page_count\": 10,\n \"get_image\": \"objects\",\n \"crop_dewarp\": 0,\n \"remove_watermark\": 0,\n \"parse_mode\": \"scan\",\n \"formula_level\": 0,\n \"table_flavor\": \"html\",\n \"pdf_pwd\": \"<string>\"\n },\n \"extract_options\": {\n \"generate_citations\": true,\n \"stamp\": true\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.textin.com/ai/service/v3/entity_extraction")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["x-ti-app-id"] = '<api-key>'
request["x-ti-secret-code"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"file\": {\n \"file_base64\": \"/9j/4AAQSk...\",\n \"file_url\": \"https://example.com/document.pdf\",\n \"file_name\": \"document.pdf\"\n },\n \"schema\": {\n \"type\": \"object\",\n \"properties\": {\n \"商品\": {\n \"type\": \"string\",\n \"description\": \"商品名称\"\n }\n },\n \"required\": [\n \"商品\"\n ]\n },\n \"parse_options\": {\n \"page_start\": 1,\n \"page_count\": 10,\n \"get_image\": \"objects\",\n \"crop_dewarp\": 0,\n \"remove_watermark\": 0,\n \"parse_mode\": \"scan\",\n \"formula_level\": 0,\n \"table_flavor\": \"html\",\n \"pdf_pwd\": \"<string>\"\n },\n \"extract_options\": {\n \"generate_citations\": true,\n \"stamp\": true\n }\n}"
response = http.request(request)
puts response.read_body{
"code": 200,
"message": "Success",
"version": "v3.0.29_20250819",
"duration": 8267,
"x_request_id": "7596b8c9d2ddbc9924b66651e9efc174",
"status": "finished",
"result": {
"success_count": 1,
"extracted_schema": {
"商品": "童装 Looney Tunes UT(短袖T恤)女装SUPIMA COTTON圆领T恤(短袖)"
},
"citations": {
"商品": {
"value": "童装 Looney Tunes UT(短袖T恤)女装SUPIMA COTTON圆领T恤(短袖)",
"bounding_regions": [
{
"page_number": 1,
"position": [
137,
599,
1129,
599,
1129,
625,
182,
625
],
"text": "童装 Looney Tunes UT(短袖T恤)女装SUPIMA COTTON圆领T恤(短袖)"
}
]
}
},
"pages": [
{
"page_number": 1,
"status": "Success",
"durations": 930.178466796875,
"image_id": "62bfe3c3a8e9c9cf.jpg",
"height": 1824,
"width": 600,
"angle": 0
}
],
"stamps": [
{
"color": "红色",
"position": [
1223,
995,
1642,
1007,
1630,
1689,
1621,
1677
],
"stamp_shape": "圆章",
"type": "公章",
"value": "电力公司专用章"
}
]
},
"part_durations": {
"parse_duration": 1080,
"retrieve_duration": 0,
"prompt_duration": 1,
"llm_duration": 7114,
"format_duration": 51
}
}授权
请求体
application/json
支持的文件格式:png, jpg, jpeg, pdf, bmp, tiff, webp, doc, docx, html, mhtml, xls, xlsx, csv, ppt, pptx, txt, ofd;
支持schema模式的结构化信息抽取,通过定义字段结构进行精确抽取。
文件信息
Show child attributes
Show child attributes
抽取数据结构,参考JSON schema说明
示例:
{
"type": "object",
"properties": {
"商品": { "type": "string", "description": "商品名称" }
},
"required": ["商品"]
}
解析阶段参数
Show child attributes
Show child attributes
高级抽取控制
Show child attributes
Show child attributes
响应
200 - application/json
返回结果
状态码
- 200: Success (成功)
- 40101: x-ti-app-id 或 x-ti-secret-code 为空
- 40102: x-ti-app-id 或 x-ti-secret-code 无效,验证失败
- 40103: 客户端IP不在白名单
- 40003: 余额不足,请充值后再使用
- 40004: Parameter error (参数错误,请检查入参)
- 40007: 机器人不存在或未发布
- 40008: 机器人未开通,请至市场开通后重试
- 40302: 上传文件大小不符,文件大小不超过 50M
- 40303: 文件类型不支持,接口会返回实际检测到的文件类型,如“当前文件类型为.gif”
- 40304: 图片尺寸不符,长宽比小于2的图片宽高需在20~20000像素范围内,其他图片的宽高需在20~10000像素范围内
- 40305: File not uploaded (识别文件未上传)
- 40306: qps超过限制
- 40400: 无效的请求链接,请检查链接是否正确
- 40422: The file is corrupted (文件损坏)
- 40423: Password required or incorrect password (PDF密码错误)
- 40424: Page number out of range (页面设置超出文件范围)
- 40425: The input file format is not supported (输入文件格式不支持)
- 40428: Process office file failed (word和ppt转pdf失败或者超时)
- 500: Engine failed (服务器内部错误)
- 50011: LLM Connection Failed (访问大模型超时)
- 50012: LLM Engine Failed (大模型引擎错误)
- 50207: Partial failed (部分页面解析失败)
可用选项:
200, 40101, 40102, 40103, 40003, 40004, 40007, 40008, 40302, 40303, 40304, 40305, 40306, 40400, 40422, 40423, 40424, 40425, 40428, 500, 50011, 50012, 50207 示例:
200
成功或错误信息
示例:
"Success"
版本号
示例:
"v3.0.29_20250819"
总耗时(ms)
示例:
8267
请求ID
示例:
"7596b8c9d2ddbc9924b66651e9efc174"
处理状态
示例:
"finished"
Show child attributes
Show child attributes
各阶段耗时统计
Show child attributes
Show child attributes
此页面对您有帮助吗?
⌘I

