curl --request POST \
--url https://api.runpulse.com/extract \
--header 'Content-Type: multipart/form-data' \
--header 'x-api-key: <api-key>' \
--form file=@example-file \
--form 'fileUrl=<string>' \
--form detectSelections=true \
--form extractionConfigId=3c90c3cc-0d44-4b50-8888-8dd25736052a \
--form 'pages=<string>' \
--form forceUrl=false \
--form 'figureProcessing={
"description": false,
"showImages": false
};type=application/json' \
--form 'extensions={
"footnoteReferences": false,
"document_metadata": false,
"chunking": {
"chunkTypes": [],
"chunkSize": 2
},
"altOutputs": {
"wlbb": false,
"returnHtml": false,
"returnXml": false
}
};type=application/json' \
--form 'spreadsheet={
"includeHiddenRows": false,
"includeHiddenCols": false,
"includeHiddenSheets": false,
"useRawValues": false,
"onlyDataRows": false,
"onlyDataCols": false,
"includeCellFormatting": false,
"cellDataMode": "inline"
};type=application/json' \
--form 'storage={
"enabled": true,
"folderName": "<string>",
"folderId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
};type=application/json' \
--form async=false \
--form 'structuredOutput={
"schema": {},
"schemaPrompt": "<string>",
"effort": false,
"agenticConfidence": false
};type=application/json' \
--form 'schema={};type=application/json' \
--form 'schemaPrompt=<string>' \
--form 'customPrompt=<string>' \
--form 'chunking=<string>' \
--form chunkSize=2 \
--form extractFigure=false \
--form figureDescription=false \
--form showImages=false \
--form returnHtml=false \
--form thinking=falseimport requests
url = "https://api.runpulse.com/extract"
files = { "file": ("example-file", open("example-file", "rb")) }
files["figureProcessing"] = (None, "{\n \"description\": false,\n \"showImages\": false\n}", "application/json")
files["extensions"] = (None, "{\n \"footnoteReferences\": false,\n \"document_metadata\": false,\n \"chunking\": {\n \"chunkTypes\": [],\n \"chunkSize\": 2\n },\n \"altOutputs\": {\n \"wlbb\": false,\n \"returnHtml\": false,\n \"returnXml\": false\n }\n}", "application/json")
files["spreadsheet"] = (None, "{\n \"includeHiddenRows\": false,\n \"includeHiddenCols\": false,\n \"includeHiddenSheets\": false,\n \"useRawValues\": false,\n \"onlyDataRows\": false,\n \"onlyDataCols\": false,\n \"includeCellFormatting\": false,\n \"cellDataMode\": \"inline\"\n}", "application/json")
files["storage"] = (None, "{\n \"enabled\": true,\n \"folderName\": \"<string>\",\n \"folderId\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\"\n}", "application/json")
files["structuredOutput"] = (None, "{\n \"schema\": {},\n \"schemaPrompt\": \"<string>\",\n \"effort\": false,\n \"agenticConfidence\": false\n}", "application/json")
files["schema"] = (None, "{}", "application/json")
payload = {
"fileUrl": "<string>",
"detectSelections": "true",
"extractionConfigId": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"pages": "<string>",
"forceUrl": "false",
"async": "false",
"schemaPrompt": "<string>",
"customPrompt": "<string>",
"chunking": "<string>",
"chunkSize": "2",
"extractFigure": "false",
"figureDescription": "false",
"showImages": "false",
"returnHtml": "false",
"thinking": "false"
}
headers = {"x-api-key": "<api-key>"}
response = requests.post(url, data=payload, files=files, headers=headers)
print(response.text)const form = new FormData();
form.append('file', '<string>');
form.append('fileUrl', '<string>');
form.append('detectSelections', 'true');
form.append('extractionConfigId', '3c90c3cc-0d44-4b50-8888-8dd25736052a');
form.append('pages', '<string>');
form.append('forceUrl', 'false');
form.append('figureProcessing', '{
"description": false,
"showImages": false
}');
form.append('extensions', '{
"footnoteReferences": false,
"document_metadata": false,
"chunking": {
"chunkTypes": [],
"chunkSize": 2
},
"altOutputs": {
"wlbb": false,
"returnHtml": false,
"returnXml": false
}
}');
form.append('spreadsheet', '{
"includeHiddenRows": false,
"includeHiddenCols": false,
"includeHiddenSheets": false,
"useRawValues": false,
"onlyDataRows": false,
"onlyDataCols": false,
"includeCellFormatting": false,
"cellDataMode": "inline"
}');
form.append('storage', '{
"enabled": true,
"folderName": "<string>",
"folderId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}');
form.append('async', 'false');
form.append('structuredOutput', '{
"schema": {},
"schemaPrompt": "<string>",
"effort": false,
"agenticConfidence": false
}');
form.append('schema', '{}');
form.append('schemaPrompt', '<string>');
form.append('customPrompt', '<string>');
form.append('chunking', '<string>');
form.append('chunkSize', '2');
form.append('extractFigure', 'false');
form.append('figureDescription', 'false');
form.append('showImages', 'false');
form.append('returnHtml', 'false');
form.append('thinking', 'false');
const options = {method: 'POST', headers: {'x-api-key': '<api-key>'}};
options.body = form;
fetch('https://api.runpulse.com/extract', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.runpulse.com/extract",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => "-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"file\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"fileUrl\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"detectSelections\"\r\n\r\ntrue\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractionConfigId\"\r\n\r\n3c90c3cc-0d44-4b50-8888-8dd25736052a\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"pages\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"forceUrl\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureProcessing\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"description\": false,\r\n \"showImages\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extensions\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"footnoteReferences\": false,\r\n \"document_metadata\": false,\r\n \"chunking\": {\r\n \"chunkTypes\": [],\r\n \"chunkSize\": 2\r\n },\r\n \"altOutputs\": {\r\n \"wlbb\": false,\r\n \"returnHtml\": false,\r\n \"returnXml\": false\r\n }\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"spreadsheet\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"includeHiddenRows\": false,\r\n \"includeHiddenCols\": false,\r\n \"includeHiddenSheets\": false,\r\n \"useRawValues\": false,\r\n \"onlyDataRows\": false,\r\n \"onlyDataCols\": false,\r\n \"includeCellFormatting\": false,\r\n \"cellDataMode\": \"inline\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"storage\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"enabled\": true,\r\n \"folderName\": \"<string>\",\r\n \"folderId\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"async\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"structuredOutput\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"schema\": {},\r\n \"schemaPrompt\": \"<string>\",\r\n \"effort\": false,\r\n \"agenticConfidence\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schema\"\r\nContent-Type: application/json\r\n\r\n{}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schemaPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"customPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunking\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunkSize\"\r\n\r\n2\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractFigure\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureDescription\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"showImages\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"returnHtml\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"thinking\"\r\n\r\nfalse\r\n-----011000010111000001101001--",
CURLOPT_HTTPHEADER => [
"Content-Type: multipart/form-data",
"x-api-key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.runpulse.com/extract"
payload := strings.NewReader("-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"file\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"fileUrl\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"detectSelections\"\r\n\r\ntrue\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractionConfigId\"\r\n\r\n3c90c3cc-0d44-4b50-8888-8dd25736052a\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"pages\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"forceUrl\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureProcessing\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"description\": false,\r\n \"showImages\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extensions\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"footnoteReferences\": false,\r\n \"document_metadata\": false,\r\n \"chunking\": {\r\n \"chunkTypes\": [],\r\n \"chunkSize\": 2\r\n },\r\n \"altOutputs\": {\r\n \"wlbb\": false,\r\n \"returnHtml\": false,\r\n \"returnXml\": false\r\n }\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"spreadsheet\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"includeHiddenRows\": false,\r\n \"includeHiddenCols\": false,\r\n \"includeHiddenSheets\": false,\r\n \"useRawValues\": false,\r\n \"onlyDataRows\": false,\r\n \"onlyDataCols\": false,\r\n \"includeCellFormatting\": false,\r\n \"cellDataMode\": \"inline\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"storage\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"enabled\": true,\r\n \"folderName\": \"<string>\",\r\n \"folderId\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"async\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"structuredOutput\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"schema\": {},\r\n \"schemaPrompt\": \"<string>\",\r\n \"effort\": false,\r\n \"agenticConfidence\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schema\"\r\nContent-Type: application/json\r\n\r\n{}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schemaPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"customPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunking\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunkSize\"\r\n\r\n2\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractFigure\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureDescription\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"showImages\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"returnHtml\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"thinking\"\r\n\r\nfalse\r\n-----011000010111000001101001--")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("x-api-key", "<api-key>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.runpulse.com/extract")
.header("x-api-key", "<api-key>")
.body("-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"file\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"fileUrl\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"detectSelections\"\r\n\r\ntrue\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractionConfigId\"\r\n\r\n3c90c3cc-0d44-4b50-8888-8dd25736052a\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"pages\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"forceUrl\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureProcessing\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"description\": false,\r\n \"showImages\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extensions\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"footnoteReferences\": false,\r\n \"document_metadata\": false,\r\n \"chunking\": {\r\n \"chunkTypes\": [],\r\n \"chunkSize\": 2\r\n },\r\n \"altOutputs\": {\r\n \"wlbb\": false,\r\n \"returnHtml\": false,\r\n \"returnXml\": false\r\n }\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"spreadsheet\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"includeHiddenRows\": false,\r\n \"includeHiddenCols\": false,\r\n \"includeHiddenSheets\": false,\r\n \"useRawValues\": false,\r\n \"onlyDataRows\": false,\r\n \"onlyDataCols\": false,\r\n \"includeCellFormatting\": false,\r\n \"cellDataMode\": \"inline\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"storage\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"enabled\": true,\r\n \"folderName\": \"<string>\",\r\n \"folderId\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"async\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"structuredOutput\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"schema\": {},\r\n \"schemaPrompt\": \"<string>\",\r\n \"effort\": false,\r\n \"agenticConfidence\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schema\"\r\nContent-Type: application/json\r\n\r\n{}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schemaPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"customPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunking\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunkSize\"\r\n\r\n2\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractFigure\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureDescription\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"showImages\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"returnHtml\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"thinking\"\r\n\r\nfalse\r\n-----011000010111000001101001--")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.runpulse.com/extract")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["x-api-key"] = '<api-key>'
request.body = "-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"file\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"fileUrl\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"detectSelections\"\r\n\r\ntrue\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractionConfigId\"\r\n\r\n3c90c3cc-0d44-4b50-8888-8dd25736052a\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"pages\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"forceUrl\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureProcessing\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"description\": false,\r\n \"showImages\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extensions\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"footnoteReferences\": false,\r\n \"document_metadata\": false,\r\n \"chunking\": {\r\n \"chunkTypes\": [],\r\n \"chunkSize\": 2\r\n },\r\n \"altOutputs\": {\r\n \"wlbb\": false,\r\n \"returnHtml\": false,\r\n \"returnXml\": false\r\n }\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"spreadsheet\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"includeHiddenRows\": false,\r\n \"includeHiddenCols\": false,\r\n \"includeHiddenSheets\": false,\r\n \"useRawValues\": false,\r\n \"onlyDataRows\": false,\r\n \"onlyDataCols\": false,\r\n \"includeCellFormatting\": false,\r\n \"cellDataMode\": \"inline\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"storage\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"enabled\": true,\r\n \"folderName\": \"<string>\",\r\n \"folderId\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"async\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"structuredOutput\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"schema\": {},\r\n \"schemaPrompt\": \"<string>\",\r\n \"effort\": false,\r\n \"agenticConfidence\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schema\"\r\nContent-Type: application/json\r\n\r\n{}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schemaPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"customPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunking\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunkSize\"\r\n\r\n2\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractFigure\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureDescription\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"showImages\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"returnHtml\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"thinking\"\r\n\r\nfalse\r\n-----011000010111000001101001--"
response = http.request(request)
puts response.read_body{
"markdown": "<string>",
"extensions": {
"document_metadata": {
"file": {
"name": "<string>",
"extension": "<string>",
"media_type": "<string>",
"size_bytes": 1,
"sha256": "<string>"
},
"properties": {
"title": "<string>",
"authors": [
"<string>"
],
"subject": "<string>",
"description": "<string>",
"keywords": [
"<string>"
],
"language": "<string>",
"application": "<string>",
"application_version": "<string>",
"producer": "<string>",
"last_modified_by": "<string>",
"revision": "<string>",
"category": "<string>",
"content_status": "<string>",
"identifier": "<string>",
"version": "<string>",
"company": "<string>",
"manager": "<string>",
"template": "<string>",
"presentation_format": "<string>",
"editing_time_minutes": 1,
"copyright": "<string>",
"created_at": "<string>",
"modified_at": "<string>"
},
"structure": {
"page_count": 1,
"page_labels": [
"<string>"
],
"page_sizes": [
{}
],
"outline_count": 1,
"attachment_count": 1,
"annotation_count": 1,
"annotation_types": {},
"form_field_count": 1,
"sheet_count": 1,
"sheet_names": [
"<string>"
],
"sheet_visibility": {},
"active_sheet": "<string>",
"slide_count": 1,
"note_count": 1,
"hidden_slide_count": 1,
"multimedia_clip_count": 1,
"word_count": 1,
"character_count": 1,
"character_count_with_spaces": 1,
"line_count": 1,
"paragraph_count": 1,
"paragraph_count_derived": 1,
"table_count": 1,
"section_count": 1,
"embedded_image_count": 1,
"width_px": 1,
"height_px": 1,
"frame_count": 1,
"row_count": 1,
"max_column_count": 1
},
"warnings": [
"<string>"
],
"custom": {},
"format_specific": {}
},
"chunking": {
"semantic": [
"<string>"
],
"header": [
"<string>"
],
"page": [
"<string>"
],
"recursive": [
"<string>"
]
},
"footnoteReferences": [
{
"symbol": "<string>",
"footnoteTextId": "<string>",
"footnotePageNumber": 2,
"footnoteText": "<string>",
"referenceTextIds": [
"<string>"
],
"references": [
{
"textId": "<string>",
"tableId": "<string>",
"row": 123,
"column": 123,
"pageNumber": 2,
"markerBoundingBox": [
123
],
"source": "native"
}
]
}
],
"altOutputs": {
"wlbb": {
"words": [
{
"id": "<string>",
"text": "<string>",
"page_number": 2,
"bounding_box": [
123
],
"average_word_confidence": 123
}
],
"error": "<string>"
},
"html": "<string>",
"xml": "<string>"
}
},
"bounding_boxes": {
"Images": [
{
"id": "<string>",
"visual_type": "chart",
"content": "<string>",
"caption": "<string>",
"page_number": 2,
"confidence": 123,
"bounding_box": [
123
],
"image_url": "<string>",
"description": "<string>",
"classification": {
"confidence": 0.5,
"model": "<string>",
"error": "<string>"
},
"sheet_name": "<string>",
"sheet_index": 123,
"workbook_sheet_index": 123,
"excel_range": "<string>",
"chart_type": "<string>",
"chart_title": "<string>",
"source_ranges": [
"<string>"
],
"render_error": "<string>",
"description_error": "<string>"
}
],
"Tables": [
{
"table_info": {
"id": "<string>",
"dimensions": [
123
],
"excel_range": "<string>",
"sheet_name": "<string>",
"sheet_index": 123,
"workbook_sheet_index": 123,
"section_index": 123,
"section_type": "<string>",
"section_name": "<string>",
"table_name": "<string>",
"layout_type": "<string>",
"is_chart": true,
"chart_type": "<string>",
"chart_title": "<string>",
"source_ranges": [
"<string>"
],
"location": {},
"confidence": 0.5
},
"confidence": 0.5,
"cell_data": [
{
"confidence": 0.5
}
]
}
],
"Text": [
{
"id": "<string>",
"content": "<string>",
"page_number": 2,
"confidence": 0.5,
"reading_order": 1,
"bounding_box": [
123
],
"excel_range": "<string>",
"sheet_name": "<string>",
"sheet_index": 123,
"workbook_sheet_index": 123,
"selected": true
}
],
"Title": [
{
"id": "<string>",
"content": "<string>",
"page_number": 2,
"confidence": 0.5,
"reading_order": 1,
"bounding_box": [
123
],
"excel_range": "<string>",
"sheet_name": "<string>",
"sheet_index": 123,
"workbook_sheet_index": 123,
"selected": true
}
],
"Footer": [
{
"id": "<string>",
"content": "<string>",
"page_number": 2,
"confidence": 0.5,
"reading_order": 1,
"bounding_box": [
123
],
"excel_range": "<string>",
"sheet_name": "<string>",
"sheet_index": 123,
"workbook_sheet_index": 123,
"selected": true
}
],
"markdown_with_ids": "<string>"
},
"extraction_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"extraction_url": "<string>",
"page_count": 2,
"plan_info": {
"tier": "<string>",
"total_credits_used": 123,
"pages_used": 1,
"note": "<string>"
},
"warnings": [
"<string>"
],
"credits_used": 123,
"html": "<string>",
"chunks": {
"semantic": [
"<string>"
],
"header": [
"<string>"
],
"page": [
"<string>"
],
"recursive": [
"<string>"
]
},
"structured_output": {
"values": {},
"citations": {}
},
"input_schema": {},
"schema_error": "<string>",
"content": "<string>",
"job_id": "<string>",
"metadata": {}
}{
"job_id": "<string>",
"status": "pending",
"message": "<string>"
}Extract File
The primary endpoint for the Pulse API. Parses uploaded documents or remote file URLs and returns rich markdown content with optional structured data extraction based on user-provided schemas and extraction options.
Set async: true to return immediately with a job_id for polling via
GET /job/. Otherwise processes synchronously.
To process many files at once, see Batch Extract or the Batch Processing guide.
curl --request POST \
--url https://api.runpulse.com/extract \
--header 'Content-Type: multipart/form-data' \
--header 'x-api-key: <api-key>' \
--form file=@example-file \
--form 'fileUrl=<string>' \
--form detectSelections=true \
--form extractionConfigId=3c90c3cc-0d44-4b50-8888-8dd25736052a \
--form 'pages=<string>' \
--form forceUrl=false \
--form 'figureProcessing={
"description": false,
"showImages": false
};type=application/json' \
--form 'extensions={
"footnoteReferences": false,
"document_metadata": false,
"chunking": {
"chunkTypes": [],
"chunkSize": 2
},
"altOutputs": {
"wlbb": false,
"returnHtml": false,
"returnXml": false
}
};type=application/json' \
--form 'spreadsheet={
"includeHiddenRows": false,
"includeHiddenCols": false,
"includeHiddenSheets": false,
"useRawValues": false,
"onlyDataRows": false,
"onlyDataCols": false,
"includeCellFormatting": false,
"cellDataMode": "inline"
};type=application/json' \
--form 'storage={
"enabled": true,
"folderName": "<string>",
"folderId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
};type=application/json' \
--form async=false \
--form 'structuredOutput={
"schema": {},
"schemaPrompt": "<string>",
"effort": false,
"agenticConfidence": false
};type=application/json' \
--form 'schema={};type=application/json' \
--form 'schemaPrompt=<string>' \
--form 'customPrompt=<string>' \
--form 'chunking=<string>' \
--form chunkSize=2 \
--form extractFigure=false \
--form figureDescription=false \
--form showImages=false \
--form returnHtml=false \
--form thinking=falseimport requests
url = "https://api.runpulse.com/extract"
files = { "file": ("example-file", open("example-file", "rb")) }
files["figureProcessing"] = (None, "{\n \"description\": false,\n \"showImages\": false\n}", "application/json")
files["extensions"] = (None, "{\n \"footnoteReferences\": false,\n \"document_metadata\": false,\n \"chunking\": {\n \"chunkTypes\": [],\n \"chunkSize\": 2\n },\n \"altOutputs\": {\n \"wlbb\": false,\n \"returnHtml\": false,\n \"returnXml\": false\n }\n}", "application/json")
files["spreadsheet"] = (None, "{\n \"includeHiddenRows\": false,\n \"includeHiddenCols\": false,\n \"includeHiddenSheets\": false,\n \"useRawValues\": false,\n \"onlyDataRows\": false,\n \"onlyDataCols\": false,\n \"includeCellFormatting\": false,\n \"cellDataMode\": \"inline\"\n}", "application/json")
files["storage"] = (None, "{\n \"enabled\": true,\n \"folderName\": \"<string>\",\n \"folderId\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\"\n}", "application/json")
files["structuredOutput"] = (None, "{\n \"schema\": {},\n \"schemaPrompt\": \"<string>\",\n \"effort\": false,\n \"agenticConfidence\": false\n}", "application/json")
files["schema"] = (None, "{}", "application/json")
payload = {
"fileUrl": "<string>",
"detectSelections": "true",
"extractionConfigId": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"pages": "<string>",
"forceUrl": "false",
"async": "false",
"schemaPrompt": "<string>",
"customPrompt": "<string>",
"chunking": "<string>",
"chunkSize": "2",
"extractFigure": "false",
"figureDescription": "false",
"showImages": "false",
"returnHtml": "false",
"thinking": "false"
}
headers = {"x-api-key": "<api-key>"}
response = requests.post(url, data=payload, files=files, headers=headers)
print(response.text)const form = new FormData();
form.append('file', '<string>');
form.append('fileUrl', '<string>');
form.append('detectSelections', 'true');
form.append('extractionConfigId', '3c90c3cc-0d44-4b50-8888-8dd25736052a');
form.append('pages', '<string>');
form.append('forceUrl', 'false');
form.append('figureProcessing', '{
"description": false,
"showImages": false
}');
form.append('extensions', '{
"footnoteReferences": false,
"document_metadata": false,
"chunking": {
"chunkTypes": [],
"chunkSize": 2
},
"altOutputs": {
"wlbb": false,
"returnHtml": false,
"returnXml": false
}
}');
form.append('spreadsheet', '{
"includeHiddenRows": false,
"includeHiddenCols": false,
"includeHiddenSheets": false,
"useRawValues": false,
"onlyDataRows": false,
"onlyDataCols": false,
"includeCellFormatting": false,
"cellDataMode": "inline"
}');
form.append('storage', '{
"enabled": true,
"folderName": "<string>",
"folderId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}');
form.append('async', 'false');
form.append('structuredOutput', '{
"schema": {},
"schemaPrompt": "<string>",
"effort": false,
"agenticConfidence": false
}');
form.append('schema', '{}');
form.append('schemaPrompt', '<string>');
form.append('customPrompt', '<string>');
form.append('chunking', '<string>');
form.append('chunkSize', '2');
form.append('extractFigure', 'false');
form.append('figureDescription', 'false');
form.append('showImages', 'false');
form.append('returnHtml', 'false');
form.append('thinking', 'false');
const options = {method: 'POST', headers: {'x-api-key': '<api-key>'}};
options.body = form;
fetch('https://api.runpulse.com/extract', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.runpulse.com/extract",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => "-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"file\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"fileUrl\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"detectSelections\"\r\n\r\ntrue\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractionConfigId\"\r\n\r\n3c90c3cc-0d44-4b50-8888-8dd25736052a\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"pages\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"forceUrl\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureProcessing\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"description\": false,\r\n \"showImages\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extensions\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"footnoteReferences\": false,\r\n \"document_metadata\": false,\r\n \"chunking\": {\r\n \"chunkTypes\": [],\r\n \"chunkSize\": 2\r\n },\r\n \"altOutputs\": {\r\n \"wlbb\": false,\r\n \"returnHtml\": false,\r\n \"returnXml\": false\r\n }\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"spreadsheet\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"includeHiddenRows\": false,\r\n \"includeHiddenCols\": false,\r\n \"includeHiddenSheets\": false,\r\n \"useRawValues\": false,\r\n \"onlyDataRows\": false,\r\n \"onlyDataCols\": false,\r\n \"includeCellFormatting\": false,\r\n \"cellDataMode\": \"inline\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"storage\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"enabled\": true,\r\n \"folderName\": \"<string>\",\r\n \"folderId\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"async\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"structuredOutput\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"schema\": {},\r\n \"schemaPrompt\": \"<string>\",\r\n \"effort\": false,\r\n \"agenticConfidence\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schema\"\r\nContent-Type: application/json\r\n\r\n{}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schemaPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"customPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunking\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunkSize\"\r\n\r\n2\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractFigure\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureDescription\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"showImages\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"returnHtml\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"thinking\"\r\n\r\nfalse\r\n-----011000010111000001101001--",
CURLOPT_HTTPHEADER => [
"Content-Type: multipart/form-data",
"x-api-key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.runpulse.com/extract"
payload := strings.NewReader("-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"file\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"fileUrl\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"detectSelections\"\r\n\r\ntrue\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractionConfigId\"\r\n\r\n3c90c3cc-0d44-4b50-8888-8dd25736052a\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"pages\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"forceUrl\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureProcessing\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"description\": false,\r\n \"showImages\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extensions\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"footnoteReferences\": false,\r\n \"document_metadata\": false,\r\n \"chunking\": {\r\n \"chunkTypes\": [],\r\n \"chunkSize\": 2\r\n },\r\n \"altOutputs\": {\r\n \"wlbb\": false,\r\n \"returnHtml\": false,\r\n \"returnXml\": false\r\n }\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"spreadsheet\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"includeHiddenRows\": false,\r\n \"includeHiddenCols\": false,\r\n \"includeHiddenSheets\": false,\r\n \"useRawValues\": false,\r\n \"onlyDataRows\": false,\r\n \"onlyDataCols\": false,\r\n \"includeCellFormatting\": false,\r\n \"cellDataMode\": \"inline\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"storage\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"enabled\": true,\r\n \"folderName\": \"<string>\",\r\n \"folderId\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"async\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"structuredOutput\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"schema\": {},\r\n \"schemaPrompt\": \"<string>\",\r\n \"effort\": false,\r\n \"agenticConfidence\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schema\"\r\nContent-Type: application/json\r\n\r\n{}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schemaPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"customPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunking\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunkSize\"\r\n\r\n2\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractFigure\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureDescription\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"showImages\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"returnHtml\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"thinking\"\r\n\r\nfalse\r\n-----011000010111000001101001--")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("x-api-key", "<api-key>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.runpulse.com/extract")
.header("x-api-key", "<api-key>")
.body("-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"file\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"fileUrl\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"detectSelections\"\r\n\r\ntrue\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractionConfigId\"\r\n\r\n3c90c3cc-0d44-4b50-8888-8dd25736052a\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"pages\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"forceUrl\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureProcessing\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"description\": false,\r\n \"showImages\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extensions\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"footnoteReferences\": false,\r\n \"document_metadata\": false,\r\n \"chunking\": {\r\n \"chunkTypes\": [],\r\n \"chunkSize\": 2\r\n },\r\n \"altOutputs\": {\r\n \"wlbb\": false,\r\n \"returnHtml\": false,\r\n \"returnXml\": false\r\n }\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"spreadsheet\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"includeHiddenRows\": false,\r\n \"includeHiddenCols\": false,\r\n \"includeHiddenSheets\": false,\r\n \"useRawValues\": false,\r\n \"onlyDataRows\": false,\r\n \"onlyDataCols\": false,\r\n \"includeCellFormatting\": false,\r\n \"cellDataMode\": \"inline\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"storage\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"enabled\": true,\r\n \"folderName\": \"<string>\",\r\n \"folderId\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"async\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"structuredOutput\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"schema\": {},\r\n \"schemaPrompt\": \"<string>\",\r\n \"effort\": false,\r\n \"agenticConfidence\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schema\"\r\nContent-Type: application/json\r\n\r\n{}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schemaPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"customPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunking\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunkSize\"\r\n\r\n2\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractFigure\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureDescription\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"showImages\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"returnHtml\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"thinking\"\r\n\r\nfalse\r\n-----011000010111000001101001--")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.runpulse.com/extract")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["x-api-key"] = '<api-key>'
request.body = "-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"file\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"fileUrl\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"detectSelections\"\r\n\r\ntrue\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractionConfigId\"\r\n\r\n3c90c3cc-0d44-4b50-8888-8dd25736052a\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"pages\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"forceUrl\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureProcessing\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"description\": false,\r\n \"showImages\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extensions\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"footnoteReferences\": false,\r\n \"document_metadata\": false,\r\n \"chunking\": {\r\n \"chunkTypes\": [],\r\n \"chunkSize\": 2\r\n },\r\n \"altOutputs\": {\r\n \"wlbb\": false,\r\n \"returnHtml\": false,\r\n \"returnXml\": false\r\n }\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"spreadsheet\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"includeHiddenRows\": false,\r\n \"includeHiddenCols\": false,\r\n \"includeHiddenSheets\": false,\r\n \"useRawValues\": false,\r\n \"onlyDataRows\": false,\r\n \"onlyDataCols\": false,\r\n \"includeCellFormatting\": false,\r\n \"cellDataMode\": \"inline\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"storage\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"enabled\": true,\r\n \"folderName\": \"<string>\",\r\n \"folderId\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\"\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"async\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"structuredOutput\"\r\nContent-Type: application/json\r\n\r\n{\r\n \"schema\": {},\r\n \"schemaPrompt\": \"<string>\",\r\n \"effort\": false,\r\n \"agenticConfidence\": false\r\n}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schema\"\r\nContent-Type: application/json\r\n\r\n{}\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"schemaPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"customPrompt\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunking\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"chunkSize\"\r\n\r\n2\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"extractFigure\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"figureDescription\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"showImages\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"returnHtml\"\r\n\r\nfalse\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"thinking\"\r\n\r\nfalse\r\n-----011000010111000001101001--"
response = http.request(request)
puts response.read_body{
"markdown": "<string>",
"extensions": {
"document_metadata": {
"file": {
"name": "<string>",
"extension": "<string>",
"media_type": "<string>",
"size_bytes": 1,
"sha256": "<string>"
},
"properties": {
"title": "<string>",
"authors": [
"<string>"
],
"subject": "<string>",
"description": "<string>",
"keywords": [
"<string>"
],
"language": "<string>",
"application": "<string>",
"application_version": "<string>",
"producer": "<string>",
"last_modified_by": "<string>",
"revision": "<string>",
"category": "<string>",
"content_status": "<string>",
"identifier": "<string>",
"version": "<string>",
"company": "<string>",
"manager": "<string>",
"template": "<string>",
"presentation_format": "<string>",
"editing_time_minutes": 1,
"copyright": "<string>",
"created_at": "<string>",
"modified_at": "<string>"
},
"structure": {
"page_count": 1,
"page_labels": [
"<string>"
],
"page_sizes": [
{}
],
"outline_count": 1,
"attachment_count": 1,
"annotation_count": 1,
"annotation_types": {},
"form_field_count": 1,
"sheet_count": 1,
"sheet_names": [
"<string>"
],
"sheet_visibility": {},
"active_sheet": "<string>",
"slide_count": 1,
"note_count": 1,
"hidden_slide_count": 1,
"multimedia_clip_count": 1,
"word_count": 1,
"character_count": 1,
"character_count_with_spaces": 1,
"line_count": 1,
"paragraph_count": 1,
"paragraph_count_derived": 1,
"table_count": 1,
"section_count": 1,
"embedded_image_count": 1,
"width_px": 1,
"height_px": 1,
"frame_count": 1,
"row_count": 1,
"max_column_count": 1
},
"warnings": [
"<string>"
],
"custom": {},
"format_specific": {}
},
"chunking": {
"semantic": [
"<string>"
],
"header": [
"<string>"
],
"page": [
"<string>"
],
"recursive": [
"<string>"
]
},
"footnoteReferences": [
{
"symbol": "<string>",
"footnoteTextId": "<string>",
"footnotePageNumber": 2,
"footnoteText": "<string>",
"referenceTextIds": [
"<string>"
],
"references": [
{
"textId": "<string>",
"tableId": "<string>",
"row": 123,
"column": 123,
"pageNumber": 2,
"markerBoundingBox": [
123
],
"source": "native"
}
]
}
],
"altOutputs": {
"wlbb": {
"words": [
{
"id": "<string>",
"text": "<string>",
"page_number": 2,
"bounding_box": [
123
],
"average_word_confidence": 123
}
],
"error": "<string>"
},
"html": "<string>",
"xml": "<string>"
}
},
"bounding_boxes": {
"Images": [
{
"id": "<string>",
"visual_type": "chart",
"content": "<string>",
"caption": "<string>",
"page_number": 2,
"confidence": 123,
"bounding_box": [
123
],
"image_url": "<string>",
"description": "<string>",
"classification": {
"confidence": 0.5,
"model": "<string>",
"error": "<string>"
},
"sheet_name": "<string>",
"sheet_index": 123,
"workbook_sheet_index": 123,
"excel_range": "<string>",
"chart_type": "<string>",
"chart_title": "<string>",
"source_ranges": [
"<string>"
],
"render_error": "<string>",
"description_error": "<string>"
}
],
"Tables": [
{
"table_info": {
"id": "<string>",
"dimensions": [
123
],
"excel_range": "<string>",
"sheet_name": "<string>",
"sheet_index": 123,
"workbook_sheet_index": 123,
"section_index": 123,
"section_type": "<string>",
"section_name": "<string>",
"table_name": "<string>",
"layout_type": "<string>",
"is_chart": true,
"chart_type": "<string>",
"chart_title": "<string>",
"source_ranges": [
"<string>"
],
"location": {},
"confidence": 0.5
},
"confidence": 0.5,
"cell_data": [
{
"confidence": 0.5
}
]
}
],
"Text": [
{
"id": "<string>",
"content": "<string>",
"page_number": 2,
"confidence": 0.5,
"reading_order": 1,
"bounding_box": [
123
],
"excel_range": "<string>",
"sheet_name": "<string>",
"sheet_index": 123,
"workbook_sheet_index": 123,
"selected": true
}
],
"Title": [
{
"id": "<string>",
"content": "<string>",
"page_number": 2,
"confidence": 0.5,
"reading_order": 1,
"bounding_box": [
123
],
"excel_range": "<string>",
"sheet_name": "<string>",
"sheet_index": 123,
"workbook_sheet_index": 123,
"selected": true
}
],
"Footer": [
{
"id": "<string>",
"content": "<string>",
"page_number": 2,
"confidence": 0.5,
"reading_order": 1,
"bounding_box": [
123
],
"excel_range": "<string>",
"sheet_name": "<string>",
"sheet_index": 123,
"workbook_sheet_index": 123,
"selected": true
}
],
"markdown_with_ids": "<string>"
},
"extraction_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"extraction_url": "<string>",
"page_count": 2,
"plan_info": {
"tier": "<string>",
"total_credits_used": 123,
"pages_used": 1,
"note": "<string>"
},
"warnings": [
"<string>"
],
"credits_used": 123,
"html": "<string>",
"chunks": {
"semantic": [
"<string>"
],
"header": [
"<string>"
],
"page": [
"<string>"
],
"recursive": [
"<string>"
]
},
"structured_output": {
"values": {},
"citations": {}
},
"input_schema": {},
"schema_error": "<string>",
"content": "<string>",
"job_id": "<string>",
"metadata": {}
}{
"job_id": "<string>",
"status": "pending",
"message": "<string>"
}Overview
extraction_id. After extraction, you can optionally split the document into topics, apply schema extraction to get structured data, or use tables for span-aware table extraction.Handling mixed document types? /classify can run before Extract to route each raw document to the right pipeline — it only chooses which pipeline runs, so extraction still happens here.async: true to process asynchronously and poll for results via GET /job/jobId./extract on each file in parallel.Async Mode
Setasync: true to return immediately with a job ID for polling:
{
"file_url": "https://example.com/document.pdf",
"async": true
}
{
"job_id": "abc123-def456",
"status": "pending",
"message": "Document processing started"
}
GET /job/{job_id} to poll for completion.
Request
Document Source
Provide the document using one of these methods:| Field | Type | Description |
|---|---|---|
file | binary | Document file to upload directly (multipart/form-data). Limited to 100 MB — larger uploads return HTTP 413 (FILE_TOO_LARGE_USE_URL). |
file_url | string | Public or pre-signed URL that Pulse will download and extract. Accepted at any size. Files over 100 MB must be submitted this way, via a file_url pointing at your own hosted or presigned URL. |
Extraction Options
| Field | Type | Default | Description |
|---|---|---|---|
model | string | organization default | Omit to use your organization’s default. pulse-ultra-2 returns the Ultra response profile (adds bounding_boxes.ordered_elements, see Reading Order). pulse-ultra-1 keeps the request on the previous-generation engine; its responses carry average_word_confidence instead of confidence and no reading_order. |
pages | string | all pages | 1-indexed page ranges such as "1-2,5". All page numbers in the response (markdown page breaks, bounding boxes, tables, words) always reference original document page numbers, even when only a subset is requested. |
refine | boolean | string[] | object | false | Post-extraction correction pass. See Refinement below. |
refine_prompt | string | "" | Additional refinement guidance. Alias for refine.prompt, convenient with the boolean or array form of refine. |
additional_prompt | string | "" | Additional document or domain context for extraction. |
detect_selections | boolean | true | Detect selected and unselected checkboxes, radio buttons, checkmarks, and similar controls. |
custom_image_prompt | string | "" | Additional context for figure and chart interpretation. |
figure_processing | object | {} | Controls visual descriptions and image delivery. |
extensions | object | {} | Adds optional derived outputs without changing the core result contract. |
spreadsheet | object | {} | Controls workbook parsing and spreadsheet cell output. |
storage | object | enabled | Controls persistence of extraction artifacts. |
async | boolean | false | Return a job ID immediately and poll GET /job/{jobId}. |
force_url | boolean | false | Deliver the complete result through a one-time result URL. Large and spreadsheet results may use URL delivery automatically. |
Refinement
refine is off unless you enable it. It accepts three forms:
// 1. Boolean — run the general refinement pass
{ "refine": true }
// 2. Array — run only the listed modes
{ "refine": ["tables", "text"] }
// 3. Object — modes plus an optional prompt in one place
{ "refine": { "modes": ["tables", "layout"], "prompt": "Keep the source row and column order." } }
| Mode | What it corrects |
|---|---|
tables | Table structure, headers, and cell content |
text | OCR text, missing content, and numerical accuracy |
formatting | Bold, italic, strikethrough, super/subscript, and LaTeX formatting |
layout | Layout correctness pass (fixed behavior; does not accept a custom prompt) |
refine.prompt accepts a string that applies to the whole refinement pass, or an object keyed by mode (e.g. {"tables": "...", "text": "..."}) to scope guidance per mode. Prompt keys must be a subset of the requested modes, and layout never accepts a custom prompt. When you use the boolean or array form, pass the guidance through the sibling refine_prompt field instead; refine.prompt takes precedence over refine_prompt when both are present.
Refinement changes the normal markdown and bounding_boxes fields in place; it does not create a parallel response object.
Figure Processing
| Field | Type | Default | Description |
|---|---|---|---|
figure_processing.description | boolean | string[] | false | Generate descriptions for detected visuals. Pass true for all visual classes, or a target list such as ["chart"], ["image"], or ["chart", "image"]. |
figure_processing.show_images | boolean | string[] | false | Return authenticated image URLs under bounding_boxes.Images[]. Accepts the same boolean or target-list forms as description. |
custom_image_prompt | string | "" | Top-level field: additional context for figure and chart interpretation. |
show_images: true returns embedded charts and images with workbook-specific metadata such as sheet_name, excel_range, chart_type, and source_ranges.
Extensions
Extensions add derived outputs while preserving the core response fields.| Field | Type | Default | Output |
|---|---|---|---|
extensions.document_metadata | boolean | false | extensions.document_metadata |
extensions.footnote_references | boolean | false | extensions.footnoteReferences |
extensions.chunking.chunk_types | string[] | none | extensions.chunking |
extensions.chunking.chunk_size | integer | strategy default | extensions.chunking |
extensions.alt_outputs.wlbb | boolean | false | extensions.altOutputs.wlbb |
extensions.alt_outputs.return_html | boolean | false | extensions.altOutputs.html |
extensions.alt_outputs.return_xml | boolean | false | extensions.altOutputs.xml when available |
bounding_boxes.Words collection is always part of the extraction response. extensions.alt_outputs.wlbb (equivalently, the top-level word_level_bounding_boxes parameter) requests the additional extensions.altOutputs.wlbb word-box output used by existing WLBB workflows. Organizations can also have word-level output enabled by default for every extraction — contact support to configure an org-wide default.
Spreadsheet Options
| Field | Type | Default | Description |
|---|---|---|---|
spreadsheet.include_hidden_rows | boolean | false | Include hidden workbook rows. |
spreadsheet.include_hidden_cols | boolean | false | Include hidden workbook columns. |
spreadsheet.include_hidden_sheets | boolean | false | Include hidden sheets. |
spreadsheet.use_raw_values | boolean | false | Return underlying numeric values instead of display-formatted text where supported. |
spreadsheet.only_data_rows | boolean | false | Trim trailing empty rows after the last data-bearing cell. |
spreadsheet.only_data_cols | boolean | false | Trim trailing empty columns after the last data-bearing cell. |
spreadsheet.include_cell_formatting | boolean | false | Include captured formatting and formula metadata in spreadsheet table cells. |
spreadsheet.cell_data_mode | inline | external | inline | Return full cell_data arrays inline, or per-table references to JSONL cell artifacts. |
Complete Processing Example
{
"file_url": "https://example.com/document.pdf",
"async": true,
"additional_prompt": "Preserve account labels and signed amounts.",
"refine": {
"modes": ["tables", "text"],
"prompt": "Keep the source row and column order."
},
"detect_selections": true,
"figure_processing": {
"description": true,
"show_images": true
},
"custom_image_prompt": "Describe axes, units, and legends.",
"extensions": {
"chunking": {
"chunk_types": ["semantic", "page"],
"chunk_size": 1000
},
"document_metadata": true
}
}
Storage Options
Control whether extractions are saved to your extraction library:| Field | Type | Default | Description |
|---|---|---|---|
storage.enabled | boolean | true | Whether to persist extraction artifacts. Set to false for temporary extractions. |
storage.folder_name | string | - | Target folder name to save the extraction to. Creates the folder if it doesn’t exist. |
storage.folder_id | string (uuid) | - | Target folder ID to save the extraction to. Takes precedence over folder_name. |
Response
The response structure varies based on document size to optimize for different use cases.Standard Inline Response
When the response payload stays below the inline threshold, results are returned directly in the response body:{
"markdown": "# Document Title\n\nExtracted content...",
"page_count": 15,
"extraction_id": "abc123-def456-ghi789",
"extraction_url": "https://platform.runpulse.com/dashboard/extractions/abc123",
"credits_used": 15.0,
"plan_info": {
"tier": "growth",
"pages_used": 240,
"total_credits_used": 240.0
},
"plan-info": {
"tier": "growth",
"pages_used": 240,
"total_credits_used": 240.0
},
"bounding_boxes": {
"Title": [
{
"id": "txt-1",
"content": "0a-Document Title",
"original_content": "Document Title",
"bounding_box": [0.1, 0.08, 0.7, 0.08, 0.7, 0.12, 0.1, 0.12],
"page_number": 1,
"confidence": 0.99,
"reading_order": 0
}
],
"Tables": [
{
"table_info": {
"id": "tbl-1",
"dimensions": {"rows": 3, "columns": 2},
"location": {
"coordinates": [0.1, 0.3, 0.9, 0.3, 0.9, 0.6, 0.1, 0.6],
"page": 1
},
"confidence": 0.96
},
"cell_data": [
{
"id": "tbl-1-r0c0",
"position": {"row": 0, "column": 0},
"text": "0t-Account",
"location": {
"coordinates": [0.1, 0.3, 0.5, 0.3, 0.5, 0.4, 0.1, 0.4],
"page": 1
},
"confidence": 0.97,
"properties": {"type": "header"}
}
]
}
],
"Words": [
{
"content": "Document",
"page_number": 1,
"bounding_box": [
{"x": 0.1, "y": 0.08},
{"x": 0.3, "y": 0.08},
{"x": 0.3, "y": 0.12},
{"x": 0.1, "y": 0.12}
],
"confidence": 0.99
}
],
"markdown_with_ids": "<div data-bb-text-id=\"txt-1\">..."
},
"extensions": {
"chunking": {
"semantic": ["chunk 1...", "chunk 2..."],
"header": ["section 1...", "section 2..."]
},
"altOutputs": {
"html": "<html>...</html>"
}
}
}
Response Fields
| Field | Type | Description |
|---|---|---|
markdown | string | Clean markdown content extracted from the document. Always present. |
page_count | integer | Total number of pages processed. |
extraction_id | string (uuid) | Persisted extraction ID. Present when storage is enabled (default). Use with /split and /schema. |
extraction_url | string | URL to view the extraction in the Pulse Platform. Present when storage is enabled. |
credits_used | number | Credits consumed by this request. Only present when the org has the credit billing system enabled. |
plan_info | object | Billing tier and cumulative usage information for the calling org, including tier, total_credits_used, pages_used, and an optional note. |
plan-info | object | Alias of plan_info with the same value. |
bounding_boxes | object | Typed layout data including text categories, Tables[].table_info, Tables[].cell_data, Words, visual items, and markdown_with_ids. Every element carries confidence (0-1, the extractor’s confidence it was read correctly) and reading_order; table cells carry their own confidence. Empty detected categories may be omitted. See Bounding Boxes. |
extensions | object | Output from enabled extensions. Only keys for enabled extensions are present. See below. |
extensions.document_metadata | object | Native properties and deterministic structure from the original file (when extensions.document_metadata is enabled). See Document Metadata below. |
extensions.chunking | object | Chunk results by strategy (when extensions.chunking is enabled). |
extensions.footnoteReferences | array | List of detected footnotes with their in-text references (when extensions.footnote_references is enabled). See Footnote References below. |
extensions.altOutputs.wlbb | object | Word-level bounding boxes (when extensions.alt_outputs.wlbb is enabled). |
extensions.altOutputs.html | string | HTML representation (when extensions.alt_outputs.return_html is enabled). |
extensions.altOutputs.xml | string | XML representation (when extensions.alt_outputs.return_xml is enabled, WIP). |
warnings | array | Optional non-fatal warnings generated during extraction. |
failed_pages | array | Only present on partial results. Each entry is {"page": number, "error": string} describing a page that could not be extracted. See Partial Results below. |
Partial Results (failed_pages)
Extraction completes with a partial result whenever at least one page succeeds. A job fails outright only when every page is unextractable. On a partial result:
failed_pagesat the top level lists each unrecoverable page with a short reason.- A matching entry is appended to
warnings. - The markdown carries a visible
<!-- PAGE N FAILED EXTRACTION: ... -->placeholder at each hole, so page-oriented consumers keep their alignment. - Only successfully extracted pages are billed.
{
"markdown": "...\n<!-- PAGE 7153 FAILED EXTRACTION: content could not be certified -->\n...",
"page_count": 11622,
"failed_pages": [
{"page": 7153, "error": "content could not be certified"}
],
"warnings": [
"1 of 11622 pages could not be certified; see failed_pages"
]
}
failed_pages as the authoritative list of holes: if it is absent, every requested page was extracted.
bounding_boxes with grouped Words, Tables[].cell_data with ids, spans, and confidences, and chunking output — and extraction bills at a flat 1 credit per page. Callers that explicitly request model="pulse-ultra-2" continue to receive today’s Ultra response format unchanged. To keep a request on the previous-generation engine, send model="pulse-ultra-1"; those responses carry average_word_confidence instead of confidence and no reading_order.Large Result Response
When a response reaches the configured inline-size threshold (5 MB by default), or whenforce_url: true is supplied, the API returns a download URL instead of inlining the payload. Some rollback deployments may also offload documents over 70 pages. The downloaded JSON is the complete extraction result with the normal response shape.
{
"is_url": true,
"url": "https://api.runpulse.com/results/abc123-def456-ghi789",
"extraction_id": "abc123-def456-ghi789"
}
Large Result Response Fields
| Field | Type | Description |
|---|---|---|
is_url | boolean | Always true for URL-backed responses. Use this to detect URL-based responses. |
url | string | Download URL for the complete JSON result. It may be a Pulse /results/{job_id} URL or a presigned storage URL, depending on organization settings. Anonymous Pulse result links are single-use and expire after 1 hour; same-org authenticated callers can replay them while the artifact remains available. |
extraction_id | string | Extraction/job identifier. The downloaded result contains the normal response fields, including metadata such as page_count, credits_used, and plan_info when available. |
Handling Large Document Responses
import requests
from pulse import Pulse
API_KEY = "YOUR_API_KEY"
client = Pulse(api_key=API_KEY)
response = client.extract(
file_url="https://platform.runpulse.com/api/examples/637e5678-30b1-45fa-acc4-877f2d636419/pdf"
)
if hasattr(response, "is_url") and response.is_url:
full_result = requests.get(
response.url,
headers={"x-api-key": API_KEY},
).json()
print(full_result["markdown"])
else:
print(response.markdown)
import { PulseClient } from 'pulse-ts-sdk';
const API_KEY = "YOUR_API_KEY";
const client = new PulseClient({ apiKey: API_KEY });
const response = await client.extract({
fileUrl: "https://platform.runpulse.com/api/examples/637e5678-30b1-45fa-acc4-877f2d636419/pdf"
});
if ((response as any).is_url) {
const fullResult = await fetch((response as any).url, {
headers: { "x-api-key": API_KEY },
}).then(r => r.json());
console.log(fullResult.markdown);
} else {
console.log(response.markdown);
}
curl -X POST https://api.runpulse.com/extract \
-H "x-api-key: YOUR_API_KEY" \
-F "file=@large_document.pdf"
# Response: {"is_url": true, "url": "https://api.runpulse.com/results/abc123-..."}
# Fetch the complete result
curl -H "x-api-key: YOUR_API_KEY" \
"https://api.runpulse.com/results/abc123-..."
storage.enabled and retrieve saved extractions from the Pulse Platform.Example Usage
Core Processing
Userefine, refine_prompt, additional_prompt, and detect_selections to steer or refine the primary extraction result. These are separate from extensions, which add derived outputs.
curl -X POST https://api.runpulse.com/extract \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"file_url": "https://example.com/bank-statement.pdf",
"additional_prompt": "Preserve account labels and signed amounts.",
"refine": {
"modes": ["tables", "text"],
"prompt": "Keep source row and column order."
},
"detect_selections": true,
"figure_processing": {
"description": true
},
"custom_image_prompt": "Describe axes, units, and legends."
}'
refine also accepts a boolean ("refine": true runs the general pass) or a bare array of modes ("refine": ["tables", "formatting", "text", "layout"]); with those forms, pass refinement guidance through the sibling refine_prompt field. See Core Processing for precedence and defaults.
Basic Extraction
from pulse import Pulse
from pulse.types import (
ExtractRequestFigureProcessing,
ExtractRequestExtensions,
ExtractRequestExtensionsAltOutputs,
)
client = Pulse(api_key="YOUR_API_KEY")
# Extract from URL with figure processing and HTML output
response = client.extract(
file_url="https://platform.runpulse.com/api/examples/637e5678-30b1-45fa-acc4-877f2d636419/pdf",
figure_processing=ExtractRequestFigureProcessing(
description=True,
),
extensions=ExtractRequestExtensions(
alt_outputs=ExtractRequestExtensionsAltOutputs(
return_html=True,
),
),
)
print(f"Markdown: {response.markdown}")
print(f"HTML: {response.extensions.alt_outputs.html}")
print(f"Extraction ID: {response.extraction_id}")
import { PulseClient } from 'pulse-ts-sdk';
const client = new PulseClient({ apiKey: "YOUR_API_KEY" });
const response = await client.extract({
fileUrl: "https://platform.runpulse.com/api/examples/637e5678-30b1-45fa-acc4-877f2d636419/pdf",
figureProcessing: { description: true },
extensions: { altOutputs: { returnHtml: true } }
});
console.log(`Markdown: ${response.markdown}`);
console.log(`HTML: ${response.extensions?.altOutputs?.html}`);
console.log(`Extraction ID: ${response.extraction_id}`);
# Extract from URL with figure processing
curl -X POST https://api.runpulse.com/extract \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"file_url": "https://platform.runpulse.com/api/examples/637e5678-30b1-45fa-acc4-877f2d636419/pdf",
"figure_processing": {"description": true},
"extensions": {"alt_outputs": {"return_html": true}}
}'
File Upload
from pulse.types import ExtractRequestFigureProcessing
# Upload and extract a local file
with open("document.pdf", "rb") as f:
response = client.extract(
file=f,
figure_processing=ExtractRequestFigureProcessing(
description=True,
),
)
import * as fs from 'fs';
const fileBuffer = fs.readFileSync("document.pdf");
const blob = new Blob([fileBuffer], { type: 'application/pdf' });
const response = await client.extract({
file: blob,
});
curl -X POST https://api.runpulse.com/extract \
-H "x-api-key: YOUR_API_KEY" \
-F "file=@document.pdf"
file=) are limited to 100 MB. Larger uploads fail fast with HTTP 413 and error code FILE_TOO_LARGE_USE_URL. Files over 100 MB must be submitted via file_url pointing at your own hosted or presigned URL — file_url submissions are accepted at any size. Very large documents are sharded and processed transparently under a single job ID; see Working with Large Documents.Structured Data (Extract → Schema)
/schema after extraction. The resulting extraction_id lets you rerun or change schemas without processing the source document again.# Step 1: Extract the document
response = client.extract(
file_url="https://platform.runpulse.com/api/examples/637e5678-30b1-45fa-acc4-877f2d636419/pdf"
)
extraction_id = response.extraction_id
# Step 2: Apply schema separately
schema_result = client.schema(
extraction_id=extraction_id,
schema_config={
"input_schema": {
"type": "object",
"properties": {
"total": {"type": "number"},
"vendor": {"type": "string"}
}
},
"schema_prompt": "Extract invoice total and vendor"
}
)
print(schema_result.schema_output)
// Step 1: Extract the document
const response = await client.extract({
fileUrl: "https://platform.runpulse.com/api/examples/637e5678-30b1-45fa-acc4-877f2d636419/pdf"
});
const extractionId = response.extraction_id;
// Step 2: Apply schema separately
const schemaResult = await client.schema({
extraction_id: extractionId,
schema_config: {
input_schema: {
type: "object",
properties: {
total: { type: "number" },
vendor: { type: "string" }
}
},
schema_prompt: "Extract invoice total and vendor"
}
});
console.log(schemaResult.schema_output);
# Step 1: Extract the document
curl -X POST https://api.runpulse.com/extract \
-H "x-api-key: YOUR_API_KEY" \
-F "file=@invoice.pdf"
# Response includes extraction_id: "abc123-..."
# Step 2: Apply schema
curl -X POST https://api.runpulse.com/schema \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"extraction_id": "abc123-...",
"schema_config": {
"input_schema": {"type": "object", "properties": {"total": {"type": "number"}, "vendor": {"type": "string"}}},
"schema_prompt": "Extract invoice total and vendor"
}
}'
Document Metadata
Enableextensions.document_metadata to read native properties from the original
file before conversion, rendering, or OCR. The option is a single boolean; Pulse
returns every safely recoverable field for the detected format.
from pulse.types import ExtractRequestExtensions
response = client.extract(
file_url="https://example.com/report.pdf",
extensions=ExtractRequestExtensions(
document_metadata=True,
),
)
metadata = response.extensions.document_metadata
print(metadata.properties.title)
print(metadata.structure.page_count)
curl -X POST https://api.runpulse.com/extract \
-H "x-api-key: YOUR_API_KEY" \
-F "file=@report.pdf" \
-F 'extensions={"document_metadata":true};type=application/json'
{
"extensions": {
"document_metadata": {
"file": {
"name": "pulse-complex-metadata-10-page.pdf",
"extension": ".pdf",
"media_type": "application/pdf",
"size_bytes": 32506
},
"properties": {
"title": "Pulse Complex Metadata Validation Report",
"authors": ["Ritvik Pandey", "Pulse Document Intelligence"],
"created_at": "2026-01-15T09:30:00-08:00"
},
"structure": {
"page_count": 10,
"outline_count": 10,
"attachment_count": 1,
"annotation_count": 5,
"form_field_count": 3
},
"warnings": []
}
}
}
null. Metadata is
evidence declared by the source file and is not independently verified. Original
camera files may contain sensitive capture timestamps or GPS coordinates.
See Document Metadata for
format-specific behavior and implementation guidance.
Page Range and Chunking
from pulse.types import (
ExtractRequestExtensions,
ExtractRequestExtensionsChunking,
)
response = client.extract(
file_url="https://platform.runpulse.com/api/examples/637e5678-30b1-45fa-acc4-877f2d636419/pdf",
pages="1-5,10", # 1-indexed
extensions=ExtractRequestExtensions(
chunking=ExtractRequestExtensionsChunking(
chunk_types=["semantic", "page"],
chunk_size=1000,
),
),
)
# Chunk data is in extensions.chunking
print(response.extensions.chunking.semantic)
print(response.extensions.chunking.page)
const response = await client.extract({
fileUrl: "https://platform.runpulse.com/api/examples/637e5678-30b1-45fa-acc4-877f2d636419/pdf",
pages: "1-5,10", // 1-indexed
extensions: {
chunking: {
chunkTypes: ["semantic", "page"],
chunkSize: 1000
}
}
});
// Chunk data is in extensions.chunking
console.log(response.extensions?.chunking?.semantic);
console.log(response.extensions?.chunking?.page);
curl -X POST https://api.runpulse.com/extract \
-H "x-api-key: YOUR_API_KEY" \
-F "file=@document.pdf" \
-F "pages=1-5,10" \
-F 'extensions={"chunking": {"chunk_types": ["semantic", "page"], "chunk_size": 1000}}'
pages= subset, every page number in the response — page-break markers, bounding boxes, tables, Words, and extension output — refers to the original document’s page numbers. Requesting pages="10-20" returns items labeled pages 10 through 20.Footnote References
Enableextensions.footnote_references to detect footnote markers (e.g. *, †, 1) in body text and link them to their footnote text. Each result item carries the marker, the footnote’s text and location, the bounding-box IDs of the body blocks that cite it, and one entry per located marker occurrence with the marker’s own bounding box where available — enough to highlight both the footnote and every citation on the page.
from pulse.types import ExtractRequestExtensions
response = client.extract(
file_url="https://example.com/research-paper.pdf",
extensions=ExtractRequestExtensions(
footnote_references=True,
),
)
# Footnote links are in extensions.footnote_references
for ref in response.extensions.footnote_references:
print(f"Marker: {ref.symbol}")
print(f" Footnote: {ref.footnote_text_id}")
print(f" Referenced by: {ref.reference_text_ids}")
const response = await client.extract({
fileUrl: "https://example.com/research-paper.pdf",
extensions: {
footnoteReferences: true
}
});
// Footnote links are in extensions.footnoteReferences
for (const ref of response.extensions?.footnoteReferences ?? []) {
console.log(`Marker: ${ref.symbol}`);
console.log(` Footnote: ${ref.footnoteTextId}`);
console.log(` Referenced by: ${ref.referenceTextIds}`);
}
curl -X POST https://api.runpulse.com/extract \
-H "x-api-key: YOUR_API_KEY" \
-F "file=@research-paper.pdf" \
-F 'extensions={"footnote_references": true}'
Example Response
{
"markdown": "...",
"bounding_boxes": { ... },
"extensions": {
"footnoteReferences": [
{
"symbol": "*",
"footnoteTextId": "txt-11",
"footnotePageNumber": 1,
"footnoteText": "Equal contribution. Listing order is random.",
"referenceTextIds": ["txt-4", "txt-5"],
"references": [
{
"textId": "txt-4",
"pageNumber": 1,
"markerBoundingBox": [0.312, 0.221, 0.319, 0.221, 0.319, 0.231, 0.312, 0.231],
"source": "native"
},
{
"textId": "txt-5",
"pageNumber": 1,
"markerBoundingBox": null,
"source": "text"
}
]
},
{
"symbol": "4",
"footnoteTextId": "txt-48",
"footnotePageNumber": 3,
"footnoteText": "See Appendix B for the full derivation.",
"referenceTextIds": [],
"references": [
{
"tableId": "tbl-2",
"row": 3,
"column": 1,
"pageNumber": 3,
"markerBoundingBox": [0.61, 0.44, 0.617, 0.44, 0.617, 0.45, 0.61, 0.45],
"source": "words"
}
]
}
]
}
}
Footnote Reference Fields
| Field | Type | Description |
|---|---|---|
symbol | string | The footnote marker as written in the footnote text (e.g. *, †, ⁴, (1)). |
footnoteTextId | string | Bounding-box ID (e.g. txt-11) of the block holding the footnote text. Several footnotes can share one block when the extractor returned them together; symbol plus footnoteTextId is the unique key. |
footnotePageNumber | integer | 1-indexed page of the footnote text. |
footnoteText | string | The footnote’s own text with its leading marker removed. |
referenceTextIds | string[] | Bounding-box IDs of body-text blocks that contain a reference to this footnote. Cross-reference with bounding_boxes.Text to get each block’s content and position. Table cells are reported in references only. |
references | object[] | One entry per located marker occurrence, in reading order. Fields below. |
references[] entry:
| Field | Type | Description |
|---|---|---|
textId | string | Bounding-box ID of the citing block. Absent when the marker sits in a table cell. |
tableId | string | table_info.id of the table when the marker sits in a cell; row and column (zero-indexed) locate the cell. |
pageNumber | integer | 1-indexed page of the citation. |
markerBoundingBox | number[] or null | Polygon of the marker in the same normalized coordinates as bounding_box. null when only the block’s text located the reference. |
source | string | How the occurrence was located: native (from the PDF’s own text, most precise), words (from word positions on the page), or text (from the block’s content only). |
1, 2, 3), superscript and parenthesized forms (⁴, (1)), symbolic (*, †, ‡, §, ¶), and lettered (a, b, c) footnotes. Available for PDFs and images; text-based PDFs give the most precise marker positions.Excel Spreadsheet Options
from pulse import Pulse
from pulse.types import ExtractRequestSpreadsheet
client = Pulse(api_key="YOUR_API_KEY")
# Extract from Excel with hidden content included
response = client.extract(
file=open("financials.xlsx", "rb"),
spreadsheet=ExtractRequestSpreadsheet(
include_hidden_rows=True,
include_hidden_cols=True,
include_hidden_sheets=False,
),
)
print(response.markdown)
import { PulseClient } from 'pulse-ts-sdk';
const client = new PulseClient({
headers: { 'x-api-key': 'YOUR_API_KEY' }
});
const response = await client.extract({
file: fs.createReadStream("financials.xlsx"),
spreadsheet: {
includeHiddenRows: true,
includeHiddenCols: true,
includeHiddenSheets: false
}
});
console.log(response.markdown);
curl -X POST https://api.runpulse.com/extract \
-H "x-api-key: YOUR_API_KEY" \
-F "file=@financials.xlsx" \
-F 'spreadsheet={"include_hidden_rows": true, "include_hidden_cols": true, "include_hidden_sheets": false, "cell_data_mode": "inline"}'
bounding_boxes.Tables[].cell_data. The default spreadsheet.cell_data_mode: "inline" returns each full cell array. Set it to "external" to return per-table JSONL artifact references when very large workbooks make inline cell arrays impractical.spreadsheet.only_data_rows: true and spreadsheet.only_data_cols: true to have Pulse trim those trailing empty “phantom” rows and columns before parsing. Surviving cells keep their original A1 coordinates, so any citation or bounding box that references a specific cell remains stable. Both flags default to false. See the extraction options above for the full reference.Excel Charts and Embedded Images
When you setfigure_processing.show_images: true on an Excel workbook, every embedded chart and image is collected from the workbook directly and returned under bounding_boxes.Images[]. Each entry carries a Pulse-hosted image_url you can fetch via results.getImage (or any HTTP client with your API key) to get the raw PNG/JPEG bytes.
import re
from pulse import Pulse
from pulse.types import ExtractRequestFigureProcessing
client = Pulse(api_key="YOUR_API_KEY")
# 1) Extract the workbook with show_images enabled.
response = client.extract(
file=open("financials.xlsx", "rb"),
figure_processing=ExtractRequestFigureProcessing(
show_images=True,
description=False,
),
)
# 2) Walk the typed Images array.
for img in response.bounding_boxes.images or []:
print(f"{img.id}: {img.visual_type} '{img.chart_title}' @ {img.excel_range}")
print(f" url: {img.image_url}")
# 3) Fetch the bytes for one chart.
img = response.bounding_boxes.images[0]
m = re.search(r"/results/([^/]+)/images/([^/?#]+)", img.image_url)
job_id, filename = m.group(1), m.group(2)
chunks = list(client.results.get_image(job_id=job_id, filename=filename))
with open("chart.png", "wb") as f:
f.write(b"".join(chunks))
import { PulseClient } from "pulse-ts-sdk";
import * as fs from "node:fs";
const client = new PulseClient({ apiKey: "YOUR_API_KEY" });
// 1) Extract the workbook with show_images enabled.
const response = await client.extract({
file: fs.createReadStream("financials.xlsx"),
figureProcessing: { showImages: true, description: false },
});
// 2) Walk the typed Images array.
for (const img of response.boundingBoxes?.Images ?? []) {
console.log(
`${img.id}: ${img.visualType} '${img.chartTitle}' @ ${img.excelRange}`,
);
console.log(` url: ${img.imageUrl}`);
}
// 3) Fetch the bytes for one chart.
const url = response.boundingBoxes?.Images?.[0]?.imageUrl;
const m = url?.match(/\/results\/([^/]+)\/images\/([^/?#]+)/);
const [, jobId, filename] = m!;
const image = await client.results.getImage({ jobId, filename });
// Persist `image` per your runtime (e.g. `await image.bytes()`).
# Step 1: extract and capture an image_url from the response.
curl -sS -X POST https://api.runpulse.com/extract \
-H "x-api-key: YOUR_API_KEY" \
-F "file=@financials.xlsx" \
-F 'figure_processing={"show_images": true}' \
| jq -r '.bounding_boxes.Images[0].image_url'
# Step 2: fetch the PNG bytes.
curl -sS -X GET "https://api.runpulse.com/results/$JOB_ID/images/excel_image_1_1.png" \
-H "x-api-key: YOUR_API_KEY" \
-o chart.png
Example bounding_boxes.Images Entry
{
"id": "excel_image_1_1",
"visual_type": "chart",
"page_number": 1,
"bounding_box": [],
"image_url": "https://api.runpulse.com/results/13e3e75f-.../images/excel_image_1_1.png",
"sheet_name": "Charts",
"excel_range": "D2",
"chart_type": "BarChart",
"chart_title": "Revenue",
"source_ranges": ["'Charts'!$A$2:$A$5", "'Charts'!$B$2:$B$5"],
"description": "Bar chart showing revenue by quarter."
}
image_url.
Disable Storage
response = client.extract(
file_url="https://platform.runpulse.com/api/examples/637e5678-30b1-45fa-acc4-877f2d636419/pdf",
storage={"enabled": False}
)
const response = await client.extract({
fileUrl: "https://platform.runpulse.com/api/examples/637e5678-30b1-45fa-acc4-877f2d636419/pdf",
storage: { enabled: false }
});
curl -X POST https://api.runpulse.com/extract \
-H "x-api-key: YOUR_API_KEY" \
-F "file=@document.pdf" \
-F 'storage={"enabled": false}'
Authorizations
Body
Input schema for extraction requests. Provide either file (direct upload) or fileUrl (remote URL).
Document to upload directly. Required unless fileUrl is provided.
Public or pre-signed URL that Pulse will download and extract. Required unless file is provided.
Extraction model to use. pulse-ultra-2 runs Pulse Ultra 2; pulse-ultra-1 runs the previous-generation engine. If omitted or set to default, your organization's default backend is used (Pulse Ultra 2 for organizations upgraded to it), so send pulse-ultra-1 explicitly to keep a request on the previous engine.
default, pulse-ultra-1, pulse-ultra-2 Pulse Ultra 2 only. Enables a specialized selection-mark detection pass that improves selected/unselected state accuracy for forms, checkboxes, radio buttons, handwritten checkmarks, X marks, and similar controls. Enabled by default when model is pulse-ultra-2; set to false to skip this pass. Passing true without model: pulse-ultra-2 returns a validation error.
UUID of a saved extraction configuration (a "preset"). When provided, the server loads the saved configuration and applies its options on top of any inline parameters supplied in this request. Inline parameters always take precedence over preset values for the same field. Saved configs are managed via the platform UI or the input_extractions admin endpoints.
Page range filter supporting segments such as 1-2 or mixed ranges like 1-2,5.
^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$When true, return the complete extraction result as a URL even if it is small. Spreadsheet extractions use URL delivery by default; set force_url: false to request inline spreadsheet output. URL delivery changes only the transport, not the result shape.
Settings that control how figures and embedded visuals are processed. Applies to both PDFs/images (where figures are detected from layout) and spreadsheets (where charts and embedded images are read directly from the workbook). These options affect the markdown output and the bounding_boxes.Images[] array; they do not produce additional output fields elsewhere in the response.
Show child attributes
Show child attributes
Settings that enable additional processing passes or alternate output formats. Each enabled extension produces a corresponding output field under response.extensions.*.
Show child attributes
Show child attributes
Settings for Excel/spreadsheet extraction. Controls handling of hidden rows, columns, and sheets, whether numeric cells are rendered using their display format or underlying raw value, where table cell metadata is returned (inline or as external artifacts), whether cell formatting is included, and optional trimming of empty phantom rows/columns past the last data-bearing cell. Applies to .xlsx, .xlsm, and .xls files. Accepts both camelCase and snake_case field names; spreadsheet_options is accepted as a legacy alias for this object.
Show child attributes
Show child attributes
Options for persisting extraction artifacts. When enabled (default), artifacts are saved to storage and a database record is created.
Show child attributes
Show child attributes
If true, returns immediately with a job_id for polling via GET /job/{jobId}. Otherwise processes synchronously.
⚠️ DEPRECATED — Use the /schema endpoint after extraction instead. Pass the extraction_id from the extract response to /schema with your schema_config. This parameter still works for backward compatibility but will be removed in a future version.
Show child attributes
Show child attributes
(Deprecated) JSON schema describing structured data to extract. Use structuredOutput instead. Accepts either a JSON object or a stringified JSON representation.
(Deprecated) Natural language prompt for schema-guided extraction. Use structuredOutput.schemaPrompt instead.
(Deprecated) Custom instructions that augment the default extraction behaviour. Use figureProcessing or extensions instead.
⚠️ DEPRECATED — Use extensions.chunking.chunkTypes instead. Comma-separated list of chunking strategies to apply (for example semantic,header,page,recursive). Still accepted for backward compatibility.
⚠️ DEPRECATED — Use extensions.chunking.chunkSize instead. Override for maximum characters per chunk when chunking is enabled.
x >= 1⚠️ DEPRECATED — Toggle to enable figure extraction in results.
⚠️ DEPRECATED — Use figureProcessing.description instead. Toggle to generate descriptive captions for extracted figures.
⚠️ DEPRECATED — Use figureProcessing.showImages instead. Embed base64-encoded images inline in figure tags in the output. Increases response size.
⚠️ DEPRECATED — Use extensions.altOutputs.returnHtml instead. Whether to include HTML representation alongside markdown in the response.
(Deprecated) Enables expanded rationale output for debugging.
Response
Extraction result. For documents under 70 pages the full result is returned inline. For larger documents and spreadsheet extractions the response can contain is_url: true and a single-use url to download the full result via GET /results/{jobId}.
- Option 1
- Option 2
Full extraction result returned by the synchronous /extract endpoint. Inherits all core fields and adds deprecated backward-compatibility fields.
Primary markdown content extracted from the document. Always present in the new format.
Output from enabled extensions. Each key corresponds to an extension that was enabled in the request under extensions.*. Only keys for enabled extensions are present.
Show child attributes
Show child attributes
Positional bounding-box data for text, titles, headers, footers, images, and tables. Images carries chart/image visuals (with image_url when figure_processing.show_images is enabled), Tables the detected tables, and Text/Title/Footer the paragraph/title/footer regions. Additional keys (e.g. markdown_with_ids, defined_names) round-trip without being typed.
Show child attributes
Show child attributes
Persisted extraction ID. Present when storage is enabled (default). Use this ID with /split and /schema endpoints.
URL to view the extraction on the Pulse platform. Present when storage is enabled.
Number of pages processed.
x >= 1Billing tier and cumulative usage information. Includes total_credits_used (primary billing metric) and pages_used (legacy compatibility).
Show child attributes
Show child attributes
Non-fatal warnings generated during extraction. Includes deprecation notices when legacy input parameters are used, as well as processing warnings (e.g. word-level bounding box limitations).
Number of credits consumed by this request. Only present when the organization has the credit billing system enabled.
Deprecated — Use extensions.altOutputs.html instead. HTML representation of the extracted content. Present when the legacy returnHtml input was used.
Deprecated — Use extensions.chunking instead. Document content split into chunks. Present when the legacy chunking input was used.
Show child attributes
Show child attributes
Deprecated — Only present when the deprecated structuredOutput input parameter was used. Use the /schema endpoint after extraction instead.
Show child attributes
Show child attributes
Deprecated — Echo of the schema that was applied. Only present when the deprecated structuredOutput input parameter was used.
Deprecated — Error message if schema processing failed via the deprecated structuredOutput input parameter.
Deprecated — Alias for markdown. Included for backward compatibility with older SDK versions. Prefer markdown.
Deprecated — Identifier assigned to the extraction job. Retained for backward compatibility.
Deprecated — Additional metadata supplied by the backend. Retained for backward compatibility.