Extraction Basics
Extract text, metadata, and structure from 107 file formats — PDFs, Office documents, images, email, HTML, archives, and more. Single files or batches, from local paths or in-memory bytes, with per-document configuration overrides and built-in content filtering.
See the Configuration Reference for all extraction settings and the Supported Formats reference for format-specific options.
Entry Points
Section titled “Entry Points”Two extraction functions are the public entry points:
| Function | Input model | Purpose |
|---|---|---|
extract |
ExtractInput |
Extract one URI or in-memory byte payload |
extract_batch |
ExtractInput[] |
Extract multiple URI and byte inputs |
ExtractInput uses kind = "uri" for local paths, file:// URIs, and HTTP(S)
URLs. Use kind = "bytes" for in-memory payloads. extract and
extract_batch return an ExtractionResult envelope with results, errors,
summary, and optional crawl metadata.
Beyond local paths and bytes, HTTP(S) URIs are fetched and can be crawled — see
UrlExtractionConfig and
CrawlConfig. Embedded images and
their preprocessing are controlled by
ImageExtractionConfig.
Extract One Input
Section titled “Extract One Input”from xberg import ExtractInput, extract
output = await extract(ExtractInput(kind="uri", uri="document.pdf"))print(output.results[0].content)import { ExtractInputKind, extract } from "@xberg-io/xberg";
const output = await extract({ kind: ExtractInputKind.Uri, uri: "document.pdf",});console.log(output.results[0].content);use xberg::{extract, ExtractInput, ExtractionConfig};
let config = ExtractionConfig::default();let output = extract(ExtractInput::from_uri("document.pdf"), &config).await?;println!("{}", output.results[0].content);Read the Result
Section titled “Read the Result”Every entry in results carries the text plus the structures found in the document. Read the content, the tables, and the format metadata from the same document:
Tests URI extraction API
import asynciofrom xberg import extract, ExtractInput, ExtractInputKind
async def main() -> None: input = ExtractInput(kind=ExtractInputKind("uri"), uri="https://example.com/pdf/fake_memo.pdf") result = await extract(input) print(result.results[0].content)
asyncio.run(main())Tests URI extraction API
import { ExtractInput, ExtractInputKind, extract } from "@xberg-io/xberg";async function main() { const input: ExtractInput = { kind: ExtractInputKind.Uri, uri: "https://example.com/pdf/fake_memo.pdf" }; const result = await extract(input); console.log(result.results?.[0]?.content);}
void main();Tests URI extraction API
import { WasmExtractInput, WasmExtractInputKind, extract } from "@xberg-io/xberg-wasm";async function main() { const input: WasmExtractInput = (() => { const _u0 = WasmExtractInput.default(); _u0.kind = WasmExtractInputKind.Uri; _u0.uri = "https://example.com/pdf/fake_memo.pdf"; return _u0; })(); const result = await extract(input, undefined); console.log(result.results[0].content);}
void main();Tests URI extraction API
use xberg::extract;use xberg::ExtractInput;
#[tokio::main]async fn main() { let input_json: serde_json::Value = serde_json::from_str(r#"{"kind":"uri","uri":"https://example.com/pdf/fake_memo.pdf"}"#).unwrap(); let input = serde_json::from_value::<ExtractInput>(input_json).unwrap(); let config = Default::default(); let result = extract(input, &config).await.expect("call failed"); println!("{:?}", result.results[0].content);}Tests URI extraction API
package main
import ( "fmt" xberg "github.com/xberg-io/xberg/packages/go")
func ptr[T any](value T) *T { return &value }func main() { input := xberg.ExtractInput{ Kind: ptr(xberg.ExtractInputKindURI), URI: ptr(`https://example.com/pdf/fake_memo.pdf`), } config := xberg.ExtractionConfig{} result, err := xberg.Extract(input, config) if err != nil { panic(err) } fmt.Printf("%+v\n", result.Results[0].Content)}Tests URI extraction API
import io.xberg.*;
public final class Example { public static void main(String[] args) throws Exception { var inputJson = "{\"kind\":\"uri\",\"uri\":\"https://example.com/pdf/fake_memo.pdf\"}"; var input = JsonUtil.fromJson(inputJson, ExtractInput.class); var result = Xberg.extract(input, ExtractionConfig.builder().build()); System.out.println(result.results().get(0).content()); }}Tests URI extraction API
import io.xberg.*import com.fasterxml.jackson.module.kotlin.jacksonObjectMapper
fun main() = kotlinx.coroutines.runBlocking { val mapper = jacksonObjectMapper().setPropertyNamingStrategy(com.fasterxml.jackson.databind.PropertyNamingStrategies.SNAKE_CASE) val input = mapper.readValue("{\"kind\":\"uri\",\"uri\":\"https://example.com/pdf/fake_memo.pdf\"}", ExtractInput::class.java) val configDefault = mapper.readValue("{\"url\":{\"crawl\":{\"ssrf\":{}}}}", ExtractionConfig::class.java) val result = Xberg.extract(input, configDefault) println(result.results.first().content)}Tests URI extraction API
using System;using System.Text.Json;using Xberg;
var ConfigOptions = new JsonSerializerOptions { PropertyNameCaseInsensitive = true };var result = await XbergConverter.ExtractAsync(new ExtractInput { Kind = JsonSerializer.Deserialize<ExtractInputKind>("\"uri\"", ConfigOptions)!, Uri = "https://example.com/pdf/fake_memo.pdf" }, new ExtractionConfig());Console.WriteLine(result.Results[0].Content);Tests URI extraction API
import Xberg
let result = try await Xberg.extract("{\"kind\":\"uri\",\"uri\":\"https://example.com/pdf/fake_memo.pdf\"}", "{}")debugPrint(result.results()[0].content())Tests URI extraction API
require "xberg"result = Xberg.extract(Xberg::ExtractInput.new(kind: 'uri', uri: 'https://example.com/pdf/fake_memo.pdf'))puts result.results[0].content.inspectTests URI extraction API
<?php
declare(strict_types=1);
require_once __DIR__ . '/vendor/autoload.php';
use Xberg\Xberg;use Xberg\ExtractInput;$input = \Xberg\ExtractInput::from_json(json_encode(["kind" => "uri", "uri" => "https://example.com/pdf/fake_memo.pdf"]));$result = Xberg::extract($input, null);var_dump($result->getResults()[0]->content);Tests URI extraction API
input_value = %Xberg.ExtractInput{kind: "uri", uri: "https://example.com/pdf/fake_memo.pdf"}result = Xberg.extract_async(input_value)IO.inspect(Enum.at(result.results, 0).content)Tests URI extraction API
import 'dart:io';import 'package:xberg/xberg.dart';import 'package:xberg/src/xberg_bridge_generated/frb_generated.dart' show RustLib;Future<void> main() async { await RustLib.init(); try { final input = await createExtractInputFromJson(json: '{"kind":"uri","uri":"https://example.com/pdf/fake_memo.pdf"}'); final config = await createExtractionConfigFromJson(json: '{}'); final result = await XbergBridge.extract(input, config: config); stdout.writeln(result.results[0].content); } finally { RustLib.dispose(); }}Tests URI extraction API
const std = @import("std");const xberg = @import("xberg");
pub fn main() !void { const _result_json = try xberg.extract("{\"kind\":\"uri\",\"uri\":\"https://example.com/pdf/fake_memo.pdf\"}", "{}"); defer std.heap.c_allocator.free(_result_json); std.debug.print("{s}\n", .{_result_json});
}Tests URI extraction API
#include <assert.h>#include <stdint.h>#include <stdio.h>#include <stdlib.h>#include <string.h>#include "xberg.h"
int main(void) { XBERGAlefHandle input_handle = xberg_extract_input_from_json("{\"kind\":\"uri\",\"uri\":\"https://example.com/pdf/fake_memo.pdf\"}"); XBERGAlefHandle result = xberg_extract(input_handle, 0); xberg_extract_input_free(input_handle); xberg_extraction_result_free(result); return EXIT_SUCCESS;}Extract from Bytes
Section titled “Extract from Bytes”When content is already loaded in memory, pass bytes through ExtractInput
with an explicit MIME type.
from xberg import ExtractInput, extract
with open("document.pdf", "rb") as file: data = file.read()
output = await extract( ExtractInput( kind="bytes", bytes=data, mime_type="application/pdf", filename="document.pdf", ))import { readFile } from "node:fs/promises";import { ExtractInputKind, extract } from "@xberg-io/xberg";
const data = await readFile("document.pdf");const output = await extract({ kind: ExtractInputKind.Bytes, bytes: data, mimeType: "application/pdf", filename: "document.pdf",});use xberg::{extract, ExtractInput, ExtractionConfig};
let data = std::fs::read("document.pdf")?;let config = ExtractionConfig::default();let output = extract( ExtractInput::from_bytes(data, "application/pdf", Some("document.pdf".to_string())), &config,).await?;Browser File Input
Section titled “Browser File Input”The Wasm package has no filesystem access, so a browser upload always goes through bytes. Read the File from the input element and pass its bytes and MIME type:
import init, { WasmExtractInputKind, extract } from "@xberg-io/xberg-wasm";
async function setupFileInput() { await init();
const fileInput = document.getElementById("file-input") as HTMLInputElement;
fileInput.addEventListener("change", async (event) => { const file = (event.target as HTMLInputElement).files?.[0]; if (!file) return;
try { const bytes = new Uint8Array(await file.arrayBuffer()); const output = await extract({ kind: "bytes", bytes, mimeType: file.type || "application/octet-stream", filename: file.name, }, undefined);
console.log("Extracted text:", output.results[0].content); displayResults(output.results[0]); } catch (error) { console.error("Extraction failed:", error); } });}
function displayResults(result: any) { const output = document.getElementById("output"); if (output) { output.textContent = `${result.content?.substring(0, 500) ?? ""}...`; }}
setupFileInput().catch(console.error);Batch Processing
Section titled “Batch Processing”extract_batch accepts a list of ExtractInput values. Mix URI and byte inputs
in one request when a pipeline receives documents from multiple sources.
extract_batch over URI inputs
import asynciofrom xberg import extract_batch
async def main() -> None: inputs = [{"kind": "uri", "uri": "https://example.com/pdf/fake_memo.pdf"}, {"kind": "uri", "uri": "https://example.com/text/fake_text.txt"}] result = await extract_batch(inputs) for result in result.results: print(result.content)
asyncio.run(main())extract_batch over URI inputs
import { ExtractInput, ExtractInputKind, extractBatch } from "@xberg-io/xberg";async function main() { const result = await extractBatch([{ kind: ExtractInputKind.Uri, uri: "https://example.com/pdf/fake_memo.pdf" } as ExtractInput, { kind: ExtractInputKind.Uri, uri: "https://example.com/text/fake_text.txt" } as ExtractInput]); for (const item of result.results ?? []) { console.log(item.content); }}
void main();extract_batch over URI inputs
import { WasmExtractInput, WasmExtractInputKind, extractBatch } from "@xberg-io/xberg-wasm";async function main() { const result = await extractBatch([(() => { const _u0 = WasmExtractInput.default(); _u0.kind = WasmExtractInputKind.Uri; _u0.uri = "https://example.com/pdf/fake_memo.pdf"; return _u0; })(), (() => { const _u0 = WasmExtractInput.default(); _u0.kind = WasmExtractInputKind.Uri; _u0.uri = "https://example.com/text/fake_text.txt"; return _u0; })()], undefined); for (const item of result.results) { console.log(item.content); }}
void main();extract_batch over URI inputs
use xberg::extract_batch;use xberg::ExtractInput;
#[tokio::main]async fn main() { let inputs_json: serde_json::Value = serde_json::from_str(r#"[{"kind":"uri","uri":"https://example.com/pdf/fake_memo.pdf"},{"kind":"uri","uri":"https://example.com/text/fake_text.txt"}]"#).unwrap(); let inputs = serde_json::from_value::<Vec<ExtractInput>>(inputs_json).unwrap(); let config = Default::default(); let result = extract_batch(inputs, &config).await.expect("call failed"); for result in result.results.iter() { println!("{}", result.content); }}extract_batch over URI inputs
package main
import ( "encoding/json" "fmt" xberg "github.com/xberg-io/xberg/packages/go")
func main() { var inputs []xberg.ExtractInput if err := json.Unmarshal([]byte(`[{"kind":"uri","uri":"https://example.com/pdf/fake_memo.pdf"},{"kind":"uri","uri":"https://example.com/text/fake_text.txt"}]`), &inputs); err != nil { panic(fmt.Sprintf("config parse failed: %v", err)) } config := xberg.ExtractionConfig{} result, err := xberg.ExtractBatch(inputs, config) if err != nil { panic(err) } for _, result := range result.Results { fmt.Printf("%v\n", result.Content) }}extract_batch over URI inputs
import io.xberg.*;
public final class Example { public static void main(String[] args) throws Exception { var result = Xberg.extractBatch(java.util.Arrays.asList(JsonUtil.fromJson("{\"kind\":\"uri\",\"uri\":\"https://example.com/pdf/fake_memo.pdf\"}", ExtractInput.class), JsonUtil.fromJson("{\"kind\":\"uri\",\"uri\":\"https://example.com/text/fake_text.txt\"}", ExtractInput.class)), ExtractionConfig.builder().build()); for (var item : result.results()) { System.out.println(item.content()); } }}extract_batch over URI inputs
import io.xberg.*import com.fasterxml.jackson.module.kotlin.jacksonObjectMapper
fun main() = kotlinx.coroutines.runBlocking { val mapper = jacksonObjectMapper().setPropertyNamingStrategy(com.fasterxml.jackson.databind.PropertyNamingStrategies.SNAKE_CASE) val configDefault = mapper.readValue("{\"url\":{\"crawl\":{\"ssrf\":{}}}}", ExtractionConfig::class.java) val result = Xberg.extractBatch(listOf(mapper.readValue("{\"kind\":\"uri\",\"uri\":\"https://example.com/pdf/fake_memo.pdf\"}", ExtractInput::class.java), mapper.readValue("{\"kind\":\"uri\",\"uri\":\"https://example.com/text/fake_text.txt\"}", ExtractInput::class.java)), configDefault) for (result in result.results) { println(result.content) }}extract_batch over URI inputs
using System;using System.Text.Json;using Xberg;
var ConfigOptions = new JsonSerializerOptions { PropertyNameCaseInsensitive = true };var result = await XbergConverter.ExtractBatchAsync(new List<ExtractInput>() { JsonSerializer.Deserialize<ExtractInput>("{\"kind\":\"uri\",\"uri\":\"https://example.com/pdf/fake_memo.pdf\"}", ConfigOptions)!, JsonSerializer.Deserialize<ExtractInput>("{\"kind\":\"uri\",\"uri\":\"https://example.com/text/fake_text.txt\"}", ConfigOptions)! }, new ExtractionConfig());foreach (var resultItem in result.Results){ Console.WriteLine(resultItem.Content);}extract_batch over URI inputs
import Xberg
let _item_inputsArray_0 = try Xberg.extractInputFromJson("{\"kind\":\"uri\",\"uri\":\"https://example.com/pdf/fake_memo.pdf\"}")let _item_inputsArray_1 = try Xberg.extractInputFromJson("{\"kind\":\"uri\",\"uri\":\"https://example.com/text/fake_text.txt\"}")let inputsArray = [_item_inputsArray_0, _item_inputsArray_1]let configObj = try Xberg.extractionConfigFromJson("{}")let result = try await Xberg.extractBatch(inputs: inputsArray, config: configObj)for result in result.results() { print(result.content())}extract_batch over URI inputs
require "xberg"result = Xberg.extract_batch([{ 'kind' => 'uri', 'uri' => 'https://example.com/pdf/fake_memo.pdf' }, { 'kind' => 'uri', 'uri' => 'https://example.com/text/fake_text.txt' }])result.results.each do |result| puts result.contentendextract_batch over URI inputs
<?php
declare(strict_types=1);
require_once __DIR__ . '/vendor/autoload.php';
use Xberg\Xberg;use Xberg\ExtractInput;use Xberg\ExtractionConfig;$result = Xberg::extractBatch([ExtractInput::from_json('{"kind":"uri","uri":"https://example.com/pdf/fake_memo.pdf"}'), ExtractInput::from_json('{"kind":"uri","uri":"https://example.com/text/fake_text.txt"}')], \Xberg\ExtractionConfig::from_json('{}'));foreach ($result->getResults() as $result) { echo $result->getContent(), PHP_EOL;}extract_batch over URI inputs
result = Xberg.extract_batch_async([%{"kind" => "uri", "uri" => "https://example.com/pdf/fake_memo.pdf"}, %{"kind" => "uri", "uri" => "https://example.com/text/fake_text.txt"}])Enum.each(result.results, fn result -> IO.puts(result.content)end)extract_batch over URI inputs
import 'dart:convert';import 'dart:io';import 'package:xberg/xberg.dart';import 'package:xberg/src/xberg_bridge_generated/frb_generated.dart' show RustLib;Future<void> main() async { await RustLib.init(); try { final inputs = await Future.wait((jsonDecode(r'[{"kind":"uri","uri":"https://example.com/pdf/fake_memo.pdf"},{"kind":"uri","uri":"https://example.com/text/fake_text.txt"}]') as List<dynamic>).map((element) => createExtractInputFromJson(json: jsonEncode(element)))); final result = await XbergBridge.extractBatch(inputs); for (final result in result.results) { stdout.writeln(result.content); } } finally { RustLib.dispose(); }}extract_batch over URI inputs
const std = @import("std");const xberg = @import("xberg");
pub fn main() !void { const _result_json = try xberg.extract_batch("[{\"kind\":\"uri\",\"uri\":\"https://example.com/pdf/fake_memo.pdf\"},{\"kind\":\"uri\",\"uri\":\"https://example.com/text/fake_text.txt\"}]", "{}"); defer std.heap.c_allocator.free(_result_json); std.debug.print("{s}\n", .{_result_json});
}extract_batch over URI inputs
#include <assert.h>#include <stdint.h>#include <stdio.h>#include <stdlib.h>#include <string.h>#include "xberg.h"
int main(void) { XBERGAlefHandle result = xberg_extract_batch("[{\"kind\":\"uri\",\"uri\":\"https://example.com/pdf/fake_memo.pdf\"},{\"kind\":\"uri\",\"uri\":\"https://example.com/text/fake_text.txt\"}]", 0); xberg_extraction_result_free(result); return EXIT_SUCCESS;}Per-Input Configuration
Section titled “Per-Input Configuration”When a batch contains a mix of document types that need different settings,
attach per-input overrides to ExtractInput while sharing a common batch config.
from xberg import ( ExtractionConfig, ExtractInput, FileExtractionConfig, extract_batch,)
config = ExtractionConfig(output_format="markdown")
inputs = [ ExtractInput(kind="uri", uri="report.pdf"), ExtractInput( kind="uri", uri="scan.tiff", config=FileExtractionConfig(force_ocr=True), ), ExtractInput( kind="uri", uri="notes.html", config=FileExtractionConfig(output_format="plain"), ),]
output = await extract_batch(inputs, config)import { ExtractInputKind, extractBatch } from "@xberg-io/xberg";
const output = await extractBatch( [ { kind: ExtractInputKind.Uri, uri: "report.pdf" }, { kind: ExtractInputKind.Uri, uri: "scan.tiff", config: { forceOcr: true }, }, { kind: ExtractInputKind.Uri, uri: "notes.html", config: { outputFormat: "plain" }, }, ], { outputFormat: "markdown" },);use xberg::{ extract_batch, ExtractInput, ExtractInputKind, ExtractionConfig, FileExtractionConfig, OutputFormat,};
let config = ExtractionConfig { output_format: OutputFormat::Markdown, ..Default::default()};
let inputs = vec![ ExtractInput::from_uri("report.pdf"), ExtractInput { kind: ExtractInputKind::Uri, uri: Some("scan.tiff".to_string()), config: Some(FileExtractionConfig { force_ocr: Some(true), ..Default::default() }), ..Default::default() }, ExtractInput { kind: ExtractInputKind::Uri, uri: Some("notes.html".to_string()), config: Some(FileExtractionConfig { output_format: Some(OutputFormat::Plain), ..Default::default() }), ..Default::default() },];
let output = extract_batch(inputs, &config).await?;Fields set to None in FileExtractionConfig inherit the batch default.
Batch-level concerns like max_concurrent_extractions, use_cache, and
security_limits cannot be overridden per input. See the
Configuration Reference
for the full list of overridable fields.
Archive and XML Bomb Protections
Section titled “Archive and XML Bomb Protections”Every archive-bearing format (.zip, .docx, .pptx, .xlsx, .odt, .ods,
.odp, .epub, .hwpx, and the iWork formats .pages/.key/.numbers) and
every XML-bearing format is checked against SecurityLimits before or during
parsing. These checks exist to stop a hostile input from exhausting memory or
CPU rather than to validate document correctness — a legitimate large document
should never come close to the defaults below.
| Limit | Defends against | Default |
|---|---|---|
max_archive_size |
ZIP bombs — declared uncompressed size accumulated across entries | 500 MiB |
max_compression_ratio |
ZIP bombs — a single entry or the archive total expanding beyond a sane ratio | 100:1 |
max_files_in_archive |
Archives with an unreasonable number of member files | 10,000 |
max_nesting_depth |
Deeply nested containers (nested archives, iWork protobuf messages) | 1,024 levels |
max_xml_depth |
Deeply nested XML elements | 1,024 levels |
max_entity_length |
Billion-laughs-class attacks — a single XML entity/attribute/token expanding to hundreds of MB | 1 MiB |
max_content_size |
Aggregate text growth across a whole document (catches long-tail expansion that max_entity_length alone would miss) |
100 MB |
max_iterations |
Infinite or near-infinite parser loops (XML token loop, HTML tokenizer, JSON parser) | 10,000,000 |
max_table_cells |
Documents claiming an unreasonable aggregate number of table cells (CSV, XLSX, HTML tables) | 100,000 |
max_pages |
Excessive per-page OCR, layout, and rendering work | Unlimited |
When both max_nesting_depth and max_xml_depth apply to the same parse, the
tighter of the two wins — lowering either one alone is enough to clamp nesting.
Archive-specific checks (max_archive_size, max_compression_ratio,
max_files_in_archive) are enforced by ZipBombValidator against the ZIP
central directory before any entry is decompressed, so an oversized or
over-compressed archive is rejected without ever running the decompressor.
Configuring limits
Section titled “Configuring limits”SecurityLimits is set on ExtractionConfig.security_limits and applies to
the whole extraction (it cannot be overridden per file in a batch).
from xberg import ExtractInput, ExtractionConfig, SecurityLimits, extract
config = ExtractionConfig( security_limits=SecurityLimits( max_archive_size=100 * 1024 * 1024, # 100 MiB max_files_in_archive=1_000, max_pages=250, ), max_embedded_file_bytes=20 * 1024 * 1024, extraction_timeout_secs=120,)
output = await extract(ExtractInput(kind="uri", uri="archive.zip"), config=config)import { ExtractInputKind, extract } from "@xberg-io/xberg";
const output = await extract( { kind: ExtractInputKind.Uri, uri: "archive.zip" }, { securityLimits: { maxArchiveSize: 100 * 1024 * 1024, // 100 MiB maxFilesInArchive: 1_000, maxPages: 250, }, maxEmbeddedFileBytes: 20 * 1024 * 1024, extractionTimeoutSecs: 120, },);use xberg::{extract, ExtractInput, ExtractionConfig, SecurityLimits};
let config = ExtractionConfig { security_limits: Some(SecurityLimits { max_archive_size: 100 * 1024 * 1024, // 100 MiB max_files_in_archive: 1_000, max_pages: Some(250), ..Default::default() }), max_embedded_file_bytes: Some(20 * 1024 * 1024), extraction_timeout_secs: Some(120), ..Default::default()};
let output = extract(ExtractInput::from_uri("archive.zip"), &config).await?;import { ExtractionConfig, SecurityLimits } from "@xberg-io/xberg-wasm";
const config = new ExtractionConfig();config.securityLimits = new SecurityLimits( 100 * 1024 * 1024, // maxArchiveSize undefined, // maxCompressionRatio 1_000, // maxFilesInArchive // ...remaining fields fall back to defaults when undefined);// Build a SecurityLimits handle from JSON and read a field back.XBERGSecurityLimits *limits = xberg_security_limits_from_json( "{\"max_archive_size\":104857600,\"max_files_in_archive\":1000}");size_t max_files = xberg_security_limits_max_files_in_archive(limits);xberg_security_limits_free(limits);Any field left unset falls back to the SecurityLimits::default() value shown
in the table above.
The max_table_cells default is intentionally conservative. For a trusted
large CSV, XLSX, or HTML input, raise security_limits.max_table_cells to a
known workload bound; doing so permits proportionally more parsing work and
output allocation. Limit errors report both the observed cell count and the
configured value.
max_pages is enforced before per-page work for PDF, PPTX, Keynote, ODP, and
multi-frame TIFF when OCR is enabled. It is not a universal document-page cap:
DOCX, ODT, XLSX, legacy Office files, Pages, Numbers, and TIFF without OCR do
not expose a reliable page count before extraction. Keep the byte, archive,
embedded-file, and timeout limits enabled even when you set max_pages.
PHP note: construct SecurityLimits, then apply it with
ExtractionConfig::setSecurityLimits(); the generated constructor does not
take securityLimits directly.
What happens when a limit is hit
Section titled “What happens when a limit is hit”All of these checks raise the same error family:
- Rust —
XbergError::Security { message, source }, wheresourceis a boxedSecurityError(ZipBombDetected,ArchiveTooLarge,TooManyFiles,NestingTooDeep,ContentTooLarge,EntityTooLong,TooManyIterations,XmlDepthExceeded,TooManyCells, orUnreadableEntryfor an archive entry whose header could not be read at all). - Python — a
xberg.SecurityErrorexception (subclass ofxberg.XbergError). - Other bindings — an error/result whose kind/name is
"Security"(or the language’s equivalent typed error), carrying the same human-readable message.
The extraction is aborted for that input; it does not silently truncate or return partial content for a security violation the way a format-parsing warning would.
PDF Reading Order Repair
Section titled “PDF Reading Order Repair”Native PDF text extraction reads glyphs in the order they were written into the content stream, not the order a human reads the page. For multi-column layouts (academic papers, magazines, dense reports) that order can jump between columns mid-sentence. Xberg can repair this by projecting text spans onto layout-detected regions, grouping them into columns, and re-emitting them top-to-bottom within each column, left-to-right across columns.
This repair is off by default. Enable it with
ExtractionConfig.pdf_options.reading_order = true
(PdfConfig::reading_order in Rust, default false).
from xberg import ExtractInput, ExtractionConfig, PdfOptions, extract
config = ExtractionConfig( pdf_options=PdfOptions(reading_order=True),)
output = await extract( ExtractInput(kind="uri", uri="two_column_paper.pdf"), config=config,)import { ExtractInputKind, extract } from "@xberg-io/xberg";
const output = await extract( { kind: ExtractInputKind.Uri, uri: "two_column_paper.pdf" }, { pdfOptions: { readingOrder: true, }, },);use xberg::{extract, ExtractInput, ExtractionConfig, PdfConfig};
let config = ExtractionConfig { pdf_options: Some(PdfConfig { reading_order: true, ..Default::default() }), ..Default::default()};
let output = extract(ExtractInput::from_uri("two_column_paper.pdf"), &config).await?;Requirements: this option only takes effect when the layout-detection
feature is compiled in and layout hints were produced for the page (the page
must have a layout model pass run over it). Without layout hints, the flag is
a no-op and extraction falls back silently to native text order.
What it fixes: the flowing body text of multi-column pages — the case where native extraction interleaves column A and column B mid-paragraph.
What it does not fix: reading order only reorders text spans, not table
cells. A page containing a rotated or scrambled table is unaffected by this
option — that class of problem lives in the table extraction path, not the
span-reordering pass, and setting reading_order will not repair it.
Cost: reordering only runs when layout hints are already available, so it adds span-projection and column-detection work on top of an existing layout-detection pass rather than triggering a new one on its own. There is no published benchmark number for the added latency; measure it against your own corpus before enabling it in a latency-sensitive pipeline.
Content Filtering
Section titled “Content Filtering”Xberg strips running headers, footers, watermarks, and cross-page repeating text
by default so downstream RAG and LLM pipelines see clean body content.
ContentFilterConfig lets you opt back in when those regions carry useful text.
By default headers, footers, and watermarks are stripped and cross-page repeating text is deduplicated; see ContentFilterConfig for field-level defaults and per-format behavior.
from xberg import ( ContentFilterConfig, ExtractionConfig, ExtractInput, extract,)
config = ExtractionConfig( content_filter=ContentFilterConfig( include_headers=True, include_footers=True, ),)
output = await extract( ExtractInput(kind="uri", uri="contract.pdf"), config=config,)import { ExtractInputKind, extract } from "@xberg-io/xberg";
const output = await extract( { kind: ExtractInputKind.Uri, uri: "brochure.pdf" }, { contentFilter: { stripRepeatingText: false, }, },);use xberg::{extract, ContentFilterConfig, ExtractInput, ExtractionConfig};
let config = ExtractionConfig { content_filter: Some(ContentFilterConfig { include_headers: true, include_footers: true, strip_repeating_text: true, include_watermarks: false, ..Default::default() }), ..Default::default()};
let output = extract(ExtractInput::from_uri("contract.pdf"), &config).await?;When a layout-detection model is active, it can independently classify regions
as page headers or footers and strip them per page. Setting
include_headers=True / include_footers=True also disables that per-page
stripping. See the
reference page for the full
field semantics and per-format behavior.
Jupyter Notebook Cells
Section titled “Jupyter Notebook Cells”Choose whether .ipynb extraction includes code-cell source, saved outputs, or
both. Xberg never executes notebook cells; output modes expose only data already
stored in the notebook. Markdown cells and structural metadata are unaffected.
from xberg import ExtractionConfig, JupyterCellRendering
config = ExtractionConfig(jupyter_cell_rendering=JupyterCellRendering.SOURCE)use xberg::{ExtractionConfig, JupyterCellRendering};
let config = ExtractionConfig { jupyter_cell_rendering: JupyterCellRendering::Source, ..Default::default()};xberg extract notebook.ipynb --jupyter-cell-rendering sourceUse source, outputs, or both (the default).
Supported Formats
Section titled “Supported Formats”Xberg supports 107 file formats across 140 unique file extensions and accepts 53 compatibility MIME aliases:
| Category | Example extensions | Notes |
|---|---|---|
.pdf |
Native text + OCR for scanned pages | |
| Images | .png, .jpg, .jpeg, .tiff, .bmp, .webp, .heic, .heics, .heif, .heifs, .hif, .avif |
OCR backend; HEIC/HEIF/AVIF need heic feature + libheif |
| Office | .docx, .pptx, .xlsx, .odt, .ods, .odp |
Modern + OpenDocument via native parsers |
| Legacy Office | .doc, .ppt, .pps, .wpd, .wp, .wp5, .wp6 |
Native OLE/CFB parsing; WordPerfect via libwpd |
.eml, .msg, .pst |
Full support including attachments | |
| Web | .html, .htm, .xhtml, .xht |
Converted to Markdown with metadata |
| Text and data | .md, .txt, .xml, .json, .geojson, .kml, .yaml, .toml, .csv, .sqlite, .sqlite3, .db, .gpkg, .gpkx |
Direct, geospatial, and bounded database extraction |
| Archives | .zip, .tar, .tgz, .gz, .7z |
Recursive extraction |
Image metadata and EXIF
Section titled “Image metadata and EXIF”For every supported image format — JPEG, PNG, TIFF, WebP, BMP, GIF, JPEG 2000,
HEIC, HEIF, AVIF — Xberg returns an ImageMetadata block on
metadata.format containing:
width/heightin pixelsformat— uppercase format tag (e.g.JPEG,PNG,HEIF)exif— a key/value map of EXIF tags
EXIF extraction is powered by the pure-Rust nom-exif integration and covers
camera identity (Make, Model, LensModel, LensSpecification, Software),
timestamps (DateTimeOriginal, CreateDate, OffsetTime, SubSecTime), full
exposure parameters (ExposureTime, FNumber, ISO, ApertureValue,
ShutterSpeedValue, ExposureProgram, ExposureMode, MeteringMode, Flash,
SceneCaptureType), the complete GPS block (GPSLatitude, GPSLongitude,
GPSAltitude, GPSTimeStamp, GPSDateStamp, GPSSpeed, GPSImgDirection,
GPSMapDatum, GPSProcessingMethod), color space, thumbnail offsets, and
provenance fields (Copyright, ImageDescription, ImageUniqueID).
EXIF works on every target, including wasm-target and android-target,
because nom-exif is pure Rust. HEIC / HEIF / AVIF pixel decoding requires
the heic Cargo feature and the system libheif library, and is therefore
native-only — see the installation guide.
When the heic feature is enabled, HEIC / HEIF / AVIF inputs are decoded to
RGBA via libheif, re-encoded as PNG, and then flow through the standard
OCR / layout pipeline. EXIF is read from the original HEIC bytes before the
PNG re-encode so no metadata is lost.
Page Tracking
Section titled “Page Tracking”Xberg can track page boundaries and extract per-page content. Page tracking availability depends on the format:
- PDF — Full byte-accurate page tracking with O(1) lookup
- PPTX — Slide boundary tracking (each slide = one page)
- DOCX — Best-effort detection using explicit
<w:br type="page"/>tags - Other formats — No page tracking
Enable page extraction with PageConfig:
config = ExtractionConfig( pages=PageConfig( insert_page_markers=True, marker_format="\n\n<!-- PAGE {page_num} -->\n\n" ))Page markers like <!-- PAGE 1 --> are inserted at boundaries in the content field — useful for LLMs that need to understand document layout. When both page tracking and chunking are enabled, chunks automatically include first_page and last_page metadata.
See PageConfig Reference for all options and Chunking for chunk-to-page mapping examples.
Code File Extraction
Section titled “Code File Extraction”Source code files (.py, .rs, .ts, .go, etc.) go through tree-sitter and produce a ProcessResult on ExtractedDocument.code_intelligence (structure, imports/exports, symbols, docstrings, diagnostics, semantic chunks). Code files bypass text chunking — TSLP’s function/class-aware CodeChunks map directly to Xberg Chunks with semantic chunk_type and heading context.
See Code Intelligence for usage and TreeSitterProcessConfig for fields.
PDF Page Rendering
Section titled “PDF Page Rendering”Render individual PDF pages as PNG images. Unlike the extraction pipeline (which parses text, tables, metadata), this API produces raw pixel data for thumbnails, vision model input, or custom OCR pipelines. It is exposed as pure-Rust functions on the core crate.
Functions
Section titled “Functions”| Function | Purpose |
|---|---|
render_pdf_page_to_png |
Render one zero-based page index to PNG bytes at a given DPI |
pdf_page_count |
Read the page count without rasterizing, to drive a render loop over pages |
Render a single page, or count first and loop to process every page without holding all images in memory:
use xberg::{pdf_page_count, render_pdf_page_to_png};
let pdf_bytes = std::fs::read("document.pdf")?;
// Render one specific page (zero-based) at 300 DPI, no password.let png = render_pdf_page_to_png(&pdf_bytes, 0, Some(300), None)?;std::fs::write("page-0.png", &png)?;
// Or count pages and render each in turn.let count = pdf_page_count(&pdf_bytes, None)?;for page_index in 0..count { let png = render_pdf_page_to_png(&pdf_bytes, page_index, Some(150), None)?; std::fs::write(format!("page-{page_index}.png"), &png)?;}dpi defaults to 150 when passed None. password unlocks encrypted PDFs.
DPI Configuration
Section titled “DPI Configuration”| DPI | Pixel size (US Letter) | Use case |
|---|---|---|
| 72 | 612 x 792 | Thumbnails, quick previews |
| 150 (default) | 1275 x 1650 | General-purpose, screen display |
| 300 | 2550 x 3300 | OCR input, print quality |
Tip: Use 300 DPI when rendering pages for OCR or vision models. The default 150 DPI may reduce recognition accuracy on small text.
MIME Type Detection
Section titled “MIME Type Detection”Xberg prefers bounded content inspection and falls back to a supported filename extension. Files with unknown or
missing extensions are sniffed instead of being rejected. A specific explicit MIME type remains authoritative;
application/octet-stream is a generic placeholder that triggers configured detection.
Set ExtractionConfig.mime_detection_policy to prefer_content (the default), trust_extension, or content_only.
trust_extension skips content sniffing when the filename has a supported extension, so use it only when filenames
come from a trusted source. content_only ignores the filename extension. A FileExtractionConfig override can select
a different policy for one batch item.
Example: Override MIME Type
Section titled “Example: Override MIME Type”from xberg import ExtractInput, extract
# File without extension — provide MIME type explicitlyresult = await extract( ExtractInput( kind="uri", uri="document_copy", mime_type="application/pdf", ), config=config,)Error Handling
Section titled “Error Handling”Extraction failures use each language’s typed error surface: exceptions in
exception-based bindings, Result in Rust, and error values or status codes in
other targets. This example handles an unsupported MIME type:
Error when extracting with unsupported MIME type
import asynciofrom pathlib import Pathfrom xberg import extract, ExtractInput, FileExtractionConfig, ExtractInputKindfrom xberg._xberg import ExtractionConfigfrom xberg import XbergError
async def main() -> None: try: input = ExtractInput(bytes=Path("text/plain.txt").read_bytes(), config=FileExtractionConfig(), filename="plain.txt", kind=ExtractInputKind("bytes"), mime_type="application/x-nonexistent") config = ExtractionConfig.from_json("{}") await extract(input, config) except XbergError as error: print(f"{type(error).__name__}: {error}")
asyncio.run(main())Error when extracting with unsupported MIME type
import { ExtractInput, ExtractInputKind, extract } from "@xberg-io/xberg";async function main() { const input: ExtractInput = { bytes: await (await import("node:fs/promises")).readFile("text/plain.txt"), config: { }, filename: "plain.txt", kind: ExtractInputKind.Bytes, mimeType: "application/x-nonexistent" }; try { await extract(input); } catch (error) { if (error instanceof Error) { console.error(`${error.name}: ${error.message}`); } }}
void main();Error when extracting with unsupported MIME type
import { WasmExtractInput, WasmExtractInputKind, WasmFileExtractionConfig, extract } from "@xberg-io/xberg-wasm";async function main() { const input: WasmExtractInput = await (async () => { const _u0 = WasmExtractInput.default(); _u0.bytes = await (await import("node:fs/promises")).readFile("text/plain.txt"); _u0.config = await (async () => { const _u1 = WasmFileExtractionConfig.default(); return _u1; })(); _u0.filename = "plain.txt"; _u0.kind = WasmExtractInputKind.Bytes; _u0.mimeType = "application/x-nonexistent"; return _u0; })(); try { await extract(input, { }); } catch (error) { console.error(String(error)); }}
void main();Error when extracting with unsupported MIME type
use xberg::extract;use xberg::ExtractInput;
#[tokio::main]async fn main() { let mut input_json: serde_json::Value = serde_json::from_str(r#"{"bytes":"text/plain.txt","config":{},"filename":"plain.txt","kind":"bytes","mime_type":"application/x-nonexistent"}"#).unwrap(); let input_file_0 = std::fs::read(r#"text/plain.txt"#).expect("file read failed"); *input_json.pointer_mut(r#"/bytes"#).expect("docs file field missing") = serde_json::json!(input_file_0); let input = serde_json::from_value::<ExtractInput>(input_json).unwrap(); let config_json: serde_json::Value = serde_json::from_str(r#"{}"#).unwrap(); let config = serde_json::from_value(config_json).unwrap(); let result = extract(input, &config).await; match result { Ok(value) => println!("{:?}", value), Err(error) => println!("{error}"), }}Error when extracting with unsupported MIME type
package main
import ( "errors" "fmt" xberg "github.com/xberg-io/xberg/packages/go" "os")
func ptr[T any](value T) *T { return &value }func mustReadFile(path string) []byte { content, err := os.ReadFile(path) if err != nil { panic(err) } return content}func main() { input := xberg.ExtractInput{ Kind: ptr(xberg.ExtractInputKindBytes), Bytes: mustReadFile(`text/plain.txt`), MimeType: ptr(`application/x-nonexistent`), Filename: ptr(`plain.txt`), Config: &xberg.FileExtractionConfig{}, } config := xberg.ExtractionConfig{} _, err := xberg.Extract(input, config) var typedError xberg.Error if errors.As(err, &typedError) { fmt.Fprintf(os.Stderr, "%T: %v\n", typedError, typedError) }}Error when extracting with unsupported MIME type
import io.xberg.*;
public final class Example { public static void main(String[] args) throws Exception { try { var inputFile0 = java.util.Base64.getEncoder().encodeToString( java.nio.file.Files.readAllBytes(java.nio.file.Path.of("text/plain.txt")) ); var inputJson = "{\"bytes\":\"__ALEF_DOC_FILE_0__\",\"config\":{},\"filename\":\"plain.txt\",\"kind\":\"bytes\",\"mime_type\":\"application/x-nonexistent\"}"; inputJson = inputJson.replace("__ALEF_DOC_FILE_0__", inputFile0); var input = JsonUtil.fromJson(inputJson, ExtractInput.class); var configJson = "{}"; var config = JsonUtil.fromJson(configJson, ExtractionConfig.class); var result = Xberg.extract(input, config); System.out.println(result); } catch (XbergRsException error) { System.err.println(error.getClass().getSimpleName() + ": " + error.getMessage()); } }}Error when extracting with unsupported MIME type
import io.xberg.*import com.fasterxml.jackson.module.kotlin.jacksonObjectMapper
fun main() = kotlinx.coroutines.runBlocking { val mapper = jacksonObjectMapper().setPropertyNamingStrategy(com.fasterxml.jackson.databind.PropertyNamingStrategies.SNAKE_CASE) try { val inputFile0 = java.util.Base64.getEncoder().encodeToString(java.nio.file.Files.readAllBytes(java.nio.file.Path.of("text/plain.txt"))) val input = mapper.readValue("{\"bytes\":\"__ALEF_DOC_FILE_0__\",\"config\":{},\"filename\":\"plain.txt\",\"kind\":\"bytes\",\"mime_type\":\"application/x-nonexistent\"}".replace("__ALEF_DOC_FILE_0__", inputFile0), ExtractInput::class.java) val config = mapper.readValue("{\"url\":{\"crawl\":{\"ssrf\":{}}}}", ExtractionConfig::class.java) val result = Xberg.extract(input, config) } catch (error: Exception) { System.err.println("${error::class.simpleName}: ${error.message}") }}Error when extracting with unsupported MIME type
using System;using System.Text.Json;using Xberg;
var ConfigOptions = new JsonSerializerOptions { PropertyNameCaseInsensitive = true };try{var result = await XbergConverter.ExtractAsync(new ExtractInput { Bytes = System.IO.File.ReadAllBytes("text/plain.txt"), Config = new FileExtractionConfig(), Filename = "plain.txt", Kind = JsonSerializer.Deserialize<ExtractInputKind>("\"bytes\"", ConfigOptions)!, MimeType = "application/x-nonexistent" }, new ExtractionConfig());}catch (Exception error){ Console.Error.WriteLine($"{error.GetType().Name}: {error.Message}");}Error when extracting with unsupported MIME type
import Xberg
do { _ = try await Xberg.extract("{\"bytes\":\"text/plain.txt\",\"config\":{},\"filename\":\"plain.txt\",\"kind\":\"bytes\",\"mime_type\":\"application/x-nonexistent\"}", "{}")} catch { print("\(type(of: error)): \(error)")}Error when extracting with unsupported MIME type
require "xberg"begin result = Xberg.extract(Xberg::ExtractInput.new(bytes: File.binread('text/plain.txt').bytes, config: { }, filename: 'plain.txt', kind: 'bytes', mime_type: 'application/x-nonexistent'), { })rescue StandardError => error warn "#{error.class}: #{error.message}"endError when extracting with unsupported MIME type
<?php
declare(strict_types=1);
require_once __DIR__ . '/vendor/autoload.php';
use Xberg\Xberg;use Xberg\ExtractInput;$input = \Xberg\ExtractInput::from_json(json_encode(["bytes" => "text/plain.txt", "config" => [], "filename" => "plain.txt", "kind" => "bytes", "mimeType" => "application/x-nonexistent"]));try { Xberg::extract($input, []);} catch (Throwable $error) { echo $error::class . ': ' . $error->getMessage() . "\n";}Error when extracting with unsupported MIME type
try do input_value = %Xberg.ExtractInput{bytes: :binary.bin_to_list(File.read!("text/plain.txt")), config: %{}, filename: "plain.txt", kind: "bytes", mime_type: "application/x-nonexistent"} result = Xberg.extract_async(input_value, "{}")rescue error -> IO.puts(:stderr, "#{inspect(error.__struct__)}: #{Exception.message(error)}")endError when extracting with unsupported MIME type
import 'dart:io';import 'package:xberg/xberg.dart';import 'package:xberg/src/xberg_bridge_generated/frb_generated.dart' show RustLib;Future<void> main() async { await RustLib.init(); try { try { final input = await createExtractInputFromJson(json: '{"bytes":"text/plain.txt","config":{},"filename":"plain.txt","kind":"bytes","mime_type":"application/x-nonexistent"}'); final config = await createExtractionConfigFromJson(json: '{}'); final result = await XbergBridge.extract(input, config: config); stdout.writeln(result); } on XbergError catch (error) { stderr.writeln('${error.runtimeType}: $error'); } } finally { RustLib.dispose(); }}Error when extracting with unsupported MIME type
const std = @import("std");const xberg = @import("xberg");
pub fn main() !void { var gpa: std.heap.DebugAllocator(.{}) = .init; defer _ = gpa.deinit(); const allocator = gpa.allocator();
var input_file_0_threaded = std.Io.Threaded.init(allocator, .{});defer input_file_0_threaded.deinit();const input_file_0_io = input_file_0_threaded.io();const input_file_0 = try std.Io.Dir.cwd().readFileAlloc(input_file_0_io, "text/plain.txt", allocator, .unlimited);defer allocator.free(input_file_0); const input_file_0_json = try std.json.Stringify.valueAlloc(allocator, input_file_0, .{ .emit_strings_as_arrays = true });defer allocator.free(input_file_0_json); const input_json_0 = try std.mem.replaceOwned(u8, allocator, "{\"bytes\":\"__ALEF_DOC_FILE_0__\",\"config\":{},\"filename\":\"plain.txt\",\"kind\":\"bytes\",\"mime_type\":\"application/x-nonexistent\"}", "\"__ALEF_DOC_FILE_0__\"", input_file_0_json);defer allocator.free(input_json_0); if (xberg.extract(input_json_0, "{}")) |_| { return error.TestUnexpectedResult; } else |err| { std.debug.print("call failed as expected: {s}\n", .{@errorName(err)}); }}Error when extracting with unsupported MIME type
#include <assert.h>#include <stdint.h>#include <stdio.h>#include <stdlib.h>#include <string.h>#include "xberg.h"
int main(void) { const char *input_json_base = "{\"bytes\":\"__ALEF_DOC_FILE_0__\",\"config\":{},\"filename\":\"plain.txt\",\"kind\":\"bytes\",\"mime_type\":\"application/x-nonexistent\"}"; FILE *input_file_0 = fopen("text/plain.txt", "rb"); if (input_file_0 == NULL) return EXIT_FAILURE; fseek(input_file_0, 0, SEEK_END); long input_size_0 = ftell(input_file_0); if (input_size_0 < 0) { fclose(input_file_0); return EXIT_FAILURE; } rewind(input_file_0); uint8_t *input_bytes_0 = malloc(input_size_0 > 0 ? (size_t)input_size_0 : 1); if (input_bytes_0 == NULL) { fclose(input_file_0); return EXIT_FAILURE; } if (fread(input_bytes_0, 1, (size_t)input_size_0, input_file_0) != (size_t)input_size_0) { free(input_bytes_0); fclose(input_file_0); return EXIT_FAILURE; } fclose(input_file_0); char *input_bytes_json_0 = malloc((size_t)input_size_0 * 4 + 3); if (input_bytes_json_0 == NULL) { free(input_bytes_0); return EXIT_FAILURE; } size_t input_offset_0 = 0; input_bytes_json_0[input_offset_0++] = '['; for (long i = 0; i < input_size_0; ++i) { input_offset_0 += (size_t)snprintf(input_bytes_json_0 + input_offset_0, 5, "%s%u", i == 0 ? "" : ",", input_bytes_0[i]); } input_bytes_json_0[input_offset_0++] = ']'; input_bytes_json_0[input_offset_0] = '\0'; free(input_bytes_0); const char *input_marker_0 = "\"__ALEF_DOC_FILE_0__\""; const char *input_position_0 = strstr(input_json_base, input_marker_0); if (input_position_0 == NULL) { free(input_bytes_json_0); return EXIT_FAILURE; } size_t input_prefix_0 = (size_t)(input_position_0 - input_json_base); size_t input_json_size_0 = strlen(input_json_base) - strlen(input_marker_0) + strlen(input_bytes_json_0) + 1; char *input_json_0 = malloc(input_json_size_0); if (input_json_0 == NULL) { free(input_bytes_json_0); return EXIT_FAILURE; } snprintf(input_json_0, input_json_size_0, "%.*s%s%s", (int)input_prefix_0, input_json_base, input_bytes_json_0, input_position_0 + strlen(input_marker_0)); free(input_bytes_json_0); XBERGAlefHandle input_handle = xberg_extract_input_from_json(input_json_0); free(input_json_0); XBERGAlefHandle config_handle = xberg_extraction_config_from_json("{}"); XBERGAlefHandle result = xberg_extract(input_handle, config_handle); if (result != 0) { return EXIT_FAILURE; } xberg_extract_input_free(input_handle); xberg_extraction_config_free(config_handle); return EXIT_SUCCESS;}Next Steps
Section titled “Next Steps”- Configuration — all configuration options and file formats
- OCR Guide — set up optical character recognition
- Chunking — split text for RAG
- Language Detection — multilingual document analysis
- Embeddings — semantic vectors for search
- Element-Based Output — structured element arrays for RAG
- Document Structure — hierarchical tree output