Packages
kreuzberg
4.6.2
4.10.2
4.10.1
4.10.0
4.9.9
4.9.7
4.9.5
4.9.4
4.9.3
4.9.2
4.9.1
4.8.6
4.8.5
4.8.4
4.8.3
4.8.2
4.8.1
4.8.0
4.7.4
4.7.3
4.7.2
4.7.1
4.7.0
4.6.3
4.6.2
4.6.1
4.6.0
4.5.4
4.5.3
4.5.2
4.5.1
4.4.6
4.4.5
4.4.4
4.4.3
4.4.2
4.4.1
4.4.0
4.3.8
4.3.7
4.3.6
4.3.5
4.3.4
4.3.3
4.3.2
4.3.0
4.2.15
4.2.14
4.2.13
4.2.12
4.2.11
4.2.10
4.2.9
4.2.8
4.2.7
4.2.6
4.2.5
4.2.4
4.2.3
4.2.2
4.2.1
4.2.0
4.1.2
4.1.1
4.1.0
4.0.8
4.0.7
4.0.6
4.0.4
4.0.3
4.0.2
4.0.1
4.0.0
4.0.0-rc.27
4.0.0-rc.26
High-performance document intelligence library with OCR support
Current section
Files
Jump to
Current section
Files
native/kreuzberg_rustler/src/extraction.rs
//! Extraction NIFs
//!
//! This module provides Native Implemented Functions (NIFs) for document extraction,
//! including single file/bytes extraction and batch operations.
use crate::atoms;
use crate::config::parse_extraction_config;
use crate::conversion::convert_extraction_result_to_term;
use rustler::{Binary, Encoder, Env, NifResult, ResourceArc, Term};
use std::sync::Mutex;
// Constants for validation
const MAX_BINARY_SIZE: usize = 500 * 1024 * 1024; // 500MB
/// Extract text and data from a document binary with default configuration
///
/// # Arguments
/// * `input` - Binary containing the document data
/// * `mime_type` - String representing the MIME type (e.g., "application/pdf")
///
/// # Returns
/// * `{:ok, result_map}` - Map containing extraction results
/// * `{:error, reason}` - Error tuple with reason string
#[rustler::nif(schedule = "DirtyCpu")]
pub fn extract<'a>(env: Env<'a>, input: Binary<'a>, mime_type: String) -> NifResult<Term<'a>> {
// Validate input
if input.is_empty() {
return Ok((atoms::error(), "Binary input cannot be empty").encode(env));
}
if input.len() > MAX_BINARY_SIZE {
return Ok((atoms::error(), "Binary input exceeds maximum size of 500MB").encode(env));
}
// Create default extraction config
let config = kreuzberg::core::config::ExtractionConfig::default();
// Call kreuzberg extraction with default config
match kreuzberg::extract_bytes_sync(input.as_slice(), &mime_type, &config) {
Ok(result) => {
// Convert ExtractionResult to Elixir term
match convert_extraction_result_to_term(env, &result) {
Ok(term) => Ok((atoms::ok(), term).encode(env)),
Err(e) => Ok((atoms::error(), format!("Failed to encode result: {}", e)).encode(env)),
}
}
Err(e) => Ok((atoms::error(), format!("Extraction failed: {}", e)).encode(env)),
}
}
/// Extract text and data from a document binary with custom configuration
///
/// # Arguments
/// * `input` - Binary containing the document data
/// * `mime_type` - String representing the MIME type (e.g., "application/pdf")
/// * `options` - Term containing extraction options (as map or keyword list)
///
/// # Returns
/// * `{:ok, result_map}` - Map containing extraction results
/// * `{:error, reason}` - Error tuple with reason string
#[rustler::nif(schedule = "DirtyCpu")]
pub fn extract_with_options<'a>(
env: Env<'a>,
input: Binary<'a>,
mime_type: String,
options: Term<'a>,
) -> NifResult<Term<'a>> {
// Validate input
if input.is_empty() {
return Ok((atoms::error(), "Binary input cannot be empty").encode(env));
}
if input.len() > MAX_BINARY_SIZE {
return Ok((atoms::error(), "Binary input exceeds maximum size of 500MB").encode(env));
}
// Parse options from Elixir term to ExtractionConfig
let config = match parse_extraction_config(env, options) {
Ok(cfg) => cfg,
Err(e) => return Ok((atoms::error(), format!("Invalid options: {}", e)).encode(env)),
};
// Call kreuzberg extraction with parsed config
match kreuzberg::extract_bytes_sync(input.as_slice(), &mime_type, &config) {
Ok(result) => {
// Convert ExtractionResult to Elixir term
match convert_extraction_result_to_term(env, &result) {
Ok(term) => Ok((atoms::ok(), term).encode(env)),
Err(e) => Ok((atoms::error(), format!("Failed to encode result: {}", e)).encode(env)),
}
}
Err(e) => Ok((atoms::error(), format!("Extraction failed: {}", e)).encode(env)),
}
}
/// Extract text and data from a file at the given path with default configuration
///
/// # Arguments
/// * `path` - String containing the file path
/// * `mime_type` - Optional string representing the MIME type; if None, MIME type is detected from file
///
/// # Returns
/// * `{:ok, result_map}` - Map containing extraction results
/// * `{:error, reason}` - Error tuple with reason string
#[rustler::nif(schedule = "DirtyCpu")]
pub fn extract_file<'a>(env: Env<'a>, path: String, mime_type: Option<String>) -> NifResult<Term<'a>> {
// Create default extraction config
let config = kreuzberg::core::config::ExtractionConfig::default();
// Call kreuzberg file extraction with default config
match kreuzberg::extract_file_sync(&path, mime_type.as_deref(), &config) {
Ok(result) => {
// Convert ExtractionResult to Elixir term
match convert_extraction_result_to_term(env, &result) {
Ok(term) => Ok((atoms::ok(), term).encode(env)),
Err(e) => Ok((atoms::error(), format!("Failed to encode result: {}", e)).encode(env)),
}
}
Err(e) => Ok((atoms::error(), format!("Extraction failed: {}", e)).encode(env)),
}
}
/// Extract text and data from a file at the given path with custom configuration
///
/// # Arguments
/// * `path` - String containing the file path
/// * `mime_type` - Optional string representing the MIME type; if None, MIME type is detected from file
/// * `options` - Term containing extraction options (as map or keyword list)
///
/// # Returns
/// * `{:ok, result_map}` - Map containing extraction results
/// * `{:error, reason}` - Error tuple with reason string
#[rustler::nif(schedule = "DirtyCpu")]
pub fn extract_file_with_options<'a>(
env: Env<'a>,
path: String,
mime_type: Option<String>,
options_term: Term<'a>,
) -> NifResult<Term<'a>> {
// Parse options from Elixir term to ExtractionConfig
let config = match parse_extraction_config(env, options_term) {
Ok(cfg) => cfg,
Err(e) => return Ok((atoms::error(), format!("Invalid options: {}", e)).encode(env)),
};
// Call kreuzberg file extraction with parsed config
match kreuzberg::extract_file_sync(&path, mime_type.as_deref(), &config) {
Ok(result) => {
// Convert ExtractionResult to Elixir term
match convert_extraction_result_to_term(env, &result) {
Ok(term) => Ok((atoms::ok(), term).encode(env)),
Err(e) => Ok((atoms::error(), format!("Failed to encode result: {}", e)).encode(env)),
}
}
Err(e) => Ok((atoms::error(), format!("Extraction failed: {}", e)).encode(env)),
}
}
/// Render a single page of a PDF file to a PNG byte buffer
///
/// # Arguments
/// * `input` - String file path to the PDF
/// * `page_index` - Zero-based page index
/// * `dpi` - Optional DPI (default 150)
///
/// # Returns
/// * `{:ok, binary}` - PNG binary
/// * `{:error, reason}` - Error tuple with reason string
#[rustler::nif(schedule = "DirtyCpu")]
pub fn render_pdf_page<'a>(env: Env<'a>, input: String, page_index: usize, dpi: Option<i32>) -> NifResult<Term<'a>> {
if input.is_empty() {
return Ok((atoms::error(), "File path cannot be empty").encode(env));
}
let pdf_bytes = match std::fs::read(&input) {
Ok(b) => b,
Err(e) => return Ok((atoms::error(), format!("Failed to read file: {}", e)).encode(env)),
};
match kreuzberg::pdf::render_pdf_page_to_png(&pdf_bytes, page_index, dpi, None) {
Ok(png) => {
let mut obin = match rustler::OwnedBinary::new(png.len()) {
Some(b) => b,
None => {
return Ok((
atoms::error(),
format!("failed to allocate binary of {} bytes", png.len()),
)
.encode(env));
}
};
obin.as_mut_slice().copy_from_slice(&png);
Ok((atoms::ok(), obin.release(env)).encode(env))
}
Err(e) => Ok((atoms::error(), format!("Rendering failed: {}", e)).encode(env)),
}
}
/// Resource wrapper for PdfPageIterator to allow passing between NIF calls.
pub struct PdfPageIteratorResource {
inner: Mutex<Option<kreuzberg::pdf::PdfPageIterator>>,
}
#[rustler::resource_impl]
impl rustler::Resource for PdfPageIteratorResource {}
/// Open a new PDF page iterator, returning a resource handle.
#[rustler::nif(schedule = "DirtyCpu")]
pub fn render_pdf_pages_iter_open<'a>(env: Env<'a>, path: String, dpi: Option<i32>) -> NifResult<Term<'a>> {
if path.is_empty() {
return Ok((atoms::error(), "File path cannot be empty").encode(env));
}
match kreuzberg::pdf::PdfPageIterator::from_file(&path, dpi, None) {
Ok(iter) => {
let resource = ResourceArc::new(PdfPageIteratorResource {
inner: Mutex::new(Some(iter)),
});
Ok(resource.encode(env))
}
Err(e) => Ok((atoms::error(), format!("Failed to open iterator: {}", e)).encode(env)),
}
}
/// Advance the iterator and return the next page.
///
/// Returns `{:ok, {page_index, png_binary}}` or `:done` when exhausted.
#[rustler::nif(schedule = "DirtyCpu")]
pub fn render_pdf_pages_iter_next<'a>(
env: Env<'a>,
resource: ResourceArc<PdfPageIteratorResource>,
) -> NifResult<Term<'a>> {
let mut guard = resource
.inner
.lock()
.map_err(|_| rustler::Error::Term(Box::new("iterator lock poisoned")))?;
let iter = match guard.as_mut() {
Some(it) => it,
None => return Ok(atoms::done().encode(env)),
};
match iter.next() {
Some(Ok((page_index, png))) => {
let mut obin = match rustler::OwnedBinary::new(png.len()) {
Some(b) => b,
None => {
return Ok((
atoms::error(),
format!("failed to allocate binary of {} bytes", png.len()),
)
.encode(env));
}
};
obin.as_mut_slice().copy_from_slice(&png);
Ok((atoms::ok(), (page_index, obin.release(env))).encode(env))
}
Some(Err(e)) => Ok((atoms::error(), format!("Iterator error: {}", e)).encode(env)),
None => Ok(atoms::done().encode(env)),
}
}
/// Free the iterator resource.
#[rustler::nif]
pub fn render_pdf_pages_iter_free(resource: ResourceArc<PdfPageIteratorResource>) -> rustler::NifResult<()> {
if let Ok(mut guard) = resource.inner.lock() {
*guard = None;
}
Ok(())
}