{"slug":"agentsvc-io-api-v1-proxy-pdf-extract-798c4b","title":"Extract all text from a PDF","host":"agentsvc.io","method":"POST","resource":"https://agentsvc.io/api/v1/proxy/pdf-extract","category":"search","description":"Extract all text from a PDF. Send as pdf_base64 (base64-encoded PDF, max ~10 MB decoded). Returns text (full concatenated text), pages array (per-page text + char_count), page_count, and metadata (title, author, creator). Encode with: Buffer.from(pdfBytes).toString('base64'). Ideal for RAG pipelines","price_listed":0.0055,"price_asked":null,"state":"effects","state_label":"Not tested: has real-world effects","checks_7d":0,"answered_7d":0,"latency_ms_median":null,"reported_calls_30d":7,"reported_payers_30d":1,"networks":["eip155:137","eip155:42161","eip155:8453","solana:5eykt4UsFv8P8NJdTREpY1vzqKqZKvdp"],"badge":"unverified","paid_checks_7d":0,"paid_ok_7d":0,"example_input":{"body":{"max_pages":10,"pdf_base64":"<base64 of a PDF>"},"bodyType":"json","method":"POST","type":"http"},"output_schema":{"$schema":"https://json-schema.org/draft/2020-12/schema","properties":{"input":{"additionalProperties":false,"properties":{"body":{"properties":{"max_pages":{"default":50,"description":"Maximum number of pages to extract. Default: 50. Use to limit processing time for large PDFs.","type":"integer"},"pdf_base64":{"description":"Base64-encoded PDF file content. Decode a PDF file to base64 and pass it here. Max ~10 MB (unencoded).","type":"string"}},"required":["pdf_base64"]},"bodyType":{"enum":["json","form-data","text"],"type":"string"},"method":{"enum":["POST","PUT","PATCH"],"type":"string"},"type":{"const":"http","type":"string"}},"required":["type","method","bodyType","body"],"type":"object"},"output":{"properties":{"example":{"properties":{"data":{"properties":{"extracted_at":{"description":"ISO 8601 timestamp of extraction","format":"date-time","type":"string"},"file_size_bytes":{"description":"Size of the decoded PDF in bytes","type":"integer"},"metadata":{"description":"PDF document metadata (if available)","properties":{"author":{"type":"string"},"creation_date":{"type":"string"},"creator":{"type":"string"},"producer":{"type":"string"},"subject":{"type":"string"},"title":{"type":"string"}},"type":"object"},"page_count":{"description":"Total number of pages in the PDF","type":"integer"},"pages":{"description":"Per-page text content (first max_pages pages)","items":{"properties":{"char_count":{"description":"Number of characters on this page","type":"integer"},"page":{"description":"Page number (1-based)","type":"integer"},"text":{"description":"Extracted text for this page","type":"string"}},"type":"object"},"type":"array"},"text":{"description":"Full extracted text from all pages, joined with newlines. Preserves paragraph structure where possible.","type":"string"}},"required":["text","page_count","extracted_at"],"type":"object"},"success":{"type":"boolean"}},"required":["success","data"],"type":"object"},"type":{"type":"string"}},"required":["type"],"type":"object"}},"required":["input"],"type":"object"},"history":[],"description_full":"Extract all text from a PDF. Send as pdf_base64 (base64-encoded PDF, max ~10 MB decoded). Returns text (full concatenated text), pages array (per-page text + char_count), page_count, and metadata (title, author, creator). Encode with: Buffer.from(pdfBytes).toString('base64'). Ideal for RAG pipelines, document QA, or LLM ingestion.","last_updated":"2026-10-10T09:02:07.683Z","schemes":["exact"]}