{"slug":"netintel-dev-pdf-parse-87f448","title":"Parse a PDF to text and Markdown","host":"netintel.dev","method":"POST","resource":"https://netintel.dev/pdf/parse","category":"image","description":"Parse a PDF to text and Markdown: send a PDF url or base64 file, get clean Markdown with paragraphs rebuilt from the text layer, per-page character offsets for citing pages, metadata (title, author, dates, page count) and the page numbers that are scanned images with no text layer. Up to 50 pages pe","price_listed":0.005,"price_asked":null,"state":"effects","state_label":"Not tested: has real-world effects","checks_7d":0,"answered_7d":0,"latency_ms_median":null,"reported_calls_30d":2,"reported_payers_30d":2,"networks":["eip155:8453","solana:5eykt4UsFv8P8NJdTREpY1vzqKqZKvdp"],"badge":"unverified","paid_checks_7d":0,"paid_ok_7d":0,"example_input":{"body":{"url":"https://netintel.dev/samples/pdf-parse-sample.pdf"},"bodyType":"json","method":"POST","type":"http"},"output_schema":{"$schema":"https://json-schema.org/draft/2020-12/schema","properties":{"input":{"additionalProperties":false,"properties":{"body":{"properties":{"file_base64":{"description":"The PDF as base64 or a data: URL (up to ~700 KB). Send url OR file_base64.","type":"string"},"max_chars":{"description":"Cap on the markdown length (default 100000)","type":"number"},"pages":{"description":"Pages to parse, 1-based, e.g. \"1-3,7\" (default: the first 50)","type":"string"},"url":{"description":"Public URL of the PDF (up to 10 MB)","type":"string"}},"type":"object"},"bodyType":{"enum":["json","form-data","text"],"type":"string"},"method":{"enum":["POST"],"type":"string"},"type":{"const":"http","type":"string"}},"required":["type","method","bodyType","body"],"type":"object"},"output":{"properties":{"example":{"properties":{"char_count":{"type":"number"},"findings":{"items":{"type":"string"},"type":"array"},"markdown":{"description":"All parsed pages, each after a <!-- page N --> marker","type":"string"},"metadata":{"type":"object"},"page_count":{"type":"number"},"pages":{"description":"Per page: page, word_count, has_text, char_start, char_end (offsets into markdown)","type":"array"},"pages_parsed":{"type":"number"},"scanned_pages":{"items":{"type":"number"},"type":"array"},"source":{"type":"object"},"truncated":{"type":"boolean"},"word_count":{"type":"number"}},"type":"object"},"type":{"type":"string"}},"required":["type"],"type":"object"}},"required":["input"],"type":"object"},"history":[],"description_full":"Parse a PDF to text and Markdown: send a PDF url or base64 file, get clean Markdown with paragraphs rebuilt from the text layer, per-page character offsets for citing pages, metadata (title, author, dates, page count) and the page numbers that are scanned images with no text layer. Up to 50 pages per call. Text layer only, no OCR: an image-only PDF returns an uncharged 422.","last_updated":"2026-10-10T22:51:16.521Z","schemes":["exact"]}