{"slug":"netintel-dev-web-extract-a58b33","title":"Extract text from a web page or PDF as clean Markdown — HTML to Markdown for any","host":"netintel.dev","method":"GET","resource":"https://netintel.dev/web/extract","category":"code","description":"Extract text from a web page or PDF as clean Markdown — HTML to Markdown for any URL: strips scripts, nav, ads, and boilerplate while preserving headings, links, lists, tables, code blocks, and blockquotes; extracts the text layer from PDFs. Returns Markdown body, title, word count, and a quality gr","price_listed":0.003,"price_asked":0.003,"state":"answering","state_label":"Answering","checks_7d":1,"answered_7d":1,"latency_ms_median":1140,"reported_calls_30d":350,"reported_payers_30d":8,"networks":["eip155:8453","solana:5eykt4UsFv8P8NJdTREpY1vzqKqZKvdp"],"badge":"unverified","paid_checks_7d":0,"paid_ok_7d":0,"example_input":{"method":"GET","queryParams":{"url":"https://www.sitemaps.org/protocol.html"},"type":"http"},"output_schema":{"$schema":"https://json-schema.org/draft/2020-12/schema","properties":{"input":{"additionalProperties":false,"properties":{"method":{"enum":["GET"],"type":"string"},"queryParams":{"properties":{"url":{"description":"Public URL of an HTML page or PDF to extract (e.g. https://www.sitemaps.org/protocol.html)","type":"string"}},"required":["url"],"type":"object"},"type":{"const":"http","type":"string"}},"required":["type","method"],"type":"object"},"output":{"properties":{"example":{"properties":{"char_count":{"type":"number"},"content_type":{"description":"article | pdf | other (non-HTML/PDF falls back to best-effort plain text)","type":"string"},"final_url":{"type":"string"},"findings":{"type":"array"},"grade":{"type":"string"},"markdown":{"type":"string"},"output_bytes":{"type":"number"},"score":{"type":"number"},"status_code":{"type":"number"},"title":{"type":"string"},"truncated":{"type":"boolean"},"url":{"type":"string"},"word_count":{"type":"number"}},"type":"object"},"type":{"type":"string"}},"required":["type"],"type":"object"}},"required":["input"],"type":"object"},"history":[{"day":"2026-09-24","reachable":true,"status":402,"valid_402":true,"asked_usdc":0.003,"price_match":true,"latency_ms":1140,"error":null}],"description_full":"Extract text from a web page or PDF as clean Markdown — HTML to Markdown for any URL: strips scripts, nav, ads, and boilerplate while preserving headings, links, lists, tables, code blocks, and blockquotes; extracts the text layer from PDFs. Returns Markdown body, title, word count, and a quality grade. Web scraper / reader for article and main content. For JS-rendered or bot-walled pages a plain fetch can't read, use /exa/contents.","last_updated":"2026-09-23T22:22:41.074Z","schemes":["exact"]}