{"slug":"gateway-stride20k-com-web-extract-2b4737","title":"Web scraping API","host":"gateway.stride20k.com","method":"GET","resource":"https://gateway.stride20k.com/web/extract","category":"search","description":"Web scraping API: fetch any public web page and get its readable content as clean markdown — title, author, canonical URL, boilerplate stripped. Honest User-Agent, robots.txt honored (explicit Disallow returns an unpaid 403), private/internal targets refused, at most 3 safety-revalidated redirects, ","price_listed":0.03,"price_asked":0.03,"state":"answering","state_label":"Answering","checks_7d":1,"answered_7d":1,"latency_ms_median":21,"reported_calls_30d":1,"reported_payers_30d":1,"networks":["eip155:8453"],"badge":"unverified","paid_checks_7d":0,"paid_ok_7d":0,"example_input":{"method":"GET","queryParams":{"url":"https://example.com/"},"type":"http"},"output_schema":{"$schema":"https://json-schema.org/draft/2020-12/schema","properties":{"input":{"additionalProperties":false,"properties":{"method":{"enum":["GET"],"type":"string"},"queryParams":{"properties":{"url":{"description":"Absolute http(s) URL of the page to extract.","pattern":"https?://.+","type":"string","urlSafety":true}},"required":["url"],"type":"object"},"type":{"const":"http","type":"string"}},"required":["type","method"],"type":"object"},"output":{"properties":{"example":{"properties":{"data":{"properties":{"author":{"description":"meta author, or null.","type":["string","null"]},"canonicalUrl":{"description":"rel=canonical link if the page declares one, else null.","type":["string","null"]},"charCount":{"description":"Length of the markdown field.","type":"integer"},"finalUrl":{"description":"URL that actually served the content.","type":"string"},"markdown":{"description":"Extracted readable content as markdown.","minLength":1,"type":"string"},"title":{"description":"Document title, or null.","type":["string","null"]},"truncated":{"description":"True if output hit the 100k-char cap.","type":"boolean"},"url":{"description":"The URL requested (after redirect re-validation).","type":"string"}},"required":["url","markdown","truncated"],"type":"object"},"endpoint":{"description":"Always \"/web/extract\".","type":"string"},"meta":{"properties":{"attribution":{"description":"Licensing attribution when the source requires it.","type":["string","null"]},"cached":{"description":"Whether this response was served from cache.","type":"boolean"},"fetchedAt":{"description":"ISO 8601 time the data was actually retrieved from the upstream.","type":"string"},"source":{"description":"Upstream source: direct fetch (buyer-directed).","type":"string"}},"required":["source","cached","fetchedAt"],"type":"object"},"ok":{"description":"true on success.","type":"boolean"}},"required":["ok","endpoint","data","meta"],"type":"object"},"type":{"type":"string"}},"required":["type"],"type":"object"}},"required":["input"],"type":"object"},"history":[{"day":"2026-09-24","reachable":true,"status":402,"valid_402":true,"asked_usdc":0.03,"price_match":true,"latency_ms":21,"error":null}],"description_full":"Web scraping API: fetch any public web page and get its readable content as clean markdown — title, author, canonical URL, boilerplate stripped. Honest User-Agent, robots.txt honored (explicit Disallow returns an unpaid 403), private/internal targets refused, at most 3 safety-revalidated redirects, 1MB input / 100k character output caps. HTML pages only. Use for research agents, content extraction, summarization pipelines, and RAG ingestion. Cached up to 5 minutes per URL.","last_updated":"2026-08-30T12:06:31.356Z","schemes":["exact"]}