{"uid":"cap_WK3dVJoTB0YDNmqAeV0_W","slug":"horizon-pulse-pdf-to-text-extractor-1d6affbd","name":"Horizon Pulse PDF to Text Extractor","description":"Pay-per-call APIs for AI agents via x402 on Base: web fetch, HTTP proxy, page extract, and crypto market data.","url":"https://horizonpulse.dev/api/pdf?utm_source=zero.xyz","method":"GET","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","method"],"properties":{"type":{"type":"string","const":"http"},"method":{"enum":["GET","HEAD","DELETE"],"type":"string"},"queryParams":{"type":"object","required":["url"],"properties":{"url":{"type":"string","description":"Absolute http(s) URL of a public PDF (required, max 10MB)."},"pages":{"type":"string","description":"Max pages to return, integer 1-50 (default 50)."}}}},"additionalProperties":false},"output":{"type":"object","required":["type"],"properties":{"type":{"type":"string"},"example":{"type":"object","description":"Horizon Pulse PDF text extraction result. The example is a trimmed real response recorded from GET /api/demo/pdf at 2026-10-01T14:36Z UTC; live values differ on every call."}}}}},"responseSchema":{"type":"json","example":{"ok":true,"meta":{"title":"Horizon Pulse sample PDF","author":"Horizon Pulse"},"bytes":858,"pages":[{"page":1,"text":"Horizon Pulse sample PDF\nPay-per-call APIs for AI agents over x402 (USDC on Base).\nThis file is the fixed input for the free /api/demo/pdf sample."}],"finalUrl":"https://horizonpulse.dev/sample.pdf","truncated":false,"totalPages":1,"requestedUrl":"https://horizonpulse.dev/sample.pdf","pagesReturned":1}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.02","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.02/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.02","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.02","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_P9exiYnFWsGkpcvpbAwY8","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.02","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Fetches a public PDF URL and returns clean extracted text per page plus title and author metadata, using pdf.js text layer (no OCR), limited to 10MB, 50 pages, and 100K characters.","exampleAgentPrompt":"Can you grab the text from this PDF — https://example.gov/report.pdf — and give me the content of the first 10 pages along with whatever title and author info is in the document?","exampleUseCases":[{"title":"Research paper text extraction","prompt":"I need the full text from this academic paper PDF at https://arxiv.org/pdf/2301.00001 — extract all the pages and tell me the title and authors listed in the document."},{"title":"Legal document reading","prompt":"Pull the text out of this public contract PDF at https://filings.sec.gov/contract.pdf so I can read through it — get up to 20 pages and include any metadata like title or author."},{"title":"Report summarization pipeline","prompt":"Fetch the text from this government report PDF at https://cdc.gov/annual-report-2023.pdf — I want all available pages extracted so I can summarize the key findings."}],"resultDescription":"Returns clean text extracted per page from the PDF, along with document-level metadata such as title and author. Output is structured by page, covering up to 50 pages and 100K characters. Files that are non-PDF, encrypted, or image-only (requiring OCR) are returned unbilled with an appropriate indication.","failureModes":["PDF exceeds 10MB size limit — request rejected or truncated","PDF requires OCR (image-only scans) — no text extracted, unbilled","PDF is password-protected or encrypted — cannot parse, unbilled","URL does not point to a valid PDF — non-PDF response, unbilled","SSRF-blocked URL (private IP ranges, internal hostnames) — request rejected","PDF has more than 50 pages — only first 50 pages returned","Text exceeds 100K character limit — output truncated at that boundary","URL is unreachable or returns non-200 status — fetch error"],"whenToPreferThis":"Choose this endpoint when you have a public HTTP/HTTPS URL pointing to a real PDF with an embedded text layer (not a scanned image PDF) and need clean per-page text plus title/author metadata. It is ideal for processing research papers, reports, filings, and other text-layer PDFs up to 10MB and 50 pages. Prefer this over generic web scrapers when the source is specifically a PDF. Do not use for scanned/image PDFs (no OCR), password-protected files, or PDFs larger than 10MB.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-01T06:41:56.591Z","isFirstParty":false,"canonicalSlug":"horizon-pulse-pdf-to-text-extractor-1d6affbd"}