{"uid":"cap_h4Vv2XLzxAS7vEBBU9zoN","slug":"horizonpulse-html-extractor-5aa58644","name":"HorizonPulse HTML Extractor","description":"Pay-per-call APIs for AI agents via x402 on Base: web fetch, HTTP proxy, page extract, and crypto market data.","url":"https://horizonpulse.dev/api/extract?utm_source=zero.xyz","method":"GET","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","method"],"properties":{"type":{"type":"string","const":"http"},"method":{"enum":["GET","HEAD","DELETE"],"type":"string"},"queryParams":{"type":"object","required":["url"],"properties":{"url":{"type":"string","description":"Absolute http(s) URL to fetch and extract (required on GET; on POST send url or html in the JSON body). Private/localhost blocked."},"html":{"type":"string","description":"POST /api/extract JSON body only (ignored on GET): raw HTML to parse instead of fetching (size-capped). If both url and html are sent, html is parsed and url is echoed."},"fields":{"type":"object","description":"Optional CSS-selector fields (max 20). Map of name to selector string or {selector, attr?: \"text\"|\"html\"|<attribute>, all?: boolean, limit?: 1-50}. GET: URL-encoded JSON. Missing fields come back null with fieldErrors; if none match, 422 no_fields_matched and no charge."}}}},"additionalProperties":false},"output":{"type":"object","required":["type"],"properties":{"type":{"type":"string"},"example":{"type":"object","description":"Horizon Pulse structured HTML extract result. The example is a trimmed real response recorded from GET /api/demo/extract at 2026-10-01T14:36Z UTC; live values differ on every call."}}}}},"responseSchema":{"type":"json","example":{"ok":true,"url":"https://horizonpulse.dev/","links":[{"href":"https://horizonpulse.dev/","text":"Horizon Pulse"},{"href":"https://horizonpulse.dev/#catalog","text":"Catalog"}],"title":"Horizon Pulse","fields":{"links":["https://horizonpulse.dev/","https://horizonpulse.dev/#catalog","https://horizonpulse.dev/#how"],"heading":"Pay-per-call APIsbuilt for agents."},"finalUrl":"https://horizonpulse.dev/","headings":[{"text":"Pay-per-call APIs built for agents.","level":1},{"text":"Thirteen routes. One protocol.","level":2}],"language":"en","elapsedMs":120,"textSample":"# Horizon Pulse\n\nLive on Base · x402 v2\n# Pay-per-call APIs\n built for agents.\n\n13 routes for market data, the web and d…","description":"Pay-per-call APIs for AI agents via x402 on Base: web fetch, HTTP proxy, page extract, and crypto market data.","fieldErrors":{},"matchedFields":2,"requestedFields":2}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.015","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.015/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.015","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.015","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_MIRQfCoaP9KzNMt_9erTG","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.015","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Extracts structured fields (title, description, canonical, links, images, headings, JSON-LD, text sample) from a public URL or raw HTML, with SSRF protection","exampleAgentPrompt":"Can you extract all the structured metadata from https://example.com/article — I need the title, description, canonical URL, headings, images, links, and any JSON-LD data on the page.","exampleUseCases":[{"title":"SEO audit of a landing page","prompt":"Pull the structured metadata from https://acme.com/landing — I need the title, meta description, canonical tag, all headings, and any JSON-LD markup so I can audit the SEO."},{"title":"Link extraction from a blog post","prompt":"Can you fetch https://techblog.io/post/123 and give me all the links and images found on that page?"},{"title":"Parse raw HTML for structured fields","prompt":"I have some HTML I copied from a product page — can you parse it and extract the title, description, headings, and any structured data like JSON-LD?"}],"resultDescription":"A structured JSON object containing extracted page fields: title, meta description, canonical URL, all links, image URLs, heading hierarchy, JSON-LD structured data blocks, and a plain-text sample of the page body. Private/localhost URLs are blocked for SSRF safety.","failureModes":["Private or localhost URLs are blocked and return an SSRF-safety error","Invalid or malformed URLs result in a fetch error","HTML size exceeding the cap is truncated or rejected","Pages behind authentication or paywalls may return incomplete or empty content","Non-HTML content types (PDF, images) may yield minimal extraction results","Network timeouts on slow or unreachable public URLs"],"whenToPreferThis":"Choose this endpoint when you need multiple structured fields from a webpage in one call (title, description, headings, links, images, JSON-LD) rather than just clean text. It is preferable to the sibling markdown/text-fetch endpoint when you need metadata, link lists, or structured schema.org data. It is ideal for SEO analysis, content enrichment, link graph construction, or validating page metadata. Use it over a generic HTTP proxy when you specifically need parsed, structured HTML fields rather than raw response bodies.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-01T18:43:04.246Z","isFirstParty":false,"canonicalSlug":"horizonpulse-html-extractor-5aa58644"}