{"uid":"cap_BvMsfmjdmWlZko5o0YL5m","slug":"web-page-to-clean-text-extractor-c425b6d0","name":"Web Page to Clean Text Extractor","description":"Web page to clean text: title, description, headings, links and readable text from any public URL — Genesis402 / UnyKorn Operator Network","url":"https://twin.unykorn.org/web-extract?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"params":{"type":"object","properties":{"url":{"type":"string","format":"uri"},"max_chars":{"type":"integer","maximum":60000,"minimum":500},"include_links":{"type":"boolean"}}}}},"responseSchema":{"type":"json","example":{"ok":true,"text":"<readable text>","type":"web-extract","links":[{"url":"https://…","text":"<anchor>"}],"title":"<title>","receipt":{"tx_hash":"0x<64hex>","amount_usd":0.004,"receipt_id":"g402-<16hex>"},"headings":[{"text":"<h1>","level":1}],"final_url":"https://www.x402.org/","description":"<meta description>","http_status":200,"content_sha256":"<64hex>"}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.004","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.004/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.004","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.004","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_GTTTylhkyAMbKVf9qtj1q","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.004","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Fetches any public URL and returns cleaned, readable text including title, description, headings, links, and body content","exampleAgentPrompt":"Can you pull the clean readable text from https://www.bbc.com/news/technology-12345678 — include the title, description, headings, and links, up to 10000 characters?","exampleUseCases":[{"title":"Research article text for summarization","prompt":"Fetch the full readable text from https://techcrunch.com/2024/05/01/ai-agents-explainer/ — I want the title, headings, and body content up to 20000 characters so I can summarize it."},{"title":"Competitive pricing page analysis","prompt":"Extract the clean text from https://www.competitor.com/pricing — get all the headings and body text, no links needed, up to 15000 characters."},{"title":"News story fact-checking","prompt":"Pull the title, description, headings, and all links from https://www.reuters.com/world/some-story-2024 so I can verify what sources they cite — up to 30000 characters please."}],"resultDescription":"Returns the page's title, meta description, structured headings, main readable body text, and optionally all hyperlinks found on the page — all cleaned of HTML markup and limited to the requested character count.","failureModes":["URL is not publicly accessible (paywalled, behind login, or blocked) — returns error or empty content","URL is malformed or invalid — returns validation error","Page returns non-HTML content (PDF, binary) — may return empty or partial text","max_chars too low to capture meaningful content — truncated output","Server timeout if target page is slow to respond","Bot-blocking or rate-limiting by the target website — returns error or incomplete content"],"whenToPreferThis":"Choose this endpoint when you need to quickly extract readable human-facing text from any public URL without running a full browser — ideal for LLM context feeding, article summarization, link extraction, or competitive research where you need clean text rather than raw HTML. It is especially useful in agent pipelines that need to process web content programmatically at low cost ($0.004 per call).","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-01T06:30:02.527Z","isFirstParty":false,"canonicalSlug":"web-page-to-clean-text-extractor-c425b6d0"}