{"uid":"cap_eQjCk51wlvR5k9BeicJU2","slug":"web-page-text-extractor-ssrf-safe-d08beb93","name":"Web Page Text Extractor (SSRF-Safe)","description":"Fetch a public web page and return clean structured text: title, description, headings, body text (50k chars) and up to 200 links. SSRF-safe (private/internal hosts blocked). For agents that need to read web content without a browser.","url":"https://simultaneously-provincial-pit-joseph.trycloudflare.com/api/v1/web-extract","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"url":{"type":"string","description":"A public http(s) URL to fetch and extract text from"}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.1","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.1/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.1","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.1","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_RDCTXeTfC-JuL2r-1cMGE","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.1","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Fetches a public web page and returns clean structured text including title, description, headings, body text, and links — without a browser.","exampleAgentPrompt":"Can you fetch the content of https://techcrunch.com/2024/05/01/openai-news/ and give me the title, main body text, and any links on the page?","exampleUseCases":[{"title":"Research article content from URL","prompt":"Go to this URL — https://www.bbc.com/news/technology-68012396 — and pull out the full article text and headings so I can summarize it."},{"title":"Competitive intelligence from a product page","prompt":"Fetch the page at https://www.competitor.com/pricing and extract all the text and headings so I can see what plans they offer."},{"title":"Extract links from a resource directory","prompt":"Can you grab the content of https://awesome-list.github.io/resources and return all the links and headings listed there?"}],"resultDescription":"A structured object containing the page title, meta description, headings (h1–h6), body text (up to 50,000 characters), and up to 200 links found on the page — all as clean text, ready to read or process.","failureModes":["Private or internal host blocked (SSRF protection) — returns an error if the URL resolves to a private IP range","Non-200 HTTP response from target — error returned with upstream status","URL is not a valid http/https URL — validation error","Page is mostly JavaScript-rendered with no static HTML content — body text may be sparse or empty","Rate limiting or connection timeout on target server — timeout error"],"whenToPreferThis":"Choose this endpoint when an agent needs to read the readable text content of a public web page without spinning up a headless browser. It is ideal for article extraction, link harvesting, and structured text retrieval from static or server-rendered pages. Prefer this over browser-based tools when speed and simplicity matter and the target page does not require JavaScript execution. The SSRF safety guarantee makes it appropriate for multi-tenant or automated agent pipelines where untrusted URLs may be submitted.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-13T18:37:25.913Z","isFirstParty":false}