{"uid":"cap_iAmCL_xw_P1x9gN3wMt25","slug":"web-page-extractor-clean-markdown-via-url-fetch-97b92290","name":"Web Page Extractor — Clean Markdown via URL Fetch","description":"Read a web page for me: fetches up to 5 URLs and returns the main article content as clean Markdown (nav, ads, and footers stripped) plus title, author, published date, canonical URL, links, images, word count and reading time. Typically 15-30x smaller than the raw HTML, so it saves far more in context tokens than it costs. Honors robots.txt by default. No browser or network access needed on your side.","url":"https://toolbelt402.tpoborne.workers.dev/web/extract?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"url":{"type":"string","description":"Single URL to extract (or use urls)"},"urls":{"type":"array","items":{"type":"string"},"maxItems":5,"description":"Up to 5 absolute http(s) URLs"},"format":{"enum":["markdown","text"],"type":"string","description":"Output format (default markdown)"},"maxBytes":{"type":"integer","description":"Cap HTML read per page (default 524288)"},"userAgent":{"type":"string","description":"User-agent to send and to evaluate robots.txt against"},"includeLinks":{"type":"boolean","description":"Include links found in the article body (default true)"},"includeImages":{"type":"boolean","description":"Include images found in the article body (default true)"},"respectRobots":{"type":"boolean","description":"Honor robots.txt for the given user-agent (default true)"}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.02","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.02/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.02","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.02","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_Oct536zwy0S77IzkJ7s__","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.02","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Fetches up to 5 URLs and returns the main article content as clean Markdown with title, author, date, links, images, word count, and reading time — stripping nav, ads, and footers.","exampleAgentPrompt":"Can you read this article for me and give me the main content as clean text? Here's the URL: https://example.com/some-long-article — strip out the ads, navigation, and footer noise.","exampleUseCases":[{"title":"Research assistant reading articles","prompt":"I need you to fetch the content from these three news articles and give me the main text from each one, stripped of ads and nav: https://techcrunch.com/article-a, https://wired.com/story-b, https://theverge.com/post-c"},{"title":"Summarizing a blog post","prompt":"Go read this blog post at https://martinfowler.com/articles/microservices.html and pull out the main content as clean Markdown so I can summarize it — I want the title, author, and publish date too."},{"title":"Checking article metadata","prompt":"Can you fetch https://www.bbc.com/news/technology-12345678 and tell me the article title, who wrote it, when it was published, and roughly how long it is to read?"}],"resultDescription":"A structured response per URL containing: cleaned Markdown body of the main article content, title, author name, published date, canonical URL, list of links and images found, word count, and estimated reading time. Nav bars, ads, and footers are removed. Content is typically 15-30x smaller than raw HTML.","failureModes":["URL is blocked by robots.txt — endpoint honors robots.txt by default and will refuse or return an error","URL is unreachable or returns a non-200 HTTP status","Page requires JavaScript rendering to load content — server-side fetch may return empty or incomplete content","Paywall or login-gated content returns minimal extractable text","More than 5 URLs submitted — batch limit exceeded","Malformed URL input causes a 400-level validation error"],"whenToPreferThis":"Use this endpoint when an agent needs to read and use the textual content of one or more web pages without consuming excessive context tokens. It is ideal over raw HTTP fetching because it automatically strips boilerplate (nav, ads, footers) and returns structured Markdown plus metadata. Prefer it over browser-based scraping tools when JavaScript rendering is not required and robots.txt compliance is acceptable. Best for article reading, research pipelines, and content summarization workflows.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-01T12:47:30.683Z","isFirstParty":false,"canonicalSlug":"web-page-extractor-clean-markdown-via-url-fetch-97b92290"}