{"uid":"cap_tywRL6pALR3AYpsLw9qj2","slug":"web-scraper-api-clean-text-extraction-e6101059","name":"Web Scraper API – Clean Text Extraction","description":"Fetch a URL and extract clean main-content text with title, description, word count, and char count; boilerplate (nav/footer/sidebar/ads) is stripped.","url":"https://web-scraper-api-production-bf20.up.railway.app/scrape/text?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"url":{"type":"string","description":"Absolute http(s) URL of the page to scrape, e.g. 'https://example.com'."}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.002","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.002/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.002","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.002","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_SkB0uMz1_PubpJIrjQJQq","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.002","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Fetches a URL and returns the main article/body text with title, description, word count, and character count, stripping boilerplate like navigation, footers, sidebars, and ads.","exampleAgentPrompt":"Can you grab the main article text from https://www.theverge.com/2024/1/15/some-article and give me a clean version without all the nav menus, ads, and footer junk?","exampleUseCases":[{"title":"Research article summarization","prompt":"Go to https://www.wired.com/story/ai-regulation-2024 and pull out the clean article text so I can summarize it — strip all the ads and navigation."},{"title":"Competitor blog content monitoring","prompt":"Fetch the main body text from https://blog.competitor.com/latest-product-update so I can analyze what they're announcing, without any of the sidebar or footer noise."},{"title":"Fact-checking a news story","prompt":"Scrape the readable content from https://apnews.com/article/some-news-story and tell me what it actually says — just the article, not the menus or ads."}],"resultDescription":"Returns the main content text of the page (boilerplate stripped), the page title, meta description, total word count, and character count as structured fields.","failureModes":["Invalid or non-HTTP(S) URL returns a validation error","Page returns a non-200 HTTP status (404, 403, 500) — error propagated to caller","JavaScript-heavy single-page apps may return incomplete or empty content if server-side rendering is unavailable","Very large pages may time out or be truncated","Paywalled or login-required pages return only teaser/preview content","Rate limiting or IP blocking by the target site may cause fetch failures"],"whenToPreferThis":"Choose this endpoint when you need the human-readable prose from a webpage — article bodies, blog posts, documentation pages — and want boilerplate (nav, footer, ads, sidebars) automatically removed. It is ideal for summarization, fact-checking, or feeding page content into an LLM. Use the sibling metadata endpoint if you need structured SEO fields, or the structured-elements endpoint if you need headings, tables, and lists rather than prose text.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T12:35:26.787Z","isFirstParty":false,"canonicalSlug":"web-scraper-api-clean-text-extraction-e6101059"}