{"uid":"cap_wWdDR2QCiqf-1IGr4IOdH","slug":"clean-text-extraction-from-html-fed2abc6","name":"Clean Text Extraction from HTML","description":"Extract clean readable text from caller-supplied HTML (no URL fetch).","url":"https://api.edifiedlab.com/v1/tools/extract-text?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"html":{"type":"string","minLength":24,"description":"Decoded HTML, minimum 24 characters. Tiny fragments such as <p>hello</p> are rejected unpaid (400) before a 402 challenge."},"html_b64":{"type":"string","description":"Base64-encoded UTF-8 HTML; decoded text must be at least 24 characters (same floor as html minLength).","contentEncoding":"base64"},"content_b64":{"type":"string","description":"Alias of html_b64; decoded text must be at least 24 characters.","contentEncoding":"base64"}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.011","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.011/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.011","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.011","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_KV6riap6-J2L4n9mXNZ5x","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.011","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Strips HTML markup and returns clean, readable plain text from caller-supplied raw HTML content (no URL fetching required).","exampleAgentPrompt":"I have this raw HTML snippet and I just need the clean, human-readable text out of it — can you strip all the tags and give me the plain text?","exampleUseCases":[{"title":"Scraper pipeline text cleanup","prompt":"I scraped some HTML from a product page and now I need to extract just the readable text — can you strip all the tags and formatting from this HTML and give me the clean content?"},{"title":"LLM context preprocessing","prompt":"Before I send this HTML email to the language model, I need the plain text version — strip the HTML tags out of this and return just the readable text so I can pass it to the LLM."},{"title":"Article content extraction for summarization","prompt":"I pulled the raw HTML of a news article and want to summarize it, but first I need the clean text — can you extract the readable content from this HTML for me?"}],"resultDescription":"Returns clean, human-readable plain text extracted from the submitted HTML, with markup, scripts, and formatting tags removed, leaving only the readable textual content.","failureModes":["HTML shorter than 24 characters returns a 400 error before any payment is charged","Malformed or empty input rejected with 400","Base64 decoding errors if html_b64 or content_b64 are improperly encoded","Very large HTML payloads may hit request size limits","Non-HTML content (e.g. plain JSON or binary) may produce garbled or empty output"],"whenToPreferThis":"Choose this endpoint when you already have raw HTML in memory and need to extract its readable text without fetching any URL. It is ideal for pipelines that scrape HTML externally and then need to clean it before passing to an LLM, summarizer, or search index. Prefer it over URL-based extractors when you control the fetching step yourself or when the HTML comes from a non-public source.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T06:37:48.264Z","isFirstParty":false,"canonicalSlug":"clean-text-extraction-from-html-fed2abc6"}