{"uid":"cap_mOpscCJCVsADhVK6jfYgL","slug":"402utils-com-entity-extractor-fd8c091e","name":"402utils.com Entity Extractor","description":"Extract emails, http(s) URLs, international phone numbers, @mentions and #hashtags from provided text OR html — normalized and de-duplicated. From html, also reads href/src attributes and mailto: links. Emails are syntax-validated; phones use Google libphonenumber (international format). Pattern-based extraction, NOT ML named-entity recognition. What a crawl agent pulls from every page. Processes provided content only; nothing is fetched.","url":"https://402utils.com/v1/extract-entities","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"html":{"type":"string","description":"HTML to scan (attributes + visible text)."},"text":{"type":"string","description":"Plain text to scan."}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.002","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.002/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.002","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.002","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_yX42l7FupaibH6LGnB7KH","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.002","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Extracts and deduplicates emails, URLs, phone numbers, @mentions, and #hashtags from plain text or HTML content using pattern-based parsing.","exampleAgentPrompt":"Pull out all the emails, phone numbers, URLs, @mentions, and hashtags from this HTML snippet for me — normalize and deduplicate them: '<div>Contact us at info@example.com or call +1-800-555-0199. Follow us @exampleco #sale. See https://example.com/offer</div>'","exampleUseCases":null,"resultDescription":"Returns a structured set of deduplicated, normalized entities found in the input: validated email addresses, full HTTP/HTTPS URLs (including those from href/src attributes and mailto links), international-format phone numbers (via Google libphonenumber), @mention strings, and #hashtag strings.","failureModes":["No text or html field provided — returns empty or error","Malformed HTML may cause some attributes to be missed","Phone numbers in non-international ambiguous formats may be skipped","Very large input payloads may hit size limits","Pattern-based extraction may miss emails/phones embedded in images or non-standard encodings"],"whenToPreferThis":"Choose this endpoint when you need fast, rule-based extraction of structured contact and social entities from raw text or HTML without ML overhead. Ideal for crawl pipelines, contact harvesting, link extraction, or preprocessing scraped pages. Prefer this over ML NER-based tools when you need high precision on well-formatted data and predictable, deterministic output.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-15T12:53:12.317Z","isFirstParty":false}