{"uid":"cap_HfHhADaX8jKDX4iLCVhxi","slug":"aayat-ai-web-page-extractor-25901259","name":"Aayat AI Web Page Extractor","description":"Web page extractor for AI agents: fetch any public URL and get clean Markdown, the page title and all links as JSON, with scripts, styles and menus removed. url is the page address (https://...). Private or internal addresses are blocked; pages over 2 MB or slower than 10 seconds are rejected and you are not charged.","url":"https://aayatai.com/extract?utm_source=zero.xyz","method":"GET","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","method"],"properties":{"type":{"type":"string","const":"http"},"method":{"enum":["GET"],"type":"string"},"queryParams":{"type":"object","required":["url"],"properties":{"url":{"type":"string","format":"uri","maxLength":2048,"description":"Full http(s) address of a public web page."}}}},"additionalProperties":false},"output":{"type":"object","required":["type"],"properties":{"type":{"type":"string"},"example":{"type":"object","required":["url","title","markdown","links"],"properties":{"url":{"type":"string","description":"Final address after redirects."},"links":{"type":"array","items":{"type":"object","required":["text","href"],"properties":{"href":{"type":"string"},"text":{"type":"string"}}}},"title":{"type":"string"},"trust":{"type":"object","description":"Third-party text, cleaned: read trust.notice; removed = what we stripped."},"markdown":{"type":"string","description":"Main page content as Markdown."}}}}}}},"responseSchema":{"type":"json","example":{"url":"https://example.com/","links":[{"href":"https://www.iana.org/domains/example","text":"More information..."}],"title":"Example Domain","markdown":"# Example Domain\n\nThis domain is for use in illustrative examples in documents..."}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.005","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.005/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.005","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.005","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_kzI6KrPOF_hVpxkkATLbm","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.005","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Fetches any public web page and returns clean Markdown content, the page title, and all links as structured JSON","exampleAgentPrompt":"Can you fetch the content of https://openai.com/blog and give me the main text as clean markdown along with all the links on the page?","exampleUseCases":[{"title":"Research article content extraction","prompt":"Go to https://www.bbc.com/news/technology-68012429 and pull out the full article text as clean markdown — I want to read it without all the navigation and ads."},{"title":"Competitor pricing page reader","prompt":"Fetch the page at https://stripe.com/pricing and extract all the text and links so I can see what their current plans and pricing look like."},{"title":"Documentation page ingestion","prompt":"Read the docs page at https://docs.python.org/3/library/asyncio.html and return the main content as markdown so I can feed it into my knowledge base."}],"resultDescription":"Returns a JSON object containing: the final URL after any redirects, the page title, the main body content converted to clean Markdown (scripts, styles, menus, and clutter stripped), and an array of all links found on the page (each with href and anchor text).","failureModes":["Private or internal IP addresses are blocked and return an error","Pages exceeding 2 MB are rejected and not charged","Pages taking longer than 10 seconds to load are rejected and not charged","Invalid or malformed URLs return a validation error","Non-public pages requiring authentication cannot be accessed","Paywalled or login-gated content may return partial or empty markdown"],"whenToPreferThis":"Choose this endpoint when an AI agent needs to read and process the textual content of a specific public web page — especially when clean, clutter-free markdown is needed for downstream LLM consumption. Ideal for single-page reads of articles, documentation, blogs, or product pages. Prefer this over raw HTTP fetching because it automatically strips navigation, scripts, styles, and ads. Not suitable for crawling entire sites, authenticated pages, or private/internal URLs.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T04:52:22.219Z","isFirstParty":false,"canonicalSlug":"aayat-ai-web-page-extractor-25901259"}