{"uid":"cap_WRbCOryi17-JRSU4IJaFh","slug":"aayat-ai-web-crawler-url-to-markdown-6d282a33","name":"Aayat AI Web Crawler — URL to Markdown","description":"Turn a small website, or one section of it, into clean Markdown: starts from your URL, finds pages from the sitemap or by following links on the same site, obeys robots.txt, and returns up to 10 pages (title, Markdown, address) in one call. Great for docs, blogs and company sites. ?url=https://example.com/docs/&maxPages=10&path=/docs/","url":"https://aayatai.com/crawl?utm_source=zero.xyz","method":"GET","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","method"],"properties":{"type":{"type":"string","const":"http"},"method":{"enum":["GET"],"type":"string"},"queryParams":{"type":"object","required":["url"],"properties":{"url":{"type":"string","format":"uri","maxLength":2048,"description":"Start page, e.g. https://example.com/docs/."},"path":{"type":"string","maxLength":300,"description":"Only crawl pages under this path (default: the start page's folder)."},"maxPages":{"type":"integer","default":5,"maximum":10,"minimum":1,"description":"Most pages to return (1-10)."}}}},"additionalProperties":false},"output":{"type":"object","required":["type"],"properties":{"type":{"type":"string"},"example":{"type":"object","required":["start","pages","count","discoveredFrom","robots","complete"],"properties":{"count":{"type":"integer"},"pages":{"type":"array","items":{"type":"object","properties":{"url":{"type":"string"},"chars":{"type":"integer"},"title":{"type":"string"},"markdown":{"type":"string"},"truncated":{"type":"boolean"}}}},"start":{"type":"string"},"trust":{"type":"object","description":"Third-party text, cleaned: read trust.notice; removed = what we stripped."},"robots":{"type":"string","description":"robots.txt status: ok, none or unreachable."},"skipped":{"type":"array","items":{"type":"object"},"description":"Pages not read and why (robots.txt, error, not HTML)."},"complete":{"type":"boolean","description":"False if more matching pages were found than returned."},"crawledAt":{"type":"string"},"discoveredFrom":{"enum":["sitemap","links"],"type":"string"}}}}}}},"responseSchema":{"type":"json","example":{"count":1,"pages":[{"url":"https://example.com/","chars":64,"title":"Example Domain","markdown":"# Example Domain\n\nThis domain is for use in illustrative examples.","truncated":false}],"start":"https://example.com/","robots":"ok","skipped":[{"url":"https://example.com/admin","reason":"robots.txt"}],"complete":true,"crawledAt":"2026-09-28T12:00:00.000Z","discoveredFrom":"links"}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.03","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.03/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.03","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.03","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_I0ZEzR9dXTETebDybFaDa","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.03","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Crawls a website or section of it starting from a given URL and returns up to 10 pages as clean Markdown, respecting robots.txt and staying within an optional path prefix.","exampleAgentPrompt":"Can you crawl https://stripe.com/docs/api/ and give me the content of up to 5 pages under /docs/api/ as clean Markdown?","exampleUseCases":[{"title":"RAG ingestion from docs site","prompt":"Fetch up to 10 pages from https://docs.langchain.com/docs/ — only pages under /docs/ — and return them as Markdown so I can load them into my vector database."},{"title":"Competitive research on company blog","prompt":"Crawl https://openai.com/blog/ and return up to 8 posts as Markdown text so I can summarize what they've been writing about lately."},{"title":"Onboarding knowledge base extraction","prompt":"Pull the content from https://help.notion.so/hc/en-us, staying under /hc/en-us/, and give me up to 10 pages as Markdown — I want to use it to answer user questions."}],"resultDescription":"A JSON object containing: the start URL, number of pages returned, how pages were discovered (sitemap or link-following), robots.txt status, a list of page objects each with URL, title, Markdown content, character count, and truncation flag, a list of skipped URLs with reasons, a completeness flag indicating if more pages existed than were returned, and a crawl timestamp.","failureModes":["Invalid or unreachable start URL returns an error or empty pages array","robots.txt disallows crawling, resulting in all pages skipped","Site has no sitemap and no followable links, returning only the start page","maxPages set to 0 or above 10 triggers validation error","Path prefix filters out all discovered pages, returning empty results","Non-HTML content (PDFs, images) is skipped and listed in the skipped array","Network timeouts on slow or firewalled sites may result in partial or empty results"],"whenToPreferThis":"Choose this endpoint when you need to ingest multi-page website content as clean Markdown in a single API call, especially for documentation sites, blogs, or company knowledge bases. It is ideal for RAG pipelines, LLM context loading, or competitive research where you want structured text from several related pages under a common URL path. Prefer it over single-page scrapers when you need up to 10 pages at once and want automatic sitemap discovery, robots.txt compliance, and link-following built in.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T00:46:19.055Z","isFirstParty":false,"canonicalSlug":"aayat-ai-web-crawler-url-to-markdown-6d282a33"}