{"uid":"cap_npIaBGqHqxJzD2VmyfM7K","slug":"apex-faucet-site-extract-api-30222e47","name":"APEX Faucet Site Extract API","description":"Website crawler: up to 25 pages of one site as clean text, robots.txt obeyed. Render a whole section of a site - up to 25 pages - in a real browser and return every page as clean text.","url":"https://apexfaucet.xyz/api/x402/site-extract?utm_source=zero.xyz","method":"GET","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","method","queryParams"],"properties":{"type":{"type":"string","const":"http"},"method":{"enum":["GET"],"type":"string"},"queryParams":{"type":"object","required":["url"],"properties":{"url":{"type":"string","description":"The page to start from. Only pages on this same host are fetched, and robots.txt is obeyed."},"full":{"type":"string","description":"Set to 1 to render every planned page ($0.14 over x402). Programs pay for every call, with or without it; only a browser sees the crawl plan without it."},"pages":{"type":"string","pattern":"^[0-9]{1,2}$","description":"How many pages to render, default 10, hard cap 25 (a query string, so digits as text)."}}}}},"output":{"type":"object"}}},"responseSchema":{"type":"json","schema":{"type":"object","properties":{"ok":{"type":"boolean"},"data":{"type":"object","properties":{"host":{"type":"string"},"pages":{"type":"array"},"start":{"type":"string"},"failed":{"type":"array"},"robots":{"type":"object"},"pageCount":{"type":"integer"},"totalCharacters":{"type":"integer"}}},"paid":{"type":"object"}}},"example":{"ok":true,"data":{"host":"docs.arc.io","pages":[{"url":"https://docs.arc.io/","title":"Welcome to Arc docs","characters":1384}],"start":"https://docs.arc.io/","pageCount":5,"totalCharacters":28012},"paid":{"asset":"USDC","amount":1,"network":"eip155:5042"}}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.14","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.14/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.14","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.14","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_JzWsYA8TI1swIZ38eB9Ru","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.14","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Crawls and extracts text content from pages on a given host, returning structured page data including titles, URLs, and character counts","exampleAgentPrompt":"Can you crawl docs.arc.io for me and extract the text content from up to 10 pages, starting from the homepage?","exampleUseCases":[{"title":"Documentation indexing for AI agent","prompt":"Pull all the text content from docs.myproject.io — up to 15 pages starting from the homepage — so I can feed it into my knowledge base."},{"title":"Competitive site content research","prompt":"Crawl competitor.io and extract the text from up to 20 pages so I can analyze what topics they cover on their website."},{"title":"Single-page article extraction","prompt":"Fetch the content of https://blog.example.com/article/intro and extract the readable text from that page."}],"resultDescription":"A JSON object with ok status, and a data object containing the host, start URL, an array of pages (each with URL, title, and character count), a list of failed URLs, robots.txt metadata, total page count, and total character count across all fetched pages.","failureModes":["URL is not reachable or returns non-200 status","robots.txt disallows crawling, resulting in zero pages fetched","Host mismatch — only pages on the same host as the start URL are fetched","Page count exceeds hard cap of 25, capped silently","Payment of 0.14 USDC not completed, resulting in 402 error","Invalid or malformed URL input causes error response"],"whenToPreferThis":"Use this endpoint when you need to extract multi-page text content from a website in a structured format, especially when you want host-scoped crawling with robots.txt compliance. Prefer this over generic scraping tools when you need page-level metadata (title, URL, character count) and want to respect crawling rules automatically. It is particularly useful for indexing documentation sites or gathering site content for downstream analysis by an AI agent.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-03T00:43:06.136Z","isFirstParty":false,"canonicalSlug":"apex-faucet-site-extract-api-30222e47"}