{"uid":"cap__UWvVe4tY7AGJw71wExw6","slug":"x402-orthogonal-com-web-crawl-structured-data-extractor-6279675d","name":"x402.orthogonal.com Web Crawl & Structured Data Extractor","description":"Crawl a website, use the provided JSON Schema and instructions to prioritize relevant internal links, and extract structured data from the selected pages.","url":"https://x402.orthogonal.com/context-dev/web/extract?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"url":{"type":"string","description":"The starting website URL to crawl and extract from. Must include http:// or https://."},"schema":{"type":"object","description":"JSON Schema for the returned data object."},"maxAgeMs":{"type":"integer","description":"Return cached scrape results if younger than this many milliseconds."},"maxDepth":{"type":"integer","description":"Optional maximum link depth from the starting URL (0 = only the starting page)."},"maxPages":{"type":"integer","description":"Maximum number of pages to analyze for extraction. Hard cap: 50. Defaults to 5."},"factCheck":{"type":"boolean","description":"When true, every returned value must be grounded in facts stated on the page."},"waitForMs":{"type":"integer","description":"Optional browser wait time in milliseconds after initial page load for each crawled page."},"stopAfterMs":{"type":"integer","description":"Soft time budget for the crawl in milliseconds. Min: 10000. Max: 110000. Default: 80000."},"instructions":{"type":"string","description":"Optional extraction guidance, such as which facts to prioritize or how to interpret fields."},"includeFrames":{"type":"boolean","description":"When true, iframe contents are included in Markdown before extraction."},"followSubdomains":{"type":"boolean","description":"When true, follow links on subdomains of the starting URL's domain."}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.03","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.03/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.03","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.03","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_Fk_slHIPJs6o8-jqvD3QD","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.03","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Crawls a website starting from a given URL, intelligently prioritizes relevant internal links using a JSON Schema and instructions, and extracts structured data from selected pages.","exampleAgentPrompt":"Crawl https://acme.com/products and extract a list of all products with their name, price, and description — look up to 3 levels deep, check up to 20 pages, and use the schema {products: [{name, price, description}]}; make sure every value is grounded in what's actually on the page.","exampleUseCases":[{"title":"Competitor pricing intelligence","prompt":"Crawl https://competitor.com/pricing and extract their plan names, monthly prices, and included features into structured JSON — go up to 2 levels deep and check up to 10 pages, and only return values that are actually stated on the page."},{"title":"Job listings aggregation","prompt":"Scrape https://careers.bigcorp.com and pull out every open job listing with its title, department, location, and apply link — go up to 2 levels deep, check up to 30 pages, and organize them as an array of job objects."},{"title":"Restaurant menu extraction","prompt":"Crawl https://bestpizza.com and extract the full menu with each item's name, category, and price into a structured JSON object — stick to just the starting page and up to 2 more levels deep, and verify every price is actually listed on the site."}],"resultDescription":"A JSON object conforming to the provided schema, populated with data extracted from the crawled pages. Values are grounded in content found on the site when factCheck is enabled. The response contains the structured fields as specified in the input schema.","failureModes":["URL is unreachable or returns non-200 status — crawl fails with no data","Schema is too complex or ambiguous — extraction may return partial or empty fields","maxPages cap of 50 hit before full coverage — some data may be missing","stopAfterMs budget exhausted — crawl stops early, returning partial results","Page requires JavaScript-heavy interaction beyond waitForMs — data not rendered","Site blocks crawlers via robots.txt or rate limiting — incomplete or failed extraction","factCheck enabled but values not found on page — fields returned as null or omitted"],"whenToPreferThis":"Choose this endpoint when you need to extract structured, schema-conformant data from an entire website or a multi-page section of one — not just a single URL. It is especially valuable when you need intelligent link prioritization (using instructions and a schema to guide which pages to follow), fact-grounded extraction, and control over crawl depth and page budget. Prefer it over a single-page scraper when the information is spread across multiple pages or when you want a structured JSON object rather than raw HTML or Markdown.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T02:04:36.050Z","isFirstParty":false,"canonicalSlug":"x402-orthogonal-com-web-crawl-structured-data-extractor-6279675d"}