{"uid":"cap_PJvlyrwnywzF_4WboUYpM","slug":"robots-txt-parser-rules-sitemaps-ai-crawler-blocks-and-path-allowed-a48f69e4","name":"robots.txt Parser: Rules, Sitemaps, AI-Crawler Blocks, and Path-Allowed Check","description":"robots.txt parser: rules, sitemaps, AI-crawler blocks, and is-this-path-allowed check — Genesis402 / UnyKorn Operator Network","url":"https://twin.unykorn.org/web/robots?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"params":{"type":"object","properties":{"path":{"type":"string","description":"optional path to check, e.g. /blog/"},"site":{"type":"string","description":"required: domain or URL"},"user_agent":{"type":"string","description":"optional, default *"}}}}},"responseSchema":{"type":"json","example":{"ok":true,"type":"robots-txt","receipt":{"tx_hash":"0x<64hex>","amount_usd":0.001,"receipt_id":"g402-<16hex>"},"sources":[{"ok":true,"name":"<source>"}],"limitations":"<text>","generated_at":"<iso time>","evidence_hash":"sha256:<64hex>"}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.001","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.001/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.001","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.001","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_AYsVavFOGODTqbkw5MSXE","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.001","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Fetches and parses a site's robots.txt, returning crawl rules, sitemaps, AI-crawler blocks, and whether a specific path is allowed for a given user-agent.","exampleAgentPrompt":"Check robots.txt for example.com and tell me whether /blog/ is allowed for the default user-agent, and list any sitemaps and AI-crawler blocks you find.","exampleUseCases":[{"title":"AI crawler policy audit","prompt":"Can you check robots.txt on nytimes.com and tell me if GPTBot or CCBot are blocked, and what rules they have?"},{"title":"Pre-scrape path permission check","prompt":"Before I scrape shopify.com/collections, can you check their robots.txt to see if that path is allowed for a generic crawler?"},{"title":"Sitemap discovery from robots.txt","prompt":"I need to find all the sitemaps listed in the robots.txt for wikipedia.org — can you pull those for me?"}],"resultDescription":"Returns the parsed robots.txt contents including crawl rules per user-agent, disallow and allow directives, sitemap URLs, whether specific AI crawlers are blocked, and a boolean indicating if the requested path is permitted for the specified user-agent.","failureModes":["robots.txt not found (404) — site may not have one, returns empty or error","network timeout if the target domain is unreachable","malformed robots.txt on the target site may produce incomplete parsing","rate limiting by the target domain when fetching robots.txt","invalid domain or URL input causes a parameter error"],"whenToPreferThis":"Use this endpoint when an agent needs to programmatically determine crawl permissions for any website before scraping, to audit AI-crawler policies across sites, or to discover sitemap URLs. Prefer this over manual fetching because it handles parsing, normalization, and AI-specific block detection automatically.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-01T18:48:02.575Z","isFirstParty":false,"canonicalSlug":"robots-txt-parser-rules-sitemaps-ai-crawler-blocks-and-path-allowed-a48f69e4"}