{"uid":"cap_lkMaR6NvV1YHNhBO3CULI","slug":"x402-forgemesh-io-robots-txt-ai-permission-checker-83f13163","name":"x402.forgemesh.io Robots.txt AI Permission Checker","description":"Fetches a domain's robots.txt and parses its Content-Signal and AI-preference directives (search, ai-input, ai-train) into structured JSON plus a plain-English summary of what's allowed — the \"can this domain be crawled by an AI agent\" check. Run before scraping content or building a RAG index, ahead of the industry's move toward stricter bot-gating defaults later this year.","url":"https://x402.forgemesh.io/robots-txt-ai-check","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"domain":{"type":"string","description":"Domain to check, e.g. example.com"}}},"responseSchema":{"type":"json","example":{"domain":"theverge.com","signals":{"search":"yes","ai-input":"no","ai-train":"no"},"declared":true,"interpretation":"site declares: search=yes, ai-input=no, ai-train=no"}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.005","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.005/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.005","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.005","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_uHbGI196Y0wAHdBp0Yx70","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.005","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Fetches and parses a domain's robots.txt file, extracting AI-specific directives (search, ai-input, ai-train) into structured JSON with a plain-English summary of what scraping or training is permitted.","exampleAgentPrompt":"Before I scrape techcrunch.com for my RAG index, check its robots.txt to see whether AI crawling, ai-input use, and ai-train use are actually permitted — give me both the structured breakdown and a plain-English summary.","exampleUseCases":null,"resultDescription":"Returns structured JSON containing parsed robots.txt directives for Content-Signal and AI-preference fields (search, ai-input, ai-train), along with a plain-English summary of what is allowed or disallowed for AI agents, scrapers, and indexers.","failureModes":["Domain does not have a robots.txt file — returns empty or default permissive interpretation","Domain is unreachable or returns non-200 — fetch error reported","Malformed robots.txt with unrecognized directive syntax — partial parse with warnings","Invalid domain format in input — validation error returned","No AI-specific directives present — summary reflects generic crawl rules only"],"whenToPreferThis":"Use this endpoint as a pre-crawl compliance check before any web scraping, RAG indexing, or AI training data collection workflow. It is specifically designed to parse emerging AI-preference and Content-Signal directives that generic robots.txt parsers do not handle, making it the right choice when you need to confirm AI-specific permissions rather than just standard Googlebot rules.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-15T07:04:57.151Z","isFirstParty":false}