{"uid":"cap_4RfaUdtrsaIyj43oiIXJP","slug":"crawlspur-robots-txt-allowance-check-for-news-documents-71dafb8f","name":"Crawlspur robots.txt Allowance Check for News Documents","description":"Prüft am Objekt news-dokument den Befund Freigabe durch robots.txt.","url":"https://crawlspur.halowerk.com/v1/pruef/crawlspur/robots-erlaubt/news-dokument","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"ziel":{"oneOf":[{"type":"string","maxLength":2048,"minLength":1,"description":"Öffentliche HTTP(S)-Adresse; bei DNS-Befunden Domain oder öffentliche IP."},{"type":"object","required":["adresse"],"properties":{"adresse":{"type":"string","maxLength":2048,"minLength":1},"selektor":{"type":"string","maxLength":240,"minLength":1,"description":"CSS-Selektor: genau ein Objekt, sonst erster Treffer des Objektfilters."},"vergleich":{"type":"string","maxLength":2048,"description":"Öffentliche Vergleichsadresse für Link-, Sitemap- oder Sprachbefunde."},"user_agent":{"type":"string","pattern":"^[A-Za-z0-9_-]{1,80}$"},"dkim_selektor":{"type":"string","pattern":"^[A-Za-z0-9_-]{1,63}$"}},"additionalProperties":false}]}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.001","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.001/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.001","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.001","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_7vZYM44fY7J-nzZc7fIwD","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.001","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Checks whether a news document URL is allowed to be crawled according to the site's robots.txt rules.","exampleAgentPrompt":"Can you check whether this news article URL https://example.com/news/article-123 is allowed to be crawled according to the site's robots.txt rules?","exampleUseCases":[{"title":"News indexing eligibility check","prompt":"I want to make sure this news article at https://publisher.com/news/breaking-story-2024 is not blocked by the site's robots.txt before I include it in my news aggregator index."},{"title":"SEO audit for news content","prompt":"I'm auditing our news website — can you check if https://mynewssite.com/articles/top-story is actually allowed by our robots.txt so I know it's eligible for Google News crawling?"},{"title":"Crawl policy monitoring for competitor news","prompt":"Can you verify whether the news page at https://competitor.com/news/market-update is permitted by their robots.txt? I need to know if their crawler is allowed to access it."}],"resultDescription":"Returns the robots.txt allowance finding for the specified news document — indicating whether the URL is permitted or blocked by the site's robots.txt directives, potentially including the specific rule matched and the effective crawl permission status for the given user agent.","failureModes":["Invalid or unreachable target URL returns an error","URL exceeds maximum length of 2048 characters","robots.txt file is unreachable or returns non-200 status","Malformed user agent string fails pattern validation","Target URL resolves to a non-public IP address and is rejected","Input object missing required 'adresse' field when using object form"],"whenToPreferThis":"Use this endpoint when you specifically need to verify whether a news document URL is permitted by a site's robots.txt, particularly for SEO auditing, news indexing pipelines, or crawl policy compliance checks. Prefer this over generic robots.txt checkers when the object type is explicitly a news document, as the endpoint is tuned for that content class.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-16T12:44:01.832Z","isFirstParty":false}