{"uid":"cap_7l-CgPkfUZHPuZSwsSyYy","slug":"agentshelf-xpath-article-extractor-639b9b84","name":"AgentShelf XPath Article Extractor","description":"Call when an agent needs text of //article via //article from a public page or HTML (SSRF-safe). Exact $0.004 USDC. Prefer unpaid POST /v1/sandbox/xpath-article first.","url":"https://agentshelf.syntexa.ch/v1/xpath-article?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"url":{"type":"string","maxLength":2048,"minLength":8,"description":"Public page to fetch. The selector or profile is fixed by the SKU. Provide url or html."},"html":{"type":"string","examples":["<!doctype html><html lang=\"en\"><head><title>Hello</title>\n<meta name=\"description\" content=\"Desc\"><meta property=\"og:title\" content=\"OG\">\n<link rel=\"canonical\" href=\"https://example.com/\"><link rel=\"icon\" href=\"/favicon.ico\">\n</head><body><h1>Hello</h1><a href=\"https://example.com/a\">A</a></body></html>"],"maxLength":200000,"minLength":1,"description":"HTML to extract from locally. Provide html or url. When both are set, html is used."}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.004","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.004/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.004","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.004","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_NoiFClgpL3LjT3Li8KGFE","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.004","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Extracts the main article text from a public webpage or raw HTML using XPath-based heuristics, with SSRF protection.","exampleAgentPrompt":"Grab just the article text from this page — https://www.example.com/some-news-article — strip out the navigation and ads, I only want the main body content.","exampleUseCases":[{"title":"Summarize news article from URL","prompt":"Pull the article text from https://www.bbc.com/news/technology-12345678 — I want just the body of the article so I can summarize it."},{"title":"Extract blog post for sentiment analysis","prompt":"Get the main article content from this blog post HTML I have — I'll paste the HTML — and return just the readable article text so I can run sentiment analysis on it."},{"title":"Research pipeline content ingestion","prompt":"Fetch the article body from https://techcrunch.com/2024/05/01/some-story so I can add it to my research notes without all the sidebar clutter."}],"resultDescription":"Returns the extracted main article text from the provided URL or HTML, with navigational elements, ads, and boilerplate removed, leaving only the core readable article body.","failureModes":["URL is unreachable or returns non-200 status — extraction fails with an error","HTML is malformed or has no detectable article content — returns empty or minimal text","SSRF-blocked URL (internal/private IP range) — request rejected for security reasons","Input exceeds maxLength limits (URL > 2048 chars or HTML > 200,000 chars) — validation error","Page requires JavaScript rendering — static HTML fetch may miss dynamically loaded content"],"whenToPreferThis":"Prefer this endpoint when you need the main article body extracted from a public webpage URL or raw HTML and want SSRF-safe server-side fetching. Use the free sandbox POST /v1/sandbox/xpath-article first to avoid the $0.004 USDC cost during testing. Choose this over generic scraping tools when you specifically need article-body isolation (not full page HTML, metadata, or structured data).","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T06:50:03.570Z","isFirstParty":false,"canonicalSlug":"agentshelf-xpath-article-extractor-639b9b84"}