{"uid":"cap_AUQkjrLaipn091MG0xtjp","slug":"agentbit-web-extract-88f58c29","name":"AgentBit Web Extract","description":"Fetch any public URL and return clean, structured content: title, meta description, main text, markdown, headings, outbound links and page metadata. Removes scripts, navigation and boilerplate. Ideal for reading articles, docs and product pages before reasoning over them.","url":"https://agentbit.app/v1/web/extract","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"url":{"type":"string"},"include_links":{"type":"boolean"},"output_format":{"type":"string"},"render_fallback":{"type":"boolean"}}},"responseSchema":{"type":"json","example":{"url":"https://example.com","text":"Example Domain This domain is for use in illustrative examples in documents.","links":[{"url":"https://www.iana.org/domains/example","text":"More information..."}],"title":"Example Domain","headings":[{"text":"Example Domain","level":1}],"markdown":"# Example Domain\n\nThis domain is for use in illustrative examples…","metadata":{"viewport":"width=device-width, initial-scale=1"},"description":null}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.005","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.005/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.005","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.005","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_kDNAznLywfLp_7i_DV7Q1","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.005","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Fetches any public URL and returns clean, structured content including title, meta description, main text, markdown, headings, outbound links, and page metadata with boilerplate removed.","exampleAgentPrompt":"Fetch the content of https://docs.example.com/getting-started and return the main text in markdown format, including all outbound links.","exampleUseCases":[{"title":"Reading an article before summarizing","prompt":"Can you grab the full article at https://techcrunch.com/2024/05/01/openai-news/ and give me the clean main text so I can summarize it?"},{"title":"Extracting product page details","prompt":"Fetch https://www.apple.com/iphone-16-pro/ and pull out the title, main text, and any outbound links so I can compare it with other product pages."},{"title":"Parsing docs before answering questions","prompt":"Read the documentation at https://stripe.com/docs/payments/accept-a-payment and extract the content as markdown so I can reason over it and answer questions about the integration steps."}],"resultDescription":"A structured JSON object containing the page title, meta description, clean main body text, markdown-formatted content, headings list, outbound links array, and additional page metadata — with scripts, navigation elements, and boilerplate stripped out.","failureModes":["URL is not publicly accessible or returns a 4xx/5xx HTTP error","Paywalled or login-required content returns incomplete or no text","JavaScript-heavy single-page apps may return empty content if render_fallback is false","Malformed or non-HTTP URL input causes a validation error","Very large pages may be truncated or time out","Anti-scraping protections (Cloudflare, CAPTCHAs) may block extraction"],"whenToPreferThis":"Choose this endpoint when you need to read and reason over the content of a specific public webpage — such as an article, documentation page, or product listing — and want clean, structured output (title, text, markdown, links) without writing custom scraping logic. It's especially useful for preprocessing web content before summarization, Q&A, or comparison tasks. Prefer it over a generic web search when you already have the URL and need the full page content rather than a snippet.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-13T18:48:00.236Z","isFirstParty":false}