{"uid":"cap_l5qRB3e0WOS_xK-sUElK1","slug":"olostep-web-page-scrape-via-x402-orthogonal-com-88681185","name":"Olostep Web Page Scrape via x402.orthogonal.com","description":"Initiate a web page scrape","url":"https://x402.orthogonal.com/olostep/v1/scrapes","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"parser":{"type":"object","properties":{"id":{"type":"string"}},"description":"When defining json as a format, you can use this parameter to specify the parser to use. Parsers are useful to extract structured content from web pages. Olostep has a few parsers built in for most common web pages, and you can also create your own parsers."},"actions":{"type":"object","properties":{"type":{"type":"string"},"milliseconds":{"type":"number"}},"description":"Actions to perform on the page before getting the content."},"country":{"type":"string","description":"Residential country to load the request from. Supported values are: * US (United States) * CA (Canada) * IT (Italy) * IN (India) * GB (England) * JP (Japan) * MX (Mexico) * AU (Australia) * ID (Indonesia) * UA (UAE) * RU (Russia) * RANDOM Some operations, like scraping Google Search and Google News, support all countries."},"formats":{"type":"array","items":{"type":"string"},"description":"Formats in which you want the content."},"metadata":{"type":"object","description":"User-defined metadata. Not supported yet"},"llm_extract":{"type":"object","properties":{"schema":{"type":"object"}}},"screen_size":{"type":"object","properties":{"screen_type":{"type":"string"},"screen_width":{"type":"number"},"screen_height":{"type":"number"}},"description":"Configuration for screen size. Preset dimensions are available through screen_type: desktop (1920x1080), mobile (414x896), or default (768x1024)."},"transformer":{"type":"string","description":"Specify the HTML transformer to use, if any. Postlight's Mercury Parser library is used to remove ads and other unwanted content from the scraped content. Available options: `postlight`, `none`"},"links_on_page":{"type":"object","properties":{"exclude_links":{"type":"array","items":{"type":"string"}},"include_links":{"type":"array","items":{"type":"string"}},"absolute_links":{"type":"boolean"},"query_to_order_links_by":{"type":"string"}},"description":"With this option, you can get all the links present on the page you scrape."},"remove_images":{"type":"boolean","description":"Option to remove images from the scraped content. Defaults to false."},"url_to_scrape":{"type":"string","description":"The URL to start scraping from."},"remove_class_names":{"type":"array","items":{"type":"string"},"description":"List of class names to remove from the content."},"remove_css_selectors":{"type":"string","description":"Option to remove certain CSS selectors from the content. Optionally, you can also pass a JSON stringified array of specific selectors you want to remove. The CSS selectors removed when this option is set to default are ['nav','footer','script','style','noscript','svg',[role=alert],[role=banner],[role=dialog],[role=alertdialog],[role=region][aria-label*=skip i],[aria-modal=true]] Available options: `default`, `none`, `array`"},"wait_before_scraping":{"type":"integer","description":"Time to wait in milliseconds before starting the scraping."}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.01","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.01/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_2mf79Es_OTOdGixCcJMvY","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.01","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Initiates a web page scrape using Olostep, returning page content in configurable formats with optional HTML transformation, structured data extraction, and geo-targeted residential proxies","exampleAgentPrompt":"Scrape the page at https://example.com/article for me and return the content as markdown, removing ads using the Postlight transformer, loading from a US residential IP on desktop screen size.","exampleUseCases":[{"title":"Geo-targeted competitor price scraping","prompt":"Scrape https://shop.competitor.com/products and return the content as JSON using my custom price-extraction schema, loading from a UK residential IP so I get the local pricing."},{"title":"Clean article extraction for summarization","prompt":"Fetch the article at https://www.nytimes.com/2024/05/01/tech/ai-update.html as markdown with the Postlight transformer to strip out ads and navigation, then I'll summarize it."},{"title":"Google Search results page scraping","prompt":"Scrape a Google Search results page for the query 'best CRM software 2024' using a US residential proxy and return all the links on the page so I can see who's ranking."}],"resultDescription":"Returns the scraped page content in the requested format(s) — e.g. raw HTML, markdown, or structured JSON — along with optional extracted links, LLM-extracted structured fields, and metadata about the scrape. The response format depends on which 'formats' were requested and whether a parser or llm_extract schema was provided.","failureModes":["Target URL is unreachable or returns a non-200 HTTP status","Country value is unsupported or misspelled, causing a 400 error","Requested parser ID does not exist or is misconfigured","Page requires JavaScript rendering not supported by the action config","LLM extraction schema is malformed or too complex","Payment of $0.01 USDC fails or x402 payment header is missing/invalid","Scrape times out due to slow target server","Anti-bot detection on target site blocks the request"],"whenToPreferThis":"Choose this endpoint when you need to scrape a single URL with fine-grained control over output format, residential proxy country, screen size, HTML transformation (ad removal via Postlight), structured data extraction via a parser or LLM schema, and link filtering. It is particularly well-suited for geo-sensitive scraping (e.g. price comparison across regions) and for extracting clean, readable article content. Prefer it over raw HTTP fetching when you need residential IP rotation, JavaScript-rendered page support, or built-in content cleaning.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-13T17:41:20.979Z","isFirstParty":false}