{"uid":"cap_qp7GizETGRmy5xCjSpOyh","slug":"apify-jwhre9stu-merit-systems-vercel-app-f19c1d72","name":"Apify RAG Web Browser Actor","description":"Start the \"RAG Web Browser\" Apify actor (apify/rag-web-browser). Web search and fetch tool for AI agents and RAG pipelines. It queries Google Search, scrapes the top N pages using a full web browser, and returns their conten… Returns a signed token for polling /api/actors/status (free SIWX) and fetching /api/actors/results (paid on first call per wallet, then free with SIWX replay).","url":"https://apify-jwhre9stu-merit-systems.vercel.app/api/actors/apify/rag-web-browser/call","method":"POST","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","method","bodyType","body"],"properties":{"body":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","properties":{"input":{"type":"object","required":["query","maxResults","outputFormats","requestTimeoutSecs","serpProxyGroup","serpMaxRetries","proxyConfiguration","scrapingTool","removeElementsCssSelector","htmlTransformer","desiredConcurrency","maxRequestRetries","dynamicContentWaitSecs","removeCookieWarnings","debugMode"],"properties":{"query":{"type":"string","pattern":"[^\\s]+","description":"Enter Google Search keywords or a URL of a specific web page. The keywords might include the [advanced search operators](https://blog.apify.com/how-to-scrape-google-like-a-pro/). Examples: - san francisco weather - https://www.cnn.com - function calling site:openai.com"},"debugMode":{"type":"boolean","default":false,"description":"If enabled, the Actor will store debugging information into the resulting dataset under the `debug` field."},"maxResults":{"type":"integer","default":3,"maximum":100,"minimum":1,"description":"The maximum number of top organic Google Search results whose web pages will be extracted. If `query` is a URL, then this field is ignored and the Actor only fetches the specific web page."},"scrapingTool":{"enum":["browser-playwright","raw-http"],"type":"string","default":"raw-http","description":"Select a scraping tool for extracting the target web pages. The Browser tool is more powerful and can handle JavaScript heavy websites, while the Plain HTML tool can't handle JavaScript but is about two times faster."},"outputFormats":{"type":"array","items":{"enum":["text","markdown","html"],"type":"string"},"default":["markdown"],"description":"Select one or more formats to which the target web pages will be extracted and saved in the resulting dataset."},"serpMaxRetries":{"type":"integer","default":2,"maximum":5,"minimum":0,"description":"The maximum number of times the Actor will retry fetching the Google Search results on error. If the last attempt fails, the entire request fails."},"serpProxyGroup":{"enum":["GOOGLE_SERP","SHADER"],"type":"string","default":"GOOGLE_SERP","description":"Enables overriding the default Apify Proxy group used for fetching Google Search results."},"htmlTransformer":{"type":"string","default":"none","description":"Specify how to transform the HTML to extract meaningful content without any extra fluff, like navigation or modals. The HTML transformation happens after removing and clicking the DOM elements. - **None** (default) - Only removes the HTML elements specified via 'Remove HTML elements' option. - **Readable text** - Extracts the main contents of the webpage, without navigation and other fluff."},"maxRequestRetries":{"type":"integer","default":1,"maximum":3,"minimum":0,"description":"The maximum number of times the Actor will retry loading the target web page on error. If the last attempt fails, the page will be skipped in the results."},"desiredConcurrency":{"type":"integer","default":5,"maximum":50,"minimum":0,"description":"The desired number of web browsers running in parallel. The system automatically scales the number based on the CPU and memory usage. If the initial value is `0`, the Actor picks the number automatically based on the available memory."},"proxyConfiguration":{"type":"object","default":{"useApifyProxy":true},"properties":{"proxyUrls":{"type":"array","items":{"type":"string"}},"useApifyProxy":{"type":"boolean"},"apifyProxyGroups":{"type":"array","items":{"type":"string"}},"apifyProxyCountry":{"type":"string"}},"description":"Apify Proxy configuration used for scraping the target web pages.","additionalProperties":{}},"requestTimeoutSecs":{"type":"integer","default":40,"maximum":300,"minimum":1,"description":"The maximum time in seconds available for the request, including querying Google Search and scraping the target web pages. For example, OpenAI allows only [45 seconds](https://platform.openai.com/docs/actions/production#timeouts) for custom actions. If a target page loading and extraction exceeds this timeout, the corresponding page will be skipped in results to ensure at least some results are returned within the timeout. If no page is extracted within the timeout, the whole request fails."},"removeCookieWarnings":{"type":"boolean","default":true,"description":"If enabled, the Actor attempts to close or remove cookie consent dialogs to improve the quality of extracted text. Note that this setting increases the latency."},"dynamicContentWaitSecs":{"type":"integer","default":10,"maximum":9007199254740991,"minimum":-9007199254740991,"description":"The maximum time in seconds to wait for dynamic page content to load. The Actor considers the web page as fully loaded once this time elapses or when the network becomes idle."},"removeElementsCssSelector":{"type":"string","default":"nav, footer, script, style, noscript, svg, img[src^='data:'],\n[role=\"alert\"],\n[role=\"banner\"],\n[role=\"dialog\"],\n[role=\"alertdialog\"],\n[role=\"region\"][aria-label*=\"skip\" i],\n[aria-modal=\"true\"]","description":"A CSS selector matching HTML elements that will be removed from the DOM, before converting it to text, Markdown, or saving as HTML. This is useful to skip irrelevant page content. The value must be a valid CSS selector as accepted by the `document.querySelectorAll()` function. By default, the Actor removes common navigation elements, headers, footers, modals, scripts, and inline image. You can disable the removal by setting this value to some non-existent CSS selector like `dummy_keep_everything`."}},"description":"Normalized actor input. See field descriptions for the simplified shape; native Apify input also accepted (passed through).","additionalProperties":{}},"options":{"type":"object","properties":{"build":{"type":"string","maxLength":100,"minLength":1},"memory":{"type":"integer","maximum":32768,"minimum":128},"maxItems":{"type":"integer","maximum":100000,"minimum":1}},"description":"Optional Apify run options.","additionalProperties":false}},"additionalProperties":false},"type":{"type":"string","const":"http"},"method":{"enum":["POST","PUT","PATCH"],"type":"string"},"bodyType":{"enum":["json","form-data","text"],"type":"string"}},"additionalProperties":false}}},"responseSchema":null,"example":{"request":{"input":{"query":"artificial intelligence","maxResults":3,"scrapingTool":"raw-http","outputFormats":["markdown"],"htmlTransformer":"none","requestTimeoutSecs":40}},"response":{"run":{"id":"G1ebqaXeElrKSe957","status":"READY","actorId":"3ox4R101TgZz67sLr","startedAt":"2026-05-29T04:46:22.166Z","usageTotalUsd":0.00005,"defaultDatasetId":"xNxmOPvjUWuRSBbwo"},"token":"eyJhbGciOiJIUzI1NiJ9.eyJydW5JZCI6IkcxZWJxYVhlRWxyS1NlOTU3IiwiYWN0b3JJZCI6ImFwaWZ5L3JhZy13ZWItYnJvd3NlciIsImRlZmF1bHREYXRhc2V0SWQiOiJ4TnhtT1B2alVXdVJTQmJ3byIsInciOiIweDljYzQyZjNkOTI0NWI4NjdhY2NjZDYzMGI0M2Y5MDZjMTY2NWIxNzYiLCJpYXQiOjE3ODAwMjk5ODIsImV4cCI6MTc4MjYyMTk4Mn0.L8gsz2vpra6vl_PgtVniX70L1ydUfucLHxxwmw_NXE0"}},"exampleRequest":{"input":{"query":"artificial intelligence","maxResults":3,"scrapingTool":"raw-http","outputFormats":["markdown"],"htmlTransformer":"none","requestTimeoutSecs":40}},"tags":["x402"],"displayCostAmount":"0.01","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"settled","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.01/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_nYH6AHhbjbFFeHGO5Mn6X","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.01","costPer":"request","priority":0,"asset":null,"unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Performs Google searches and scrapes the top N result pages using a full browser, returning page content for use in AI/RAG pipelines","exampleAgentPrompt":"Search the web for 'best practices for LLM prompt engineering' and scrape the top 5 result pages, returning the full text content so I can use it as context in my RAG pipeline.","exampleUseCases":[{"title":"Real-time competitor pricing research","prompt":"Search for the current pricing pages of our top 3 competitors and scrape their full offerings so I can analyze how we compare on cost and features."},{"title":"Gather latest industry news summaries","prompt":"Search for 'AI regulations 2024' and fetch the complete text from the top 7 news articles so my knowledge base stays current with the latest developments."},{"title":"Build FAQ knowledge base from web","prompt":"Search Google for 'common questions about API rate limiting' and scrape the full content from the top 10 results so I can compile a comprehensive FAQ for customer support."}],"resultDescription":"Returns a signed token used to poll /api/actors/status for job completion and fetch /api/actors/results for the scraped page content from the top N Google search results, including full text extracted by a real browser.","failureModes":["Search query returns no results — empty results array","Target pages block scraping — partial or empty content for those URLs","Actor timeout if pages are slow to load — incomplete results","Invalid or missing query parameter — 400 error","Payment/wallet issues — 402 error on first call per wallet","Polling token expires before results are ready — re-fetch required"],"whenToPreferThis":"Use this endpoint when an AI agent or RAG pipeline needs real, current web content from multiple pages for a given search query, especially when JavaScript rendering is required. Prefer over simple HTTP fetchers when pages require a full browser to render, or when you need Google Search integration combined with scraping in a single call.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-16T04:06:24.923Z","isFirstParty":false}