{"uid":"cap_gvjMpkV2ZNz4z-qosHJiQ","slug":"agentshelf-json-ld-dataset-extractor-48caf258","name":"AgentShelf JSON-LD Dataset Extractor","description":"Call when an agent needs JSON-LD @type Dataset extracted from a public page or HTML (SSRF-safe). Exact $0.01 USDC. Prefer unpaid POST /v1/sandbox/profile-jsonld-dataset first.","url":"https://agentshelf.syntexa.ch/v1/profile-jsonld-dataset?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"url":{"type":"string","maxLength":2048,"minLength":8,"description":"Public page to fetch. The selector or profile is fixed by the SKU. Provide url or html."},"html":{"type":"string","examples":["<!doctype html><html lang=\"en\"><head><title>Hello</title>\n<meta name=\"description\" content=\"Desc\"><meta property=\"og:title\" content=\"OG\">\n<link rel=\"canonical\" href=\"https://example.com/\"><link rel=\"icon\" href=\"/favicon.ico\">\n</head><body><h1>Hello</h1><a href=\"https://example.com/a\">A</a></body></html>"],"maxLength":200000,"minLength":1,"description":"HTML to extract from locally. Provide html or url. When both are set, html is used."}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.01","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.01/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_c1o3qeT3i1h5gUHjj5Ft5","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.01","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Extracts JSON-LD @type Dataset structured data from a public webpage URL or raw HTML, returning the parsed schema.org Dataset objects found on the page.","exampleAgentPrompt":"Can you pull out any JSON-LD Dataset structured data from this page — https://data.gov/dataset/air-quality-2023 — and tell me what dataset metadata it contains like the title, description, and keywords?","exampleUseCases":[{"title":"Research data catalog enrichment","prompt":"I need you to extract the JSON-LD Dataset metadata from this data catalog page — https://zenodo.org/record/7890123 — so I can get the structured title, description, and author information without manually parsing the HTML."},{"title":"Open data portal metadata harvest","prompt":"Fetch the schema.org Dataset structured data from https://opendata.cityofnewyork.us/dataset/street-closures and give me whatever JSON-LD @type Dataset objects are embedded on that page."},{"title":"Extract dataset info from raw HTML","prompt":"I've got some HTML from a data publisher's page — can you pull out any JSON-LD Dataset annotations from it? Here's the HTML: <html>...<script type='application/ld+json'>...</script>...</html>"}],"resultDescription":"A parsed representation of all JSON-LD @type Dataset objects found within the page at the given URL or in the provided HTML, including schema.org properties such as name, description, keywords, license, creator, distribution, and other dataset metadata fields present in the markup.","failureModes":["URL points to a private or non-public IP address (SSRF blocked, request rejected)","Page contains no JSON-LD Dataset markup (empty result returned)","URL is unreachable or returns non-200 HTTP status","HTML input exceeds 200,000 character limit","URL exceeds 2,048 character limit","Payment of $0.01 USDC not attached or insufficient (402 response)","Malformed URL input rejected by schema validation"],"whenToPreferThis":"Use this endpoint when you need to programmatically extract machine-readable schema.org Dataset structured data embedded as JSON-LD in a webpage — ideal for enriching data catalog pipelines, harvesting open data portal metadata, or parsing research dataset descriptions without relying on LLM-based extraction. Prefer the free /v1/sandbox/profile-jsonld-dataset endpoint first for testing. Choose this over general-purpose scrapers when you specifically need @type Dataset JSON-LD objects in a structured, parsed form.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T04:54:07.166Z","isFirstParty":false,"canonicalSlug":"agentshelf-json-ld-dataset-extractor-48caf258"}