{"uid":"cap_0TFiTEe-enwNa4sSCUReR","slug":"forgemesh-vision-caption-generator-e53f56b6","name":"ForgeMesh Vision Caption Generator","description":"Visual caption generator: turns any image URL into a written description covering the scene, objects, people, and readable text within it. No images are stored during or after processing. A drop-in way for text-based agents and pipelines to gain image understanding for cataloging, moderation pre-checks, or accessibility workflows.","url":"https://x402.forgemesh.io/vision-caption-generator","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"image_url":{"type":"string","description":"Public URL of the image (jpg/png/webp, max 8MB)"}}},"responseSchema":{"type":"json","example":{"text":"A teal rectangular graphic with the words FORGEMESH UTILITY GRID in bold white capital letters centered on it.","model":"moondream"}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.01","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.01/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_km91kChvbA0afDGR-t2Nb","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.01","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Converts any public image URL into a detailed written description covering scenes, objects, people, and readable text.","exampleAgentPrompt":"Can you describe what's in this image for me? Here's the URL: https://example.com/photo.jpg — tell me about the scene, any people or objects, and any readable text you can see.","exampleUseCases":null,"resultDescription":"A written natural-language caption describing the full contents of the image, including the overall scene, identifiable objects, people present, and any text readable within the image. No image data is stored after processing.","failureModes":["Image URL is not publicly accessible or returns a 4xx/5xx error","Image exceeds 8MB size limit","Unsupported image format (not JPG, PNG, or WebP)","Image URL is malformed or unreachable","Image content is ambiguous or too low-resolution to caption accurately"],"whenToPreferThis":"Choose this endpoint when a text-based agent or pipeline needs to understand image content without storing images — ideal for accessibility alt-text generation, content moderation pre-checks, image cataloging, or any workflow where visual content must be converted to text. Prefer this over OCR-only tools when you need a holistic scene description, not just text extraction.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-15T01:10:59.582Z","isFirstParty":false}