{"uid":"cap_W3tmTlVB2oUq-HoqfXf0m","slug":"x402-forgemesh-image-to-text-description-2c0315d0","name":"x402 ForgeMesh Image-to-Text Description","description":"Image-to-text description generator: submit any image URL and receive a detailed written account of what's shown, subjects, setting, actions, and any visible text, rendered entirely in natural language. Processed in memory with nothing retained. Useful for building searchable captions, screening uploads before publishing, or giving text-only agents a way to understand visual content.","url":"https://x402.forgemesh.io/image-to-text-description","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"image_url":{"type":"string","description":"Public URL of the image (jpg/png/webp, max 8MB)"}}},"responseSchema":{"type":"json","example":{"text":"A teal rectangular graphic with the words FORGEMESH UTILITY GRID in bold white capital letters centered on it.","model":"moondream"}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.01","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.01/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_aSaSrpR6RyLnU4yOjZk4i","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.01","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Submits an image URL and returns a detailed natural-language description of its contents, subjects, setting, actions, and visible text.","exampleAgentPrompt":"Can you describe what's in this image for me — https://upload.wikimedia.org/wikipedia/commons/thumb/3/3a/Cat03.jpg/1200px-Cat03.jpg — tell me the subjects, setting, any visible text, and what's happening?","exampleUseCases":null,"resultDescription":"A detailed natural-language paragraph describing the image's subjects, setting, actions occurring in the scene, and any readable text visible in the image. No image data is stored; processing is in-memory only.","failureModes":["Image URL is not publicly accessible or returns a non-200 status — endpoint cannot fetch the image","Image exceeds 8MB size limit — request rejected","Unsupported image format (not JPG, PNG, or WebP) — processing fails","URL points to a non-image resource — returns an error","Network timeout fetching the remote image — request fails"],"whenToPreferThis":"Use this endpoint when a text-only agent needs to understand visual content, when you need to generate searchable captions for images at scale, when screening user-uploaded images before publishing, or when you need alt-text or accessibility descriptions. Prefer this over generic OCR tools when you need both scene description AND text extraction in a single natural-language output.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-15T07:06:18.150Z","isFirstParty":false}