{"uid":"cap_FuxHCGVnxvohv_wJ-3cH3","slug":"relaystation-image-ocr-e6d08b15","name":"RelayStation Image OCR","description":"$0.003/call. Extract text from an image — Tesseract, 10 languages, plain text out. 1¢ x402 min; remainder auto-credits — relaystation.ai/penny","url":"https://api.relaystation.ai/v1/image/ocr","method":"POST","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","method","bodyType","body"],"properties":{"body":{"type":"object","properties":{"tsv":{"type":"boolean","description":"Also return tesseract TSV (per-word boxes + confidences)."},"file":{"type":"object","description":"cputools input-source: { inline: <base64 ≤ 4 MiB> } or { inputKey: <scratch key from /v1/cputools/upload-url, ≤ 50 MB> }."},"lang":{"enum":["eng","spa","fra","deu","ita","por","nld","pol","rus","chi_sim"],"type":"string","description":"OCR language tag (default \"eng\"). The live roster is the operator-tunable cputools.ocr.langs (deployed traineddata: eng, spa, fra, deu, ita, por, nld, pol, rus, chi_sim); an off-roster lang is a free 422 UNSUPPORTED_LANG naming the roster."}}},"type":{"type":"string","const":"http"},"method":{"enum":["POST","PUT","PATCH"],"type":"string"},"bodyType":{"enum":["json","form-data","text"],"type":"string"}},"additionalProperties":false},"output":{"type":"object","required":["type"],"properties":{"type":{"type":"string"},"example":{"type":"object"}}}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.01","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"registry","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.01/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_EXwI2ExyUnJ458HZSKw69","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.01","costPer":"request","priority":0,"asset":null,"unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Extracts text from an image using Tesseract OCR, supporting 10 languages and returning plain text output","exampleAgentPrompt":"Can you extract all the text from this image of my receipt? It's in English — just give me the plain text.","exampleUseCases":[{"title":"Digitize scanned invoice","prompt":"I've got a scanned PNG of an invoice — can you pull out all the text so I can copy the line items and totals?"},{"title":"Read text from a screenshot","prompt":"Here's a screenshot of an error message I got on my screen. Can you read the text out of it for me?"},{"title":"Extract text from foreign-language flyer","prompt":"I have a photo of a French-language flyer. Can you OCR it and give me the plain text so I can paste it into a translator?"}],"resultDescription":"Plain text string containing all text recognized in the image, extracted via Tesseract OCR across up to 10 supported languages","failureModes":["Image URL is inaccessible or returns a non-image response — endpoint returns an error","Image format is unsupported — only common raster formats (JPEG, PNG, etc.) are handled","Low image quality or resolution results in garbled or incomplete text extraction","Language not in supported set of 10 — may produce poor results","Payment insufficient — x402 minimum charge is 1¢ USDC"],"whenToPreferThis":"Choose this endpoint when you need fast, low-cost OCR on a single image and plain text output is sufficient. It is ideal for agent pipelines that need to digitize receipts, screenshots, scanned documents, or image-embedded text without requiring structured data extraction. Prefer it over vision-LLM approaches when cost per call matters and Tesseract accuracy is adequate for the language and image quality.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-15T00:44:11.918Z","isFirstParty":false}