{"uid":"cap_ICQR1ZBjuZD3ZVcIuKydd","slug":"relaystation-ocr-layout-da043067","name":"Relaystation OCR Layout","description":"$0.010/document. OCR a document with line- and word-level layout coordinates (Textract) — position-aware text for downstream parsing. 1¢ x402 min; remainder auto-credits — relaystation.ai/penny","url":"https://api.relaystation.ai/v1/doc/ocr-layout","method":"POST","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","method","bodyType","body"],"properties":{"body":{"type":"object","properties":{"file":{"type":"object","description":"cputools input-source: { inline: <base64 ≤ 4 MiB> } or { inputKey: <scratch key from /v1/cputools/upload-url, ≤ 50 MB> }."}}},"type":{"type":"string","const":"http"},"method":{"enum":["POST","PUT","PATCH"],"type":"string"},"bodyType":{"enum":["json","form-data","text"],"type":"string"}},"additionalProperties":false},"output":{"type":"object","required":["type"],"properties":{"type":{"type":"string"},"example":{"type":"object"}}}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.01","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"registry","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.01/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_43fXAjYh73PawzzyNsdks","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.01","costPer":"request","priority":0,"asset":null,"unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"OCR a document and return full line- and word-level layout coordinates (via AWS Textract) for position-aware text extraction and downstream parsing.","exampleAgentPrompt":"Can you OCR this scanned invoice PDF and give me the text with word-level and line-level position coordinates so I can parse the layout programmatically?","exampleUseCases":[{"title":"Invoice field extraction with layout","prompt":"I have a scanned invoice image — can you OCR it and return all the text with word positions and bounding boxes so I can automatically locate the total amount and line items?"},{"title":"Form digitization with coordinates","prompt":"I've got a scanned paper form that I need to digitize. Run OCR on it and give me word-level layout coordinates so I can map each field value back to its position on the page."},{"title":"Scanned contract text extraction","prompt":"Can you extract all the text from this scanned contract PDF along with the line and word coordinates? I need to know where each clause appears on each page for my document parser."}],"resultDescription":"A structured response containing all recognized text from the document, organized with line-level and word-level bounding box coordinates (position data) as produced by AWS Textract, enabling downstream layout-aware parsing and field extraction.","failureModes":["File too large (>50 MB for inputKey, >4 MiB for inline base64) returns an error","Unsupported file format or corrupted document may cause OCR failure","Missing required 'file' body field returns a validation error","Network or Textract service errors may return a 5xx response","Insufficient prepaid credit balance may block the call"],"whenToPreferThis":"Choose this endpoint when you need not just the raw text from a scanned document but also precise word- and line-level spatial coordinates (bounding boxes) — for example, when building a parser that needs to locate fields by position, reconstruct table structures, or map extracted values to their visual location on the page. Prefer this over a plain OCR endpoint any time layout matters for downstream processing.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-14T06:54:08.099Z","isFirstParty":false}