{"uid":"cap_8C538gT4mPyWeWhHnUhT5","slug":"pdf-extraction-api-bismuth-10ce622d","name":"PDF Extraction API (Bismuth)","description":"Extract text from PDF documents with OCR (Tesseract) and vision mode support. Part of the Bismuth utility API suite for AI agents.","url":"https://pdf-api-production-cf1e.up.railway.app/extract","method":"POST","headers":{},"bodySchema":{"type":"object","required":["file"],"properties":{"file":{"type":"string"}}},"responseSchema":{},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.01","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.01/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_Lw02PSKoT1Ns7LFYDJxFO","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.01","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Extracts text content from PDF documents using OCR (Tesseract) and vision-mode processing","exampleAgentPrompt":"Can you extract all the text from this PDF file I have — it's a scanned document so it'll need OCR to read it properly?","exampleUseCases":[{"title":"Scanned invoice text extraction","prompt":"I have a scanned invoice PDF and I need to pull out all the text so I can process the line items — can you run OCR on it and give me the full text?"},{"title":"Legal document digitization","prompt":"Extract all the text from this legal contract PDF so I can search through it and summarize the key clauses."},{"title":"Research paper parsing","prompt":"I've got a research paper in PDF format and I need the raw text extracted from it — including any figures or captions — so I can feed it into my analysis pipeline."}],"resultDescription":"Returns the extracted text content from the PDF document, processed using Tesseract OCR and optionally a vision mode for handling complex or scanned documents.","failureModes":["Corrupted or password-protected PDF results in extraction failure","Low-quality scans may produce garbled OCR output","Very large PDFs may time out or exceed processing limits","Non-PDF file format submitted as input returns an error","Empty or image-only PDF with no extractable text returns blank output"],"whenToPreferThis":"Choose this endpoint when you need to extract text from PDF documents, especially scanned or image-based PDFs that require OCR. It is ideal for agent workflows where structured or unstructured PDF content needs to be read programmatically, particularly when documents are not machine-readable. Prefer this over general document converters when OCR and vision-mode accuracy are priorities.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-15T18:37:42.666Z","isFirstParty":false}