{"uid":"cap_7wLM5FQzyrwD19wFfZbiB","slug":"forgemesh-speech-to-text-6a538725","name":"ForgeMesh Speech-to-Text","description":"Speech-to-text API: audio URL in, transcript out — a neural speech-recognition engine on our own hardware, ~99 languages, mp3/wav/m4a/ogg up to 25MB. For voicemail, podcast, meeting, and voice-agent pipelines. Audio deleted after processing.","url":"https://x402.forgemesh.io/speech-to-text","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"audio_url":{"type":"string","description":"Public URL of an audio file, max 25MB"}}},"responseSchema":{"type":"json","example":{"text":"The quick brown fox jumps over the lazy dog. ForgeMesh utility grid speech fixture."}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.03","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.03/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.03","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.03","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_j5592tTxRFYeEe1SFeatX","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.03","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Transcribes audio files from a public URL into text using a neural speech recognition engine supporting ~99 languages and common audio formats.","exampleAgentPrompt":"Transcribe this voicemail audio file for me — it's an mp3 at https://storage.example.com/voicemail-2024-06-10.mp3 and I need the full text of what was said.","exampleUseCases":null,"resultDescription":"A text transcript of the spoken content in the audio file, derived from neural speech recognition. The audio is deleted after processing. Supports approximately 99 languages.","failureModes":["Audio file URL is not publicly accessible or returns 4xx/5xx — endpoint cannot fetch the file","Audio file exceeds 25MB size limit — request rejected","Unsupported audio format (not mp3, wav, m4a, or ogg) — processing fails","Audio contains no intelligible speech — empty or low-quality transcript returned","Network timeout fetching large audio files — processing error"],"whenToPreferThis":"Use this endpoint when you have a publicly accessible audio file URL and need a text transcript quickly without managing your own speech recognition infrastructure. Well-suited for voicemail pipelines, podcast transcription, meeting notes, and voice-agent workflows where privacy matters since audio is deleted post-processing. Supports ~99 languages and common formats (mp3, wav, m4a, ogg) up to 25MB per file.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-14T00:49:42.155Z","isFirstParty":false}