{"uid":"cap_4yLTDmZiVpT-RpUU1expM","slug":"forgemesh-audio-transcription-6fdedd74","name":"ForgeMesh Audio Transcription","description":"Audio transcription API: transcribe any public audio URL (mp3, wav, m4a, ogg — up to 25MB / ~20 min) to text with a neural speech-recognition engine running locally on our hardware. Multilingual (~99 languages), auto language detection. Use as a speech-to-text step for voicemail, podcasts, meetings, and voice agents. Audio is processed in a sandbox and deleted immediately; we keep nothing.","url":"https://x402.forgemesh.io/transcribe","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"audio_url":{"type":"string","description":"Public URL of an audio file, max 25MB"}}},"responseSchema":{"type":"json","example":{"text":"The quick brown fox jumps over the lazy dog. ForgeMesh utility grid speech fixture."}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.03","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.03/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.03","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.03","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_MsuK51HIarRLDxvU52eCz","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.03","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Transcribes a public audio file URL (mp3, wav, m4a, ogg) to text using a local neural speech-recognition engine with multilingual support and auto language detection.","exampleAgentPrompt":"Can you transcribe this podcast audio file for me? Here's the public URL: https://example.com/episode42.mp3 — I need the full text transcript.","exampleUseCases":null,"resultDescription":"Returns a text transcript of the audio content extracted from the provided public URL, with support for approximately 99 languages and automatic language detection. The audio is processed in a sandboxed environment and deleted immediately after transcription.","failureModes":["Audio file exceeds 25MB size limit — request rejected","Unsupported audio format (e.g. video files, flac) — not processed","Private or authentication-required URL — cannot be fetched","Audio URL is unreachable or returns a non-200 response — fetch error","Audio duration exceeds ~20 minutes — may be rejected","Corrupted or malformed audio file — recognition fails","Very low audio quality or heavy background noise — inaccurate transcript"],"whenToPreferThis":"Use this endpoint when you need to convert a publicly accessible audio file (mp3, wav, m4a, ogg) to text, especially for voicemail processing, podcast transcription, meeting notes, or powering voice agents. Prefer it over cloud STT services when privacy is a concern, since audio is processed locally on ForgeMesh hardware and deleted immediately. Well-suited for multilingual audio where automatic language detection is needed without pre-specifying a language.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":1,"lastUsedAt":"2026-07-22T16:46:19.311Z","lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-14T19:11:44.766Z","isFirstParty":false}