{"uid":"cap_roaJ-wynYx-Sl8yK_03L-","slug":"forgemesh-voice-pro-long-form-tts-29bad06c","name":"ForgeMesh Voice Pro Long-Form TTS","description":"Text-to-speech with speed (0.7x-2.0x) and quality-step (1-100) control, 1 of 10 standard voices across 31 languages, for 501-2000 characters, returned as WAV audio. Use it to: pace a longer narration like a podcast intro, read a multi-paragraph script at a chosen speed, produce a slow deliberate reading of instructions, voice an extended alert or briefing. USDC on Base via x402.","url":"https://voice.forgemesh.io/v1/tts/pro-long","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"lang":{"enum":["en","ko","ja","ar","bg","cs","da","de","el","es","et","fi","fr","hi","hr","hu","id","it","lt","lv","nl","pl","pt","ro","ru","sk","sl","sv","tr","uk","vi"],"type":"string","description":"Language code; 31 languages supported"},"text":{"type":"string","maxLength":2000,"minLength":1,"description":"Text to synthesize into speech, max 2000 characters. Use non-long routes for 1-500 chars and -long routes for 501-2000 chars."},"speed":{"type":"number","maximum":2,"minimum":0.7,"description":"Pro/Custom expressive speed control, 0.7-2.0. Presets: slow 0.7, normal 1.0, fast 1.3, rapid 1.6"},"steps":{"type":"integer","maximum":100,"minimum":1,"description":"Pro/Custom quality control, 1-100. Presets: draft 4, standard 8, high 16, ultra 24"},"voice":{"enum":["M1","M2","M3","M4","M5","F1","F2","F3","F4","F5"],"type":"string","description":"Standard voice: M1-M5 or F1-F5"}}},"responseSchema":{"type":"json","example":{"description":"44.1kHz 16-bit mono WAV speech audio returned inline","content_type":"audio/wav"}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.006","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.006/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.006","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.006","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_1tvOmw_Tzdt8zhUs6uehS","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.006","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Converts long-form text (up to 2000 chars) to high-quality WAV audio using diffusion-based synthesis with selectable persona voices and 31 language support, billed per call via x402 micropayments.","exampleAgentPrompt":"Read this paragraph aloud using ForgeMesh's Storyteller persona voice in English, at a slightly slow pace of 0.8 speed with 20 diffusion steps: 'In the age of artificial minds, the boundary between machine and muse grew ever thinner...'","exampleUseCases":null,"resultDescription":"A 44.1kHz 16-bit mono WAV audio file containing the synthesized speech of the submitted text, rendered in the selected voice persona, language, speed, and diffusion quality level.","failureModes":["Text exceeds 2000 character limit — request rejected","Invalid or unsupported ISO language code — synthesis fails","Voice name not recognized (typo in persona name) — error returned","Speed value outside 0.7-2.0 range — validation error","Steps value outside 1-100 range — validation error","Insufficient USDC balance for x402 micropayment — payment failure","Network timeout during diffusion synthesis for high step counts"],"whenToPreferThis":"Choose this endpoint when you need long-form (up to 2000 chars) high-quality diffusion-based TTS with persona voices (Storyteller, Narrator, etc.), multilingual output across 31 languages, and pay-per-call pricing via x402 micropayments — especially for AI agents, audiobooks, video narration, or apps where quality and voice character matter more than raw speed.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-14T17:52:20.921Z","isFirstParty":false}