{"uid":"cap_Wm3U_cJU_kluDhZmK6oAV","slug":"forgemesh-voice-custom-long-form-tts-c7953734","name":"ForgeMesh Voice Custom Long-Form TTS","description":"Expressive text-to-speech in any of 20 voices, standard M1-M5/F1-F5 plus personas like Storyteller, Narrator, Announcer, Urgent, Velvet, with speed and quality control, for 501-2000 characters, WAV output. Use it to: narrate a longer story or audiobook chapter, voice an extended character monologue, produce a branded multi-sentence announcement, script a persona-driven podcast segment. USDC on Base via x402.","url":"https://voice.forgemesh.io/v1/tts/custom-long","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"lang":{"enum":["en","ko","ja","ar","bg","cs","da","de","el","es","et","fi","fr","hi","hr","hu","id","it","lt","lv","nl","pl","pt","ro","ru","sk","sl","sv","tr","uk","vi"],"type":"string","description":"Language code; 31 languages supported"},"text":{"type":"string","maxLength":2000,"minLength":1,"description":"Text to synthesize into speech, max 2000 characters. Use non-long routes for 1-500 chars and -long routes for 501-2000 chars."},"speed":{"type":"number","maximum":2,"minimum":0.7,"description":"Pro/Custom expressive speed control, 0.7-2.0. Presets: slow 0.7, normal 1.0, fast 1.3, rapid 1.6"},"steps":{"type":"integer","maximum":100,"minimum":1,"description":"Pro/Custom quality control, 1-100. Presets: draft 4, standard 8, high 16, ultra 24"},"voice":{"enum":["M1","M2","M3","M4","M5","F1","F2","F3","F4","F5","Storyteller","Narrator","Announcer","Assistant","Urgent","Sage","Spark","Anchor","Velvet","Echo"],"type":"string","description":"Voice name. Standard voices M1-M5/F1-F5; persona voices require Custom tier."}}},"responseSchema":{"type":"json","example":{"description":"44.1kHz 16-bit mono WAV speech audio returned inline","content_type":"audio/wav"}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.01","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.01/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_hW4Qrf9b9hRXlk0_dDMwX","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.01","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Converts long-form text (up to 2000 characters) to speech using customizable persona voices across 31 languages, returning a 44.1kHz WAV audio file via pay-per-call x402 API.","exampleAgentPrompt":"Read out the following passage in English using the Storyteller persona voice at 1.2x speed: 'In the beginning, the world was without form, and the silence stretched across the deep...' — give me a WAV file I can use in my video.","exampleUseCases":null,"resultDescription":"A 44.1kHz 16-bit mono WAV audio file containing the synthesized speech of the submitted text, returned as audio/wav binary content ready for playback or embedding in media projects.","failureModes":["Text exceeds 2000 character limit — request rejected","Invalid or unsupported ISO language code — error response","Invalid voice identifier (not in M1-M5, F1-F5, or named personas) — error response","Speed value out of Pro tier range (0.7-2.0) — error or clamped","Payment failure via x402 protocol — 402 Payment Required response","Steps value out of range (1-100) — validation error"],"whenToPreferThis":"Choose this endpoint when you need long-form text narration (up to 2000 chars), want persona-based voices (Storyteller, Narrator, etc.), require multilingual support across 31 languages, or need fine-grained control over speech speed and diffusion quality steps. Prefer it over generic TTS APIs when your agent uses x402 micropayment rails and you need WAV output specifically.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-13T18:46:36.469Z","isFirstParty":false}