{"uid":"cap_QtEd9jw8ajBpBRnS8LC0c","slug":"forgemesh-tts-custom-long-form-text-to-speech-c01701ca","name":"ForgeMesh TTS Custom Long-Form Text-to-Speech","description":"Paid text-to-speech via x402. WAV audio in 31 languages, OpenAI-compatible, no API keys.","url":"https://tts.forgemesh.io/v1/tts/custom-long","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"lang":{"enum":["en","ko","ja","ar","bg","cs","da","de","el","es","et","fi","fr","hi","hr","hu","id","it","lt","lv","nl","pl","pt","ro","ru","sk","sl","sv","tr","uk","vi"],"type":"string","description":"Language code; 31 languages supported"},"text":{"type":"string","maxLength":2000,"minLength":1,"description":"Text to synthesize into speech, max 2000 characters. Use non-long routes for 1-500 chars and -long routes for 501-2000 chars."},"speed":{"type":"number","maximum":2,"minimum":0.7,"description":"Pro/Custom expressive speed control, 0.7-2.0. Presets: slow 0.7, normal 1.0, fast 1.3, rapid 1.6"},"steps":{"type":"integer","maximum":100,"minimum":1,"description":"Pro/Custom quality control, 1-100. Presets: draft 4, standard 8, high 16, ultra 24"},"voice":{"enum":["M1","M2","M3","M4","M5","F1","F2","F3","F4","F5","Storyteller","Narrator","Announcer","Assistant","Urgent","Sage","Spark","Anchor","Velvet","Echo"],"type":"string","description":"Voice name. Standard voices M1-M5/F1-F5; persona voices require Custom tier."}}},"responseSchema":{"type":"json","example":{"description":"44.1kHz 16-bit mono WAV speech audio returned inline","content_type":"audio/wav"}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.01","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.01/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_rgQNTrhqoA_WLghNCL7Qe","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.01","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Converts long-form text (501–2000 characters) into 44.1kHz WAV audio with configurable voice, speed, and quality across 31 languages, paid via x402.","exampleAgentPrompt":"Turn this 800-character product description into spoken audio using the 'Narrator' voice in English at normal speed and standard quality: 'Welcome to our flagship store. Our new line of ergonomic office chairs combines cutting-edge design with unparalleled comfort, featuring adjustable lumbar support, breathable mesh fabric, and a five-year warranty — perfect for long work sessions.'","exampleUseCases":[{"title":"Multilingual e-learning narration","prompt":"Read this 700-character Spanish lesson script aloud using voice F2 in Spanish at a slow pace (speed 0.7) with high quality (16 steps) so students can follow along clearly."},{"title":"Podcast episode voiceover","prompt":"Generate a voiceover for this 1500-character podcast intro using the 'Anchor' voice in English at fast speed and ultra quality (24 steps) — I need it to sound punchy and professional."},{"title":"Localized product announcement audio","prompt":"Convert this 900-character Japanese product announcement into speech using the M3 voice in Japanese at normal speed and standard quality (8 steps) for our Tokyo launch video."}],"resultDescription":"Returns a 44.1kHz 16-bit mono WAV audio file inline (Content-Type: audio/wav) containing the synthesized speech of the submitted text, ready for playback or further processing.","failureModes":["Text shorter than 501 characters — use non-long routes instead","Text exceeding 2000 characters — truncation or rejection","Unsupported language code — enum validation error","Invalid speed value outside 0.7–2.0 range — validation error","Persona voices (Storyteller, Narrator, etc.) may require Custom tier — tier error","Payment failure via x402 — 402 response if USDC payment not processed","High steps values (e.g. ultra 24+) may increase latency noticeably"],"whenToPreferThis":"Choose this endpoint when you need to synthesize longer passages of 501–2000 characters with fine-grained control over voice persona, speed, and quality. Prefer it over standard TTS routes for editorial content, narration, voiceovers, or multilingual audio where expressive control matters. It requires no API key (payments via x402/USDC) and returns a ready-to-use WAV file, making it ideal for agent workflows that need audio without credential management overhead.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-14T17:51:15.201Z","isFirstParty":false}