{"uid":"cap_VUFIe3GfcJ3JDCVsQXL0B","slug":"x402engine-voice-cloning-text-to-speech-lux-191ab22e","name":"x402engine Voice-Cloning Text-to-Speech (LUX)","description":"Voice-cloning text-to-speech — provide a reference audio clip and generate speech in that voice at 48kHz","url":"https://x402engine.app/api/tts/lux","method":"POST","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","bodyType","body","method"],"properties":{"body":{"type":"object","required":["text","audio_url"],"properties":{"seed":{"type":"number","description":"Random seed for reproducibility"},"text":{"type":"string","example":"Hey, what's up? I'm feeling really great today!","description":"Text to convert to speech"},"audio_url":{"type":"string","example":"https://storage.googleapis.com/falserverless/example_inputs/reference_audio.wav","description":"URL of reference audio file for voice cloning"},"guidance_scale":{"type":"number","default":3,"example":3,"description":"Classifier-free guidance scale (0-10)"},"max_ref_length":{"type":"number","default":5,"example":5,"description":"Max reference audio duration in seconds (1-15)"},"num_inference_steps":{"type":"number","default":4,"example":4,"description":"Flow-matching inference steps (1-16)"}},"additionalProperties":false},"type":{"type":"string","const":"http"},"method":{"enum":["POST"],"type":"string"},"bodyType":{"enum":["json","form-data","text"],"type":"string"}},"additionalProperties":false}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.02","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"registry","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.02/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.02","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.02","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_ga4KjhDJnvFOcd5T5JAT7","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.02","costPer":"request","priority":0,"asset":null,"unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Clones a voice from a reference audio URL and synthesizes new speech from text at 48kHz quality","exampleAgentPrompt":"Clone the voice from this audio clip — https://storage.googleapis.com/falserverless/example_inputs/reference_audio.wav — and say 'Welcome back, it's great to have you here today!' using 4 inference steps and a guidance scale of 3.","exampleUseCases":[{"title":"Personalized podcast narration","prompt":"Use the voice sample at https://mycdn.com/host_voice.wav to generate audio saying 'Today on the show, we dive deep into AI trends shaping 2025' — use a guidance scale of 5 and 8 inference steps for higher quality."},{"title":"Custom voice assistant responses","prompt":"I have a reference audio clip of my brand voice at https://assets.mybrand.com/brand_voice.wav — generate the phrase 'Your order has been confirmed and is on its way!' in that cloned voice at default settings."},{"title":"Audiobook narration in author's voice","prompt":"Clone the voice from https://recordings.example.com/author_sample.wav and read aloud: 'Chapter one. The morning sun crept over the mountains, casting long golden shadows across the valley.' Use 4 inference steps and guidance scale 3."}],"resultDescription":"Returns synthesized 48kHz audio spoken in the cloned voice of the reference speaker, matching the prosody and timbre of the provided audio URL. Output is a high-fidelity audio file or stream reproducing the target voice speaking the requested text.","failureModes":["Reference audio URL is unreachable or returns a non-audio file","Reference audio clip is too long and exceeds max_ref_length limit","Text input is empty or missing","Guidance scale or inference steps out of valid range","Payment of $0.02 USDC not provided or rejected via x402","Voice cloning fails due to poor-quality or noisy reference audio","Network timeout during inference"],"whenToPreferThis":"Choose this endpoint when you need voice-cloning TTS — i.e., you have a reference audio sample and want synthesized speech that matches that specific speaker's voice. Prefer it over standard TTS endpoints when speaker identity matters, for personalized assistants, branded voice experiences, or when replicating a specific person's voice. The 48kHz output and configurable inference steps make it suitable for production-quality audio generation via pay-per-call micropayment with no subscription.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-13T18:56:14.695Z","isFirstParty":false}