{"uid":"cap_46jjHVpjlxzwjk_9sBfrU","slug":"x402-gateway-production-up-railway-app-9acac2e8","name":"x402 Gateway Voice-Cloning TTS (Lux)","description":"Voice-cloning text-to-speech — provide a reference audio clip and generate speech in that voice at 48kHz","url":"https://x402-gateway-production.up.railway.app/api/tts/lux","method":"POST","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","bodyType","body","method"],"properties":{"body":{"type":"object","required":["text","audio_url"],"properties":{"seed":{"type":"number","description":"Random seed for reproducibility"},"text":{"type":"string","example":"Hey, what's up? I'm feeling really great today!","description":"Text to convert to speech"},"audio_url":{"type":"string","example":"https://storage.googleapis.com/falserverless/example_inputs/reference_audio.wav","description":"URL of reference audio file for voice cloning"},"guidance_scale":{"type":"number","default":3,"example":3,"description":"Classifier-free guidance scale (0-10)"},"max_ref_length":{"type":"number","default":5,"example":5,"description":"Max reference audio duration in seconds (1-15)"},"num_inference_steps":{"type":"number","default":4,"example":4,"description":"Flow-matching inference steps (1-16)"}},"additionalProperties":false},"type":{"type":"string","const":"http"},"method":{"enum":["POST"],"type":"string"},"bodyType":{"enum":["json","form-data","text"],"type":"string"}},"additionalProperties":false}}},"responseSchema":null,"example":{"request":{"text":"Welcome to our product demo, thanks for joining us today","audio_url":"https://github.com/openai/whisper/raw/main/tests/jfk.flac"},"response":{"seed":1091889596,"audio":{"url":"https://v3b.fal.media/files/b/0a9e15ca/SyCUCOA1JexGrzB7-WOFi_output.wav","file_size":571436,"content_type":"audio/wav"},"inference_time_ms":2997}},"exampleRequest":{"text":"Welcome to our product demo, thanks for joining us today","audio_url":"https://github.com/openai/whisper/raw/main/tests/jfk.flac"},"tags":["x402"],"displayCostAmount":"0.020000","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"settled","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.020000/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.02","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.02","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_ergKFF10uEvVYCxVRNKhA","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.02","costPer":"request","priority":0,"asset":null,"unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Generates 48kHz speech audio by cloning a voice from a reference audio clip using text-to-speech synthesis","exampleAgentPrompt":"Use my uploaded voice sample to generate speech that says 'Welcome to our product demo, thanks for joining us today' — clone my voice at 48kHz quality.","exampleUseCases":[{"title":"Personalized podcast intro narration","prompt":"I've got a short recording of my voice — use it to generate a podcast intro that says 'Welcome back to The Future Forward podcast, I'm your host Alex Rivera and today we're diving deep into the world of AI.' Make it sound exactly like me at high quality."},{"title":"Branded customer support audio","prompt":"We have a reference clip of our brand spokesperson's voice. Use it to generate audio that says 'Thank you for contacting us today. A member of our team will be with you shortly.' We want all our automated messages to sound like her, not a generic robot."},{"title":"Audiobook narration in author's voice","prompt":"The author recorded a short sample of themselves reading a passage. Using that voice clip, generate narration for the first paragraph of their book: 'It was a morning like any other, yet something in the air whispered of change — a feeling she could not quite name but had long been waiting for.'"}],"resultDescription":"Returns a high-quality 48kHz audio file of the input text spoken in the voice cloned from the provided reference audio clip.","failureModes":["Reference audio clip is too short or low quality for voice cloning","Text input is empty or missing","Unsupported audio format for reference clip","Payment of $0.02 USDC not fulfilled via x402 protocol","Reference audio contains too much background noise for accurate cloning","Server error or timeout during synthesis"],"whenToPreferThis":"Choose this endpoint when you need personalized or branded voice audio that matches a specific speaker's voice characteristics, rather than a generic TTS voice. Ideal for agents that need to produce audio content in a user's own voice or a specific reference voice at high 48kHz fidelity.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":7,"lastUsedAt":"2026-06-28T03:09:04.278Z","lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-13T18:45:31.182Z","isFirstParty":false}