{"uid":"cap_8gs8GoFPLgDkZxdLxYcRP","slug":"hypercli-speaking-video-generation-25175470","name":"HyperCLI Speaking Video Generation","description":"speaking-video — fixed price $0.25","url":"https://api.hypercli.com/api/x402/flow/speaking-video","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"width":{"anyOf":[{"type":"integer"},{"type":"null"}],"default":null},"height":{"anyOf":[{"type":"integer"},{"type":"null"}],"default":null},"prompt":{"type":"string"},"dry_run":{"type":"boolean","default":false},"file_ids":{"anyOf":[{"type":"array","items":{"type":"string"}},{"type":"null"}],"default":null},"negative":{"anyOf":[{"type":"string"},{"type":"null"}],"default":null},"audio_url":{"anyOf":[{"type":"string"},{"type":"null"}],"default":null},"image_url":{"anyOf":[{"type":"string"},{"type":"null"}],"default":null},"notify_url":{"anyOf":[{"type":"string"},{"type":"null"}],"default":null}}},"responseSchema":{"type":"json","example":{"render":{},"access_key":"example","cancel_url":"example","status_url":"example"}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.25","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"settled","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.25/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.25","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.25","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_e0Hvt4iK4AKUMMvJ2kI4Q","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.25","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Generates a speaking/talking-head video by animating an image with a provided audio track and text prompt","exampleAgentPrompt":"Using HyperCLI, generate a speaking video where the person in this portrait photo (https://example.com/portrait.jpg) talks along with this voiceover audio (https://example.com/speech.mp3) — use the prompt 'professional business presentation' and render it at 512x512.","exampleUseCases":[{"title":"Create AI avatar for sales pitch","prompt":"I need to turn a headshot of our CEO into a talking-head video that delivers our company pitch. Can you animate this photo with the voiceover recording I prepared and make it look professional?"},{"title":"Generate lip-sync video for training content","prompt":"We've got an instructional voiceover for our employee training module, and I want to pair it with an animated portrait of our trainer. Can you create a speaking video that syncs the audio with the image?"},{"title":"Produce personalized greeting messages","prompt":"I want to create custom video greetings where a character appears to speak directly to customers. Can you take this character portrait, add our welcome voiceover, and generate a few speaking avatar videos with different backgrounds?"}],"resultDescription":"Returns a JSON object with an access key to retrieve the rendered video, a status URL to poll for job completion, a cancel URL to abort the job, and a render object with job details. The video is generated asynchronously and can be monitored via the status URL.","failureModes":["Invalid or inaccessible image_url returns an error or failed render","Invalid or inaccessible audio_url causes audio processing failure","Unsupported width/height dimensions may cause rendering errors","Missing required prompt field results in a validation error","Payment failure via x402 protocol prevents job creation","Async job timeout if render takes too long on the backend"],"whenToPreferThis":"Use this endpoint when you need to animate a static image or portrait to sync with audio, creating a talking-head or lip-sync video. Ideal for AI avatar creation, video presentations, or any use case combining a still image with spoken audio into a dynamic video. Prefer over text-to-video when you already have a specific image and audio track to combine.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":2,"lastUsedAt":"2026-07-15T10:18:36.481Z","lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-16T00:43:10.844Z","isFirstParty":false}