{"uid":"cap_iJhepNIYb-csQL6H1ZrCJ","slug":"nexari-speech-to-text-with-speaker-diarization-2-min-ed3d0515","name":"Nexari Speech-to-Text with Speaker Diarization (2 min)","description":"Transkription mit Sprechertrennung bis 2 Min Audio (pyannote + Whisper), verarbeitet in Deutschland. Speech-to-text with speaker diarization for audio up to 2 min, German-optimised, processed in Germany. Voraussetzung/required: accept_terms_hash, business_name (B2B) – /legal/terms.json","url":"https://x402.nexari.cloud/transcribe/2min/speakers?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"filename":{"type":"string","description":"Dateiname mit Endung, z. B. memo.m4a"},"language":{"type":"string","description":"de (Standard), en, auto ..."},"audio_url":{"type":"string","description":"HTTPS-URL der Audiodatei / public URL of the audio file"},"audio_base64":{"type":"string","description":"Audio als Base64 / audio as base64"},"num_speakers":{"type":"integer","description":"Nur /speakers: bekannte Sprecherzahl (1-10), verbessert die Trennung"},"business_name":{"type":"string","description":"Unternehmen, für das gehandelt wird (nur B2B)"},"privacy_contact":{"type":"string","description":"E-Mail für Datenschutz-Meldungen (Art. 33 DSGVO)"},"accept_terms_hash":{"type":"string","description":"SHA-256 der AGB aus /legal/terms.json"}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.04","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.04/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.04","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.04","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_Qo0bBjWWW2xqJDIKWcstc","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.04","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Transcribes audio up to 2 minutes with speaker diarization using pyannote and Whisper, processed in Germany, German-optimised.","exampleAgentPrompt":"Can you transcribe this short audio memo for me and separate out who said what? The file is at https://storage.example.com/meeting_clip.m4a, there are 2 speakers, and the language is German.","exampleUseCases":[{"title":"Diarize a short customer call","prompt":"I have a 90-second customer support call recording at https://storage.example.com/call_001.mp3 with 2 speakers — can you transcribe it and show me who said what, in German?"},{"title":"Transcribe a bilingual voice memo","prompt":"Please transcribe this voice memo at https://files.example.com/memo.m4a and split it by speaker — I'm not sure how many speakers there are so use auto-detection, and try to auto-detect the language too."},{"title":"Meeting snippet attribution for notes","prompt":"I've got a 1-minute clip from a team meeting encoded as base64, filename snippet.wav, with 3 speakers — can you transcribe it with speaker labels so I can attribute who said what in my meeting notes?"}],"resultDescription":"Returns a transcript broken down by speaker segments, with speaker labels (e.g. SPEAKER_00, SPEAKER_01), timestamps, and the transcribed text for each segment. Processed entirely within Germany for GDPR compliance.","failureModes":["Audio exceeds 2-minute limit — request rejected","Unsupported audio format or corrupted file","audio_url not publicly accessible — fetch failure","num_speakers out of 1-10 range — validation error","accept_terms_hash missing or invalid — terms not accepted error","Base64 audio malformed — decoding error","Network timeout fetching remote audio URL"],"whenToPreferThis":"Choose this endpoint when you need both transcription and speaker diarization for short audio clips (under 2 minutes), particularly for German-language content or when GDPR/data residency in Germany is required. Prefer this over the plain transcription endpoint when identifying who said what matters. Use the 10-minute sibling endpoint for longer audio.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T17:23:17.613Z","isFirstParty":false,"canonicalSlug":"nexari-speech-to-text-with-speaker-diarization-2-min-ed3d0515"}