> For clean Markdown of any page, append .md to the page URL. > For a complete documentation index, see https://docs.resemble.ai/guides/detect/streaming-audio/llms.txt. > For AI client integration (Claude Code, Cursor, etc.), connect to the MCP server at https://docs.resemble.ai/_mcp/server. # Stream Audio for Deepfake Detection Send an audio file to the Deepfake Detection WebSocket in real time and print each per-window verdict followed by the final aggregate result. ## What You Will Build The Python client in this guide: 1. Opens the authenticated production WebSocket. 2. Uses `ffmpeg` to decode and resample an input file to mono, 16 kHz, signed 16-bit PCM. 3. Sends approximately 100 milliseconds of audio per binary frame at playback speed. 4. Receives result messages concurrently with the upload. 5. Sends the `end` control message and waits for `final` and a normal close. ## Prerequisites * A Resemble API key with audio Deepfake Detection access * Python 3.10 or newer * [`ffmpeg`](https://ffmpeg.org/download.html) available on your `PATH` Install the Python dependency: ```bash python -m pip install aiohttp ``` Set your API key: ```bash export RESEMBLE_API_KEY="YOUR_API_KEY" ``` ## Create the Client Save the following as `stream_audio_detection.py`: ```python #!/usr/bin/env python3 from __future__ import annotations import argparse import asyncio import json import os import shutil import struct from pathlib import Path from typing import Any import aiohttp STREAM_URL = "wss://stream.resemble.ai/api/v1/detect/audio" SAMPLE_RATE = 16_000 FRAME_MILLISECONDS = 100 def streaming_wav_header() -> bytes: """Create a mono PCM WAV header for a stream of unknown final length.""" channels = 1 bits_per_sample = 16 bytes_per_sample = bits_per_sample // 8 byte_rate = SAMPLE_RATE * channels * bytes_per_sample block_align = channels * bytes_per_sample return struct.pack( "<4sL4s4sLHHLLHH4sL", b"RIFF", 0xFFFFFFFF, b"WAVE", b"fmt ", 16, 1, channels, SAMPLE_RATE, byte_rate, block_align, bits_per_sample, b"data", 0xFFFFFFFF, ) def decode_message(message: aiohttp.WSMessage) -> dict[str, Any]: if message.type != aiohttp.WSMsgType.TEXT: raise RuntimeError(f"Expected a text message, received {message.type.name}") payload = json.loads(message.data.strip()) if not isinstance(payload, dict): raise RuntimeError("The server returned an unexpected JSON value") return payload def print_result(payload: dict[str, Any]) -> None: message_type = payload.get("type") if message_type == "ready": print(f"[ready] stream_id={payload.get('stream_id')}") elif message_type == "chunk": chunk = payload.get("chunk_info") or {} label = chunk.get("chunk_label") score = chunk.get("chunk_aggregated_score") score_text = "" if score is None or label == "skipped" else f" score={float(score):.4f}" print( f"[chunk {chunk.get('chunk_id')}] " f"{float(chunk.get('begin_timestamp_s', 0)):.1f}-" f"{float(chunk.get('end_timestamp_s', 0)):.1f}s " f"label={label}{score_text}" ) elif message_type == "final": score = payload.get("aggregated_score") score_text = "N/A" if score is None else f"{float(score):.4f}" print(f"[final] label={payload.get('label')} aggregated_score={score_text}") elif message_type == "error": code = payload.get("error_code") or "stream_error" message = payload.get("error") or payload.get("error_message") or "Unknown error" raise RuntimeError(f"{code}: {message}") async def send_audio(websocket: aiohttp.ClientWebSocketResponse, audio_path: Path) -> float: """Decode the input with ffmpeg and send paced binary PCM frames.""" ffmpeg = shutil.which("ffmpeg") if not ffmpeg: raise RuntimeError("ffmpeg was not found on PATH") bytes_per_second = SAMPLE_RATE * 2 # mono signed 16-bit PCM frame_bytes = bytes_per_second * FRAME_MILLISECONDS // 1000 process = await asyncio.create_subprocess_exec( ffmpeg, "-v", "error", "-i", str(audio_path), "-f", "s16le", "-acodec", "pcm_s16le", "-ar", str(SAMPLE_RATE), "-ac", "1", "pipe:1", stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE, ) assert process.stdout is not None assert process.stderr is not None sent_bytes = 0 started_at = asyncio.get_running_loop().time() await websocket.send_bytes(streaming_wav_header()) try: while True: frame = await process.stdout.read(frame_bytes) if not frame: break await websocket.send_bytes(frame) sent_bytes += len(frame) target_elapsed = sent_bytes / bytes_per_second actual_elapsed = asyncio.get_running_loop().time() - started_at if target_elapsed > actual_elapsed: await asyncio.sleep(target_elapsed - actual_elapsed) except BaseException: if process.returncode is None: process.kill() await process.wait() raise try: await asyncio.wait_for(process.wait(), timeout=5) except asyncio.TimeoutError: process.kill() await process.wait() stderr = (await process.stderr.read()).decode(errors="replace").strip() if process.returncode != 0: raise RuntimeError(f"ffmpeg failed: {stderr or f'exit code {process.returncode}'}") if sent_bytes == 0: raise RuntimeError("ffmpeg decoded no audio") await websocket.send_str(json.dumps({"type": "end"})) duration = sent_bytes / bytes_per_second print(f"[sent] {duration:.2f}s of audio; end marker sent") return duration async def receive_results( websocket: aiohttp.ClientWebSocketResponse, ) -> dict[str, Any] | None: final_result: dict[str, Any] | None = None async for message in websocket: if message.type == aiohttp.WSMsgType.ERROR: raise RuntimeError(f"WebSocket error: {websocket.exception()}") if message.type != aiohttp.WSMsgType.TEXT: continue payload = decode_message(message) print_result(payload) if payload.get("type") == "final": final_result = payload return final_result async def stream_audio(audio_path: Path, api_key: str) -> None: headers = {"Authorization": f"Bearer {api_key}"} params = {"filename": audio_path.name} timeout = aiohttp.ClientTimeout(total=None, sock_connect=30) final_result: dict[str, Any] | None async with aiohttp.ClientSession(timeout=timeout) as session: try: async with session.ws_connect( STREAM_URL, headers=headers, params=params, heartbeat=30, ) as websocket: first_message = await asyncio.wait_for(websocket.receive(), timeout=30) first_payload = decode_message(first_message) print_result(first_payload) if first_payload.get("type") != "ready": raise RuntimeError("The first server message was not ready") sender = asyncio.create_task(send_audio(websocket, audio_path)) receiver = asyncio.create_task(receive_results(websocket)) try: await asyncio.gather(sender, receiver) finally: for task in (sender, receiver): if not task.done(): task.cancel() await asyncio.gather(sender, receiver, return_exceptions=True) final_result = receiver.result() print(f"[closed] code={websocket.close_code}") if websocket.close_code != 1000: raise RuntimeError( f"The WebSocket closed unexpectedly with code {websocket.close_code}" ) except aiohttp.WSServerHandshakeError as exc: raise RuntimeError( f"WebSocket handshake rejected with HTTP {exc.status}: {exc.message}" ) from exc if final_result is None: raise RuntimeError("The server closed without returning a final result") def main() -> None: parser = argparse.ArgumentParser(description="Stream audio for deepfake detection") parser.add_argument("audio_path", type=Path) args = parser.parse_args() audio_path = args.audio_path.expanduser().resolve() if not audio_path.is_file(): parser.error(f"Audio file not found: {audio_path}") api_key = os.environ.get("RESEMBLE_API_KEY", "").strip() if not api_key: parser.error("Set RESEMBLE_API_KEY before running the client") asyncio.run(stream_audio(audio_path, api_key)) if __name__ == "__main__": main() ``` ## Run the Client Pass any audio format supported by your `ffmpeg` installation: ```bash RESEMBLE_API_KEY="YOUR_API_KEY" \ python stream_audio_detection.py /path/to/audio.wav ``` For a voice-active input, output resembles: ```text [ready] stream_id=f241df9c-4738-48f8-8098-34e712e63bd1 [chunk 0] 0.0-4.0s label=real score=0.0812 [chunk 1] 4.0-8.0s label=real score=0.1029 [sent] 8.74s of audio; end marker sent [chunk 2] 8.0-8.7s label=real score=0.1184 [final] label=real aggregated_score=0.0974 [closed] code=1000 ``` The client intentionally sends audio at playback speed. To stream a microphone or telephony source, keep the same WAV header and PCM format, then replace the `ffmpeg` file reader with frames from your live audio source. See [Streaming Audio Detection (WebSocket)](/detect/streaming) for message schemas, authorization behavior, session limits, and error handling.