Stream Audio for Deepfake Detection
Send an audio file to the Deepfake Detection WebSocket in real time and print each per-window verdict followed by the final aggregate result.
What You Will Build
The Python client in this guide:
- Opens the authenticated production WebSocket.
- Uses
ffmpegto decode and resample an input file to mono, 16 kHz, signed 16-bit PCM. - Sends approximately 100 milliseconds of audio per binary frame at playback speed.
- Receives result messages concurrently with the upload.
- Sends the
endcontrol message and waits forfinaland a normal close.
Prerequisites
- A Resemble API key with audio Deepfake Detection access
- Python 3.10 or newer
ffmpegavailable on yourPATH
Install the Python dependency:
python -m pip install aiohttp
Set your API key:
export RESEMBLE_API_KEY="YOUR_API_KEY"
Create the Client
Save the following as stream_audio_detection.py:
#!/usr/bin/env python3from __future__ import annotationsimport argparseimport asyncioimport jsonimport osimport shutilimport structfrom pathlib import Pathfrom typing import Anyimport aiohttpSTREAM_URL = "wss://stream.resemble.ai/api/v1/detect/audio"SAMPLE_RATE = 16_000FRAME_MILLISECONDS = 100def streaming_wav_header() -> bytes:"""Create a mono PCM WAV header for a stream of unknown final length."""channels = 1bits_per_sample = 16bytes_per_sample = bits_per_sample // 8byte_rate = SAMPLE_RATE * channels * bytes_per_sampleblock_align = channels * bytes_per_samplereturn struct.pack("<4sL4s4sLHHLLHH4sL",b"RIFF",0xFFFFFFFF,b"WAVE",b"fmt ",16,1,channels,SAMPLE_RATE,byte_rate,block_align,bits_per_sample,b"data",0xFFFFFFFF,)def decode_message(message: aiohttp.WSMessage) -> dict[str, Any]:if message.type != aiohttp.WSMsgType.TEXT:raise RuntimeError(f"Expected a text message, received {message.type.name}")payload = json.loads(message.data.strip())if not isinstance(payload, dict):raise RuntimeError("The server returned an unexpected JSON value")return payloaddef print_result(payload: dict[str, Any]) -> None:message_type = payload.get("type")if message_type == "ready":print(f"[ready] stream_id={payload.get('stream_id')}")elif message_type == "chunk":chunk = payload.get("chunk_info") or {}label = chunk.get("chunk_label")score = chunk.get("chunk_aggregated_score")score_text = "" if score is None or label == "skipped" else f" score={float(score):.4f}"print(f"[chunk {chunk.get('chunk_id')}] "f"{float(chunk.get('begin_timestamp_s', 0)):.1f}-"f"{float(chunk.get('end_timestamp_s', 0)):.1f}s "f"label={label}{score_text}")elif message_type == "final":score = payload.get("aggregated_score")score_text = "N/A" if score is None else f"{float(score):.4f}"print(f"[final] label={payload.get('label')} aggregated_score={score_text}")elif message_type == "error":code = payload.get("error_code") or "stream_error"message = payload.get("error") or payload.get("error_message") or "Unknown error"raise RuntimeError(f"{code}: {message}")async def send_audio(websocket: aiohttp.ClientWebSocketResponse, audio_path: Path) -> float:"""Decode the input with ffmpeg and send paced binary PCM frames."""ffmpeg = shutil.which("ffmpeg")if not ffmpeg:raise RuntimeError("ffmpeg was not found on PATH")bytes_per_second = SAMPLE_RATE * 2 # mono signed 16-bit PCMframe_bytes = bytes_per_second * FRAME_MILLISECONDS // 1000process = await asyncio.create_subprocess_exec(ffmpeg,"-v","error","-i",str(audio_path),"-f","s16le","-acodec","pcm_s16le","-ar",str(SAMPLE_RATE),"-ac","1","pipe:1",stdout=asyncio.subprocess.PIPE,stderr=asyncio.subprocess.PIPE,)assert process.stdout is not Noneassert process.stderr is not Nonesent_bytes = 0started_at = asyncio.get_running_loop().time()await websocket.send_bytes(streaming_wav_header())try:while True:frame = await process.stdout.read(frame_bytes)if not frame:breakawait websocket.send_bytes(frame)sent_bytes += len(frame)target_elapsed = sent_bytes / bytes_per_secondactual_elapsed = asyncio.get_running_loop().time() - started_atif target_elapsed > actual_elapsed:await asyncio.sleep(target_elapsed - actual_elapsed)except BaseException:if process.returncode is None:process.kill()await process.wait()raisetry:await asyncio.wait_for(process.wait(), timeout=5)except asyncio.TimeoutError:process.kill()await process.wait()stderr = (await process.stderr.read()).decode(errors="replace").strip()if process.returncode != 0:raise RuntimeError(f"ffmpeg failed: {stderr or f'exit code {process.returncode}'}")if sent_bytes == 0:raise RuntimeError("ffmpeg decoded no audio")await websocket.send_str(json.dumps({"type": "end"}))duration = sent_bytes / bytes_per_secondprint(f"[sent] {duration:.2f}s of audio; end marker sent")return durationasync def receive_results(websocket: aiohttp.ClientWebSocketResponse,) -> dict[str, Any] | None:final_result: dict[str, Any] | None = Noneasync for message in websocket:if message.type == aiohttp.WSMsgType.ERROR:raise RuntimeError(f"WebSocket error: {websocket.exception()}")if message.type != aiohttp.WSMsgType.TEXT:continuepayload = decode_message(message)print_result(payload)if payload.get("type") == "final":final_result = payloadreturn final_resultasync def stream_audio(audio_path: Path, api_key: str) -> None:headers = {"Authorization": f"Bearer {api_key}"}params = {"filename": audio_path.name}timeout = aiohttp.ClientTimeout(total=None, sock_connect=30)final_result: dict[str, Any] | Noneasync with aiohttp.ClientSession(timeout=timeout) as session:try:async with session.ws_connect(STREAM_URL,headers=headers,params=params,heartbeat=30,) as websocket:first_message = await asyncio.wait_for(websocket.receive(), timeout=30)first_payload = decode_message(first_message)print_result(first_payload)if first_payload.get("type") != "ready":raise RuntimeError("The first server message was not ready")sender = asyncio.create_task(send_audio(websocket, audio_path))receiver = asyncio.create_task(receive_results(websocket))try:await asyncio.gather(sender, receiver)finally:for task in (sender, receiver):if not task.done():task.cancel()await asyncio.gather(sender, receiver, return_exceptions=True)final_result = receiver.result()print(f"[closed] code={websocket.close_code}")if websocket.close_code != 1000:raise RuntimeError(f"The WebSocket closed unexpectedly with code {websocket.close_code}")except aiohttp.WSServerHandshakeError as exc:raise RuntimeError(f"WebSocket handshake rejected with HTTP {exc.status}: {exc.message}") from excif final_result is None:raise RuntimeError("The server closed without returning a final result")def main() -> None:parser = argparse.ArgumentParser(description="Stream audio for deepfake detection")parser.add_argument("audio_path", type=Path)args = parser.parse_args()audio_path = args.audio_path.expanduser().resolve()if not audio_path.is_file():parser.error(f"Audio file not found: {audio_path}")api_key = os.environ.get("RESEMBLE_API_KEY", "").strip()if not api_key:parser.error("Set RESEMBLE_API_KEY before running the client")asyncio.run(stream_audio(audio_path, api_key))if __name__ == "__main__":main()
Run the Client
Pass any audio format supported by your ffmpeg installation:
RESEMBLE_API_KEY="YOUR_API_KEY" \python stream_audio_detection.py /path/to/audio.wav
For a voice-active input, output resembles:
[ready] stream_id=f241df9c-4738-48f8-8098-34e712e63bd1[chunk 0] 0.0-4.0s label=real score=0.0812[chunk 1] 4.0-8.0s label=real score=0.1029[sent] 8.74s of audio; end marker sent[chunk 2] 8.0-8.7s label=real score=0.1184[final] label=real aggregated_score=0.0974[closed] code=1000
The client intentionally sends audio at playback speed. To stream a microphone or telephony source, keep the same WAV header and PCM format, then replace the ffmpeg file reader with frames from your live audio source.
See Streaming Audio Detection (WebSocket) for message schemas, authorization behavior, session limits, and error handling.
