Skip to navigation

Stream Audio for Deepfake Detection

Send an audio file to the Deepfake Detection WebSocket in real time and print each per-window verdict followed by the final aggregate result.

What You Will Build

The Python client in this guide:

  1. Opens the authenticated production WebSocket.
  2. Uses ffmpeg to decode and resample an input file to mono, 16 kHz, signed 16-bit PCM.
  3. Sends approximately 100 milliseconds of audio per binary frame at playback speed.
  4. Receives result messages concurrently with the upload.
  5. Sends the end control message and waits for final and a normal close.

Prerequisites

  • A Resemble API key with audio Deepfake Detection access
  • Python 3.10 or newer
  • ffmpeg available on your PATH

Install the Python dependency:

python -m pip install aiohttp

Set your API key:

export RESEMBLE_API_KEY="YOUR_API_KEY"

Create the Client

Save the following as stream_audio_detection.py:

#!/usr/bin/env python3
from __future__ import annotations
import argparse
import asyncio
import json
import os
import shutil
import struct
from pathlib import Path
from typing import Any
import aiohttp
STREAM_URL = "wss://stream.resemble.ai/api/v1/detect/audio"
SAMPLE_RATE = 16_000
FRAME_MILLISECONDS = 100
def streaming_wav_header() -> bytes:
"""Create a mono PCM WAV header for a stream of unknown final length."""
channels = 1
bits_per_sample = 16
bytes_per_sample = bits_per_sample // 8
byte_rate = SAMPLE_RATE * channels * bytes_per_sample
block_align = channels * bytes_per_sample
return struct.pack(
"<4sL4s4sLHHLLHH4sL",
b"RIFF",
0xFFFFFFFF,
b"WAVE",
b"fmt ",
16,
1,
channels,
SAMPLE_RATE,
byte_rate,
block_align,
bits_per_sample,
b"data",
0xFFFFFFFF,
)
def decode_message(message: aiohttp.WSMessage) -> dict[str, Any]:
if message.type != aiohttp.WSMsgType.TEXT:
raise RuntimeError(f"Expected a text message, received {message.type.name}")
payload = json.loads(message.data.strip())
if not isinstance(payload, dict):
raise RuntimeError("The server returned an unexpected JSON value")
return payload
def print_result(payload: dict[str, Any]) -> None:
message_type = payload.get("type")
if message_type == "ready":
print(f"[ready] stream_id={payload.get('stream_id')}")
elif message_type == "chunk":
chunk = payload.get("chunk_info") or {}
label = chunk.get("chunk_label")
score = chunk.get("chunk_aggregated_score")
score_text = "" if score is None or label == "skipped" else f" score={float(score):.4f}"
print(
f"[chunk {chunk.get('chunk_id')}] "
f"{float(chunk.get('begin_timestamp_s', 0)):.1f}-"
f"{float(chunk.get('end_timestamp_s', 0)):.1f}s "
f"label={label}{score_text}"
)
elif message_type == "final":
score = payload.get("aggregated_score")
score_text = "N/A" if score is None else f"{float(score):.4f}"
print(f"[final] label={payload.get('label')} aggregated_score={score_text}")
elif message_type == "error":
code = payload.get("error_code") or "stream_error"
message = payload.get("error") or payload.get("error_message") or "Unknown error"
raise RuntimeError(f"{code}: {message}")
async def send_audio(websocket: aiohttp.ClientWebSocketResponse, audio_path: Path) -> float:
"""Decode the input with ffmpeg and send paced binary PCM frames."""
ffmpeg = shutil.which("ffmpeg")
if not ffmpeg:
raise RuntimeError("ffmpeg was not found on PATH")
bytes_per_second = SAMPLE_RATE * 2 # mono signed 16-bit PCM
frame_bytes = bytes_per_second * FRAME_MILLISECONDS // 1000
process = await asyncio.create_subprocess_exec(
ffmpeg,
"-v",
"error",
"-i",
str(audio_path),
"-f",
"s16le",
"-acodec",
"pcm_s16le",
"-ar",
str(SAMPLE_RATE),
"-ac",
"1",
"pipe:1",
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
)
assert process.stdout is not None
assert process.stderr is not None
sent_bytes = 0
started_at = asyncio.get_running_loop().time()
await websocket.send_bytes(streaming_wav_header())
try:
while True:
frame = await process.stdout.read(frame_bytes)
if not frame:
break
await websocket.send_bytes(frame)
sent_bytes += len(frame)
target_elapsed = sent_bytes / bytes_per_second
actual_elapsed = asyncio.get_running_loop().time() - started_at
if target_elapsed > actual_elapsed:
await asyncio.sleep(target_elapsed - actual_elapsed)
except BaseException:
if process.returncode is None:
process.kill()
await process.wait()
raise
try:
await asyncio.wait_for(process.wait(), timeout=5)
except asyncio.TimeoutError:
process.kill()
await process.wait()
stderr = (await process.stderr.read()).decode(errors="replace").strip()
if process.returncode != 0:
raise RuntimeError(f"ffmpeg failed: {stderr or f'exit code {process.returncode}'}")
if sent_bytes == 0:
raise RuntimeError("ffmpeg decoded no audio")
await websocket.send_str(json.dumps({"type": "end"}))
duration = sent_bytes / bytes_per_second
print(f"[sent] {duration:.2f}s of audio; end marker sent")
return duration
async def receive_results(
websocket: aiohttp.ClientWebSocketResponse,
) -> dict[str, Any] | None:
final_result: dict[str, Any] | None = None
async for message in websocket:
if message.type == aiohttp.WSMsgType.ERROR:
raise RuntimeError(f"WebSocket error: {websocket.exception()}")
if message.type != aiohttp.WSMsgType.TEXT:
continue
payload = decode_message(message)
print_result(payload)
if payload.get("type") == "final":
final_result = payload
return final_result
async def stream_audio(audio_path: Path, api_key: str) -> None:
headers = {"Authorization": f"Bearer {api_key}"}
params = {"filename": audio_path.name}
timeout = aiohttp.ClientTimeout(total=None, sock_connect=30)
final_result: dict[str, Any] | None
async with aiohttp.ClientSession(timeout=timeout) as session:
try:
async with session.ws_connect(
STREAM_URL,
headers=headers,
params=params,
heartbeat=30,
) as websocket:
first_message = await asyncio.wait_for(websocket.receive(), timeout=30)
first_payload = decode_message(first_message)
print_result(first_payload)
if first_payload.get("type") != "ready":
raise RuntimeError("The first server message was not ready")
sender = asyncio.create_task(send_audio(websocket, audio_path))
receiver = asyncio.create_task(receive_results(websocket))
try:
await asyncio.gather(sender, receiver)
finally:
for task in (sender, receiver):
if not task.done():
task.cancel()
await asyncio.gather(sender, receiver, return_exceptions=True)
final_result = receiver.result()
print(f"[closed] code={websocket.close_code}")
if websocket.close_code != 1000:
raise RuntimeError(
f"The WebSocket closed unexpectedly with code {websocket.close_code}"
)
except aiohttp.WSServerHandshakeError as exc:
raise RuntimeError(
f"WebSocket handshake rejected with HTTP {exc.status}: {exc.message}"
) from exc
if final_result is None:
raise RuntimeError("The server closed without returning a final result")
def main() -> None:
parser = argparse.ArgumentParser(description="Stream audio for deepfake detection")
parser.add_argument("audio_path", type=Path)
args = parser.parse_args()
audio_path = args.audio_path.expanduser().resolve()
if not audio_path.is_file():
parser.error(f"Audio file not found: {audio_path}")
api_key = os.environ.get("RESEMBLE_API_KEY", "").strip()
if not api_key:
parser.error("Set RESEMBLE_API_KEY before running the client")
asyncio.run(stream_audio(audio_path, api_key))
if __name__ == "__main__":
main()

Run the Client

Pass any audio format supported by your ffmpeg installation:

RESEMBLE_API_KEY="YOUR_API_KEY" \
python stream_audio_detection.py /path/to/audio.wav

For a voice-active input, output resembles:

[ready] stream_id=f241df9c-4738-48f8-8098-34e712e63bd1
[chunk 0] 0.0-4.0s label=real score=0.0812
[chunk 1] 4.0-8.0s label=real score=0.1029
[sent] 8.74s of audio; end marker sent
[chunk 2] 8.0-8.7s label=real score=0.1184
[final] label=real aggregated_score=0.0974
[closed] code=1000

The client intentionally sends audio at playback speed. To stream a microphone or telephony source, keep the same WAV header and PCM format, then replace the ffmpeg file reader with frames from your live audio source.

See Streaming Audio Detection (WebSocket) for message schemas, authorization behavior, session limits, and error handling.