> For clean Markdown of any page, append .md to the page URL.
> For a complete documentation index, see https://docs.resemble.ai/llms.txt.
> For AI client integration (Claude Code, Cursor, etc.), connect to the MCP server at https://docs.resemble.ai/_mcp/server.

# Stream Audio for Deepfake Detection

Send an audio file to the Deepfake Detection WebSocket in real time and print each per-window verdict followed by the final aggregate result.

## What You Will Build

The Python client in this guide:

1. Opens the authenticated production WebSocket.
2. Uses `ffmpeg` to decode and resample an input file to mono, 16 kHz, signed 16-bit PCM.
3. Sends approximately 100 milliseconds of audio per binary frame at playback speed.
4. Receives result messages concurrently with the upload.
5. Sends the `end` control message and waits for `final` and a normal close.

## Prerequisites

* A Resemble API key with audio Deepfake Detection access
* Python 3.10 or newer
* [`ffmpeg`](https://ffmpeg.org/download.html) available on your `PATH`

Install the Python dependency:

```bash
python -m pip install aiohttp
```

Set your API key:

```bash
export RESEMBLE_API_KEY="YOUR_API_KEY"
```

## Create the Client

Save the following as `stream_audio_detection.py`:

```python
#!/usr/bin/env python3

from __future__ import annotations

import argparse
import asyncio
import json
import os
import shutil
import struct
from pathlib import Path
from typing import Any

import aiohttp


STREAM_URL = "wss://stream.resemble.ai/api/v1/detect/audio"
SAMPLE_RATE = 16_000
FRAME_MILLISECONDS = 100


def streaming_wav_header() -> bytes:
    """Create a mono PCM WAV header for a stream of unknown final length."""
    channels = 1
    bits_per_sample = 16
    bytes_per_sample = bits_per_sample // 8
    byte_rate = SAMPLE_RATE * channels * bytes_per_sample
    block_align = channels * bytes_per_sample
    return struct.pack(
        "<4sL4s4sLHHLLHH4sL",
        b"RIFF",
        0xFFFFFFFF,
        b"WAVE",
        b"fmt ",
        16,
        1,
        channels,
        SAMPLE_RATE,
        byte_rate,
        block_align,
        bits_per_sample,
        b"data",
        0xFFFFFFFF,
    )


def decode_message(message: aiohttp.WSMessage) -> dict[str, Any]:
    if message.type != aiohttp.WSMsgType.TEXT:
        raise RuntimeError(f"Expected a text message, received {message.type.name}")
    payload = json.loads(message.data.strip())
    if not isinstance(payload, dict):
        raise RuntimeError("The server returned an unexpected JSON value")
    return payload


def print_result(payload: dict[str, Any]) -> None:
    message_type = payload.get("type")
    if message_type == "ready":
        print(f"[ready] stream_id={payload.get('stream_id')}")
    elif message_type == "chunk":
        chunk = payload.get("chunk_info") or {}
        label = chunk.get("chunk_label")
        score = chunk.get("chunk_aggregated_score")
        score_text = "" if score is None or label == "skipped" else f" score={float(score):.4f}"
        print(
            f"[chunk {chunk.get('chunk_id')}] "
            f"{float(chunk.get('begin_timestamp_s', 0)):.1f}-"
            f"{float(chunk.get('end_timestamp_s', 0)):.1f}s "
            f"label={label}{score_text}"
        )
    elif message_type == "final":
        score = payload.get("aggregated_score")
        score_text = "N/A" if score is None else f"{float(score):.4f}"
        print(f"[final] label={payload.get('label')} aggregated_score={score_text}")
    elif message_type == "error":
        code = payload.get("error_code") or "stream_error"
        message = payload.get("error") or payload.get("error_message") or "Unknown error"
        raise RuntimeError(f"{code}: {message}")


async def send_audio(websocket: aiohttp.ClientWebSocketResponse, audio_path: Path) -> float:
    """Decode the input with ffmpeg and send paced binary PCM frames."""
    ffmpeg = shutil.which("ffmpeg")
    if not ffmpeg:
        raise RuntimeError("ffmpeg was not found on PATH")

    bytes_per_second = SAMPLE_RATE * 2  # mono signed 16-bit PCM
    frame_bytes = bytes_per_second * FRAME_MILLISECONDS // 1000
    process = await asyncio.create_subprocess_exec(
        ffmpeg,
        "-v",
        "error",
        "-i",
        str(audio_path),
        "-f",
        "s16le",
        "-acodec",
        "pcm_s16le",
        "-ar",
        str(SAMPLE_RATE),
        "-ac",
        "1",
        "pipe:1",
        stdout=asyncio.subprocess.PIPE,
        stderr=asyncio.subprocess.PIPE,
    )
    assert process.stdout is not None
    assert process.stderr is not None

    sent_bytes = 0
    started_at = asyncio.get_running_loop().time()
    await websocket.send_bytes(streaming_wav_header())

    try:
        while True:
            frame = await process.stdout.read(frame_bytes)
            if not frame:
                break

            await websocket.send_bytes(frame)
            sent_bytes += len(frame)

            target_elapsed = sent_bytes / bytes_per_second
            actual_elapsed = asyncio.get_running_loop().time() - started_at
            if target_elapsed > actual_elapsed:
                await asyncio.sleep(target_elapsed - actual_elapsed)
    except BaseException:
        if process.returncode is None:
            process.kill()
        await process.wait()
        raise

    try:
        await asyncio.wait_for(process.wait(), timeout=5)
    except asyncio.TimeoutError:
        process.kill()
        await process.wait()

    stderr = (await process.stderr.read()).decode(errors="replace").strip()
    if process.returncode != 0:
        raise RuntimeError(f"ffmpeg failed: {stderr or f'exit code {process.returncode}'}")
    if sent_bytes == 0:
        raise RuntimeError("ffmpeg decoded no audio")

    await websocket.send_str(json.dumps({"type": "end"}))
    duration = sent_bytes / bytes_per_second
    print(f"[sent] {duration:.2f}s of audio; end marker sent")
    return duration


async def receive_results(
    websocket: aiohttp.ClientWebSocketResponse,
) -> dict[str, Any] | None:
    final_result: dict[str, Any] | None = None
    async for message in websocket:
        if message.type == aiohttp.WSMsgType.ERROR:
            raise RuntimeError(f"WebSocket error: {websocket.exception()}")
        if message.type != aiohttp.WSMsgType.TEXT:
            continue

        payload = decode_message(message)
        print_result(payload)
        if payload.get("type") == "final":
            final_result = payload
    return final_result


async def stream_audio(audio_path: Path, api_key: str) -> None:
    headers = {"Authorization": f"Bearer {api_key}"}
    params = {"filename": audio_path.name}
    timeout = aiohttp.ClientTimeout(total=None, sock_connect=30)
    final_result: dict[str, Any] | None

    async with aiohttp.ClientSession(timeout=timeout) as session:
        try:
            async with session.ws_connect(
                STREAM_URL,
                headers=headers,
                params=params,
                heartbeat=30,
            ) as websocket:
                first_message = await asyncio.wait_for(websocket.receive(), timeout=30)
                first_payload = decode_message(first_message)
                print_result(first_payload)
                if first_payload.get("type") != "ready":
                    raise RuntimeError("The first server message was not ready")

                sender = asyncio.create_task(send_audio(websocket, audio_path))
                receiver = asyncio.create_task(receive_results(websocket))
                try:
                    await asyncio.gather(sender, receiver)
                finally:
                    for task in (sender, receiver):
                        if not task.done():
                            task.cancel()
                    await asyncio.gather(sender, receiver, return_exceptions=True)

                final_result = receiver.result()

                print(f"[closed] code={websocket.close_code}")
                if websocket.close_code != 1000:
                    raise RuntimeError(
                        f"The WebSocket closed unexpectedly with code {websocket.close_code}"
                    )
        except aiohttp.WSServerHandshakeError as exc:
            raise RuntimeError(
                f"WebSocket handshake rejected with HTTP {exc.status}: {exc.message}"
            ) from exc

    if final_result is None:
        raise RuntimeError("The server closed without returning a final result")


def main() -> None:
    parser = argparse.ArgumentParser(description="Stream audio for deepfake detection")
    parser.add_argument("audio_path", type=Path)
    args = parser.parse_args()

    audio_path = args.audio_path.expanduser().resolve()
    if not audio_path.is_file():
        parser.error(f"Audio file not found: {audio_path}")

    api_key = os.environ.get("RESEMBLE_API_KEY", "").strip()
    if not api_key:
        parser.error("Set RESEMBLE_API_KEY before running the client")

    asyncio.run(stream_audio(audio_path, api_key))


if __name__ == "__main__":
    main()
```

## Run the Client

Pass any audio format supported by your `ffmpeg` installation:

```bash
RESEMBLE_API_KEY="YOUR_API_KEY" \
  python stream_audio_detection.py /path/to/audio.wav
```

For a voice-active input, output resembles:

```text
[ready] stream_id=f241df9c-4738-48f8-8098-34e712e63bd1
[chunk 0] 0.0-4.0s label=real score=0.0812
[chunk 1] 4.0-8.0s label=real score=0.1029
[sent] 8.74s of audio; end marker sent
[chunk 2] 8.0-8.7s label=real score=0.1184
[final] label=real aggregated_score=0.0974
[closed] code=1000
```

The client intentionally sends audio at playback speed. To stream a microphone or telephony source, keep the same WAV header and PCM format, then replace the `ffmpeg` file reader with frames from your live audio source.

See [Streaming Audio Detection (WebSocket)](/detect/streaming) for message schemas, authorization behavior, session limits, and error handling.