Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
62 changes: 62 additions & 0 deletions .github/scripts/fuzz.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,62 @@
#!/usr/bin/env bash
#
# Builds the extension with libFuzzer coverage, AddressSanitizer and UndefinedBehaviorSanitizer and runs a fuzz
# target of tests/fuzz with Atheris. Linux only: Apple Clang has no libFuzzer. In the manylinux image:
# docker run --rm --platform linux/amd64 -v "$PWD:/src" -w /src \
# quay.io/pypa/manylinux_2_28_x86_64 .github/scripts/fuzz.sh video_frame -max_total_time=600
#
# The first argument is the target (fuzz_<target>.py), the rest are libFuzzer's. The corpus grows in
# tests/fuzz/corpus/<target>, crashes are written to tests/fuzz/crashes. Pass a crash file to replay it.

set -euo pipefail

SRC="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
BUILD="${WRTC_FUZZ_BUILD_DIR:-/tmp/wrtc-fuzz}"
PYTHON="${PYTHON:-/opt/python/cp313-cp313/bin/python}"
TARGET="${1:?usage: fuzz.sh <target> [libFuzzer arguments]}"
shift

if ! command -v clang > /dev/null; then
dnf install -y -q clang lld compiler-rt llvm
fi

if [ ! -x "$BUILD/venv/bin/python" ]; then
"$PYTHON" -m venv "$BUILD/venv"
# built from source with this Clang, so its sanitizer runtime is the one the extension is instrumented for
CLANG_BIN="$(command -v clang)" LIBFUZZER_LIB="$(clang -print-runtime-dir)/libclang_rt.fuzzer_no_main.a" \
"$BUILD/venv/bin/pip" install -q --no-binary atheris atheris \
cmake ninja "pybind11>=3.0" pytest
fi
# shellcheck disable=SC1091
source "$BUILD/venv/bin/activate"

CC=clang CXX=clang++ cmake -S "$SRC" -B "$BUILD/build" -G Ninja \
-DCMAKE_BUILD_TYPE=RelWithDebInfo \
-DCMAKE_UNITY_BUILD=ON \
-DWRTC_SANITIZE=fuzzer-no-link,address,undefined \
-Dpybind11_DIR="$(python -m pybind11 --cmakedir)" > /dev/null
cmake --build "$BUILD/build"

# The interpreter isn't instrumented: libFuzzer and ASan must be loaded before anything else
LD_PRELOAD="$(python -c 'import atheris; print(atheris.path())')/asan_with_fuzzer.so"
export LD_PRELOAD
export PYTHONMALLOC=malloc
export ASAN_OPTIONS="detect_leaks=0:halt_on_error=1:abort_on_error=0:detect_stack_use_after_return=0"
export UBSAN_OPTIONS="print_stacktrace=1:halt_on_error=1"
ASAN_SYMBOLIZER_PATH="$(command -v llvm-symbolizer)"
export ASAN_SYMBOLIZER_PATH
export PYTHONPATH="$BUILD/build/python-webrtc/cpp:$SRC/python-webrtc/python:$SRC/tests/fuzz"
export PYTHONDONTWRITEBYTECODE=1

CORPUS="$SRC/tests/fuzz/corpus/$TARGET"
mkdir -p "$CORPUS" "$SRC/tests/fuzz/crashes"
# a file argument replays it instead of fuzzing
if [ $# -gt 0 ] && [ -f "$1" ]; then
CRASH="$(realpath "$1")"
shift
cd "$SRC/tests/fuzz"
exec python "fuzz_$TARGET.py" "$CRASH" "$@"
fi
cd "$SRC/tests/fuzz"
exec python "fuzz_$TARGET.py" "$CORPUS" -artifact_prefix="$SRC/tests/fuzz/crashes/$TARGET-" \
-max_len=4096 -timeout=30 -rss_limit_mb=4096 "$@"
4 changes: 4 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -36,3 +36,7 @@ python-webrtc/python/*.session-journal
python-webrtc/python/*.raw
python-webrtc/python/snippet_test.py
python-webrtc/python/test_ub.py

# fuzzing (tests/fuzz): crashes and the corpora libFuzzer grows
/tests/fuzz/crashes/
/tests/fuzz/corpus/
8 changes: 7 additions & 1 deletion Makefile
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
.PHONY: dev test asan tsan lint format format-check tidy stub wheels doc clean
.PHONY: dev test asan tsan fuzz lint format format-check tidy stub wheels doc clean

# pinned to the clang-tidy of .github/scripts/tidy.sh
CLANG_FORMAT := uvx clang-format==22.1.8
Expand All @@ -20,6 +20,12 @@ asan:
tsan:
SANITIZE=thread .github/scripts/sanitizers-macos.sh $(O)

# a target of tests/fuzz with Atheris in the Linux image, the build kept in build/fuzz: make fuzz T=video_frame O=-max_total_time=600
fuzz:
docker run --rm -it --platform linux/amd64 -v "$(CURDIR):/src" -v "$(CURDIR)/build/fuzz:/tmp/wrtc-fuzz" \
-e WRTC_CACHE_DIR=/tmp/wrtc-fuzz/cache -w /src \
quay.io/pypa/manylinux_2_28_x86_64 .github/scripts/fuzz.sh $(T) $(O)

lint: format-check
uvx ruff check
uvx ruff format --check
Expand Down
5 changes: 4 additions & 1 deletion python-webrtc/cpp/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -27,7 +27,10 @@ if(WRTC_SANITIZE)
-fsanitize=${WRTC_SANITIZE} -fno-sanitize=vptr -fno-sanitize-recover=all -fno-omit-frame-pointer)
target_link_options(${MODULE} PRIVATE -fsanitize=${WRTC_SANITIZE} -fno-sanitize=vptr)
if(WRTC_SANITIZE MATCHES "address")
target_link_options(${MODULE} PRIVATE -shared-libasan)
# Atheris preloads its own runtime with libFuzzer (asan_with_fuzzer.so), which must provide the symbols
if(NOT WRTC_SANITIZE MATCHES "fuzzer")
target_link_options(${MODULE} PRIVATE -shared-libasan)
endif()
# The libc++ inside libwebrtc isn't instrumented (no _LIBCPP_INSTRUMENTED_WITH_ASAN): containers it touches
# aren't annotated, so annotating them in our code alone reports false container-overflows.
target_compile_definitions(${MODULE} PRIVATE __SANITIZER_DISABLE_CONTAINER_OVERFLOW__)
Expand Down
8 changes: 7 additions & 1 deletion python-webrtc/cpp/src/media/audio_samples.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
#include "audio_samples.h"

#include <algorithm>
#include <cmath>
#include <cstdint>
#include <cstring>
#include <string_view>
Expand Down Expand Up @@ -92,7 +93,12 @@ namespace python_webrtc {
case SampleType::S32:
return Load<int32_t>(data);
default: {
const double scaled = std::clamp(static_cast<double>(Load<float>(data)), -1.0, 1.0) * kS32Scale;
const double sample = Load<float>(data);
// NaN is silence: clamping keeps it, and converting it to an integer is undefined
if (std::isnan(sample)) {
return 0;
}
const double scaled = std::clamp(sample, -1.0, 1.0) * kS32Scale;
return static_cast<int32_t>(std::clamp(scaled, -kS32Scale, kS32Scale - 1));
}
}
Expand Down
2 changes: 2 additions & 0 deletions python-webrtc/python/webrtc/models/video_frame.py
Original file line number Diff line number Diff line change
Expand Up @@ -346,6 +346,8 @@ def _layout(value: Any) -> Optional[List[PlaneLayout]]:

def _rotation(value: float) -> int:
"""The nearest multiple of 90, ties rounded up, from 0 to 270"""
if not math.isfinite(value):
raise TypeError(f'The rotation must be finite, not {value!r}')
return int(math.floor(value / 90 + 0.5) * 90) % 360


Expand Down
22 changes: 22 additions & 0 deletions tests/fuzz/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
# Fuzzing

Coverage-guided fuzz targets for [Atheris](https://github.com/google/atheris) (libFuzzer for Python). The extension
is built with libFuzzer coverage, ASan and UBSan; libwebrtc isn't instrumented, but its `RTC_CHECK` aborts and
crashes are caught.

| Target | What it fuzzes |
| --- | --- |
| `video_frame` | `VideoFrame` from buffers, `copy_to` with rects, layouts and RGB conversions, frames of frames |
| `audio_data` | `AudioData` and `copy_to` with every format and layout conversion |
| `native_buffers` | The native `wrtc.VideoFrameBuffer` and `wrtc.copyAudioSamples` directly, without the Python checks |
| `generator` | `AudioData` and `VideoFrame` written to generators sent over a connection, read back by processors |

Linux only (Apple Clang has no libFuzzer), in the manylinux image; the build is kept in `build/fuzz`:

```sh
make fuzz T=video_frame O=-max_total_time=600
make fuzz T=audio_data O=tests/fuzz/crashes/audio_data-crash-... # replays a crash
```

Only the documented exceptions are expected, and a few oracles check results (a frame or samples copied out as they
are give the same bytes). A crash becomes a regression test in `tests/`.
87 changes: 87 additions & 0 deletions tests/fuzz/_input.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,87 @@
#
# Copyright 2026 Ilya (Marshal) <https://github.com/MarshalX>. All rights reserved.
#
# Use of this source code is governed by a BSD-style license
# that can be found in the LICENSE.md file in the root of the project.
#

"""Values for the fuzz targets, drawn from the fuzzer's bytes and biased to the edges where checks go wrong."""

import atheris

# where sizes overflow: 16, 24 (the native limit of a frame side), 31, 32 and 64 bits
EDGES = [0, 1, 2, 3, 4, 7, 8, 15, 16, 255, 256, 2**16 - 1, 2**16, 2**24, 2**24 + 1, 2**31 - 1, 2**31]
EDGES += [2**32 - 1, 2**32, 2**32 + 1, 2**63 - 1, 2**63, 2**64 - 1, 2**64, 2**65]
FLOATS = [0.0, -0.0, 0.5, 1.5, -1.0, 1e-300, 1e300, float('inf'), float('-inf'), float('nan')]


class Input:
def __init__(self, data: bytes):
self._fdp = atheris.FuzzedDataProvider(data)

def flag(self) -> bool:
return self._fdp.ConsumeBool()

def choice(self, values):
return self._fdp.PickValueInList(list(values))

def small(self, limit: int = 64) -> int:
return self._fdp.ConsumeIntInRange(0, limit)

def unsigned(self, limit: int = 64) -> int:
"""Mostly small, sometimes an edge"""
mode = self._fdp.ConsumeIntInRange(0, 7)
if mode == 0:
return self.choice(EDGES)
if mode == 1:
return self._fdp.ConsumeIntInRange(0, 2**64 - 1)
return self.small(limit)

def integer(self, limit: int = 64):
"""An unsigned value, a negative one, or something that isn't an integer"""
mode = self._fdp.ConsumeIntInRange(0, 15)
if mode == 0:
return -self.unsigned(limit) - 1
if mode == 1:
return self.number()
if mode == 2:
return self.choice([None, True, '1', b'1'])
return self.unsigned(limit)

def number(self, limit: int = 64):
mode = self._fdp.ConsumeIntInRange(0, 3)
if mode == 0:
return self.choice(FLOATS)
if mode == 1:
return self.small(limit) + self.choice([0, 0, 0.5, 1e-9])
return self.small(limit)

def maybe(self, make):
return make() if self.flag() else None

def buffer(self, size: int):
"""size bytes as bytes, a bytearray, or a view of one, which may be strided"""
data = self._fdp.ConsumeBytes(size)
data += bytes(size - len(data))
kind = self._fdp.ConsumeIntInRange(0, 7)
if kind == 0:
return bytearray(data)
if kind == 1:
return memoryview(bytearray(data * 2))[::2]
if kind == 2:
return memoryview(bytearray(data))[::-1]
if kind == 3:
return memoryview(bytearray(data)).cast('B', [size, 1]) if size else memoryview(bytearray())
return data

def destination(self, size: int):
"""A writable buffer of about size bytes"""
size = max(0, size + self.choice([0, 0, 0, -1, 1, 64]))
kind = self._fdp.ConsumeIntInRange(0, 7)
if kind == 0:
return memoryview(bytearray(size * 2))[::2]
if kind == 1:
return memoryview(bytearray(size))[::-1]
if kind == 2:
return bytes(size)
return bytearray(size)
79 changes: 79 additions & 0 deletions tests/fuzz/fuzz_audio_data.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,79 @@
#
# Copyright 2026 Ilya (Marshal) <https://github.com/MarshalX>. All rights reserved.
#
# Use of this source code is governed by a BSD-style license
# that can be found in the LICENSE.md file in the root of the project.
#

"""Fuzzes AudioData: creation, and copy_to with every conversion of format and layout."""

import sys

import atheris

with atheris.instrument_imports():
from _input import Input

import webrtc

FORMATS = list(webrtc.AudioSampleFormat)
EXPECTED = (TypeError, ValueError, BufferError, webrtc.NotSupportedError, webrtc.InvalidStateError)
SAMPLE_BYTES = {'u8': 1, 's16': 2, 's32': 4, 'f32': 4}


def check_identity(audio: webrtc.AudioData, data: bytes) -> None:
"""Interleaved samples copied out in their own format are the same bytes"""
if audio.format.value.endswith('-planar'):
return
out = bytearray(audio.allocation_size({'plane_index': 0}))
audio.copy_to(out, {'plane_index': 0})
assert bytes(out) == data[: len(out)], f'{audio!r} copied out differently'


def test_one_input(data: bytes) -> None:
inp = Input(data)
format = inp.choice(FORMATS)
frames, channels = inp.integer(512), inp.integer(8)
sample_rate = inp.number(96000) if inp.flag() else 48000
size = inp.small(1 << 14)
if isinstance(frames, int) and isinstance(channels, int) and inp.flag():
# exactly enough samples, when it's not too many
exact = frames * channels * SAMPLE_BYTES[format.value.split('-')[0]]
size = exact if 0 <= exact < 1 << 16 else size
source = inp.buffer(size)
try:
audio = webrtc.AudioData(
format=format,
sample_rate=sample_rate,
number_of_frames=frames,
number_of_channels=channels,
timestamp=inp.integer(),
data=source,
)
except EXPECTED:
return
check_identity(audio, bytes(source))
for _ in range(inp.small(4)):
options = {'plane_index': inp.integer(8)}
if inp.flag():
options['frame_offset'] = inp.integer(512)
if inp.flag():
options['frame_count'] = inp.integer(512)
if inp.flag():
options['format'] = inp.choice(FORMATS)
try:
size = audio.allocation_size(options)
audio.copy_to(inp.destination(min(size, 1 << 20)), options)
except EXPECTED:
pass
if inp.small(8) == 0:
audio = audio.clone()


def main() -> None:
atheris.Setup(sys.argv, test_one_input)
atheris.Fuzz()


if __name__ == '__main__':
main()
Loading
Loading