diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..5076716 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,197 @@ +name: CI + +on: + push: + pull_request: + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: ci-${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + +env: + CARGO_BUILD_JOBS: 3 + CARGO_TERM_COLOR: always + +jobs: + style: + runs-on: ubuntu-24.04 + timeout-minutes: 15 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: dtolnay/rust-toolchain@7e38f4b43b4db5c8dd498af069a4f6196df1d067 + with: + toolchain: stable + components: rustfmt,clippy + - name: Install tools + run: | + sudo apt-get update + sudo apt-get install -y nasm pipx + pipx install clang-format==23.1.1 + echo "$HOME/.local/bin" >> "$GITHUB_PATH" + - name: Format and lint + run: | + ./scripts/stylecheck.sh + git diff --exit-code + + tests: + runs-on: ubuntu-24.04 + timeout-minutes: 25 + strategy: + fail-fast: false + matrix: + asm: ['true', 'false'] + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: halidecx/wpd-test-data + ref: f8c31341db3ab4400f048a96e7b3736fed303b34 + path: wpd-test-data + persist-credentials: false + - uses: dtolnay/rust-toolchain@7e38f4b43b4db5c8dd498af069a4f6196df1d067 + with: + toolchain: stable + - name: Install tools + run: | + sudo apt-get update + sudo apt-get install -y pipx ninja-build nasm cmake + pipx install meson==1.12.1 + echo "$HOME/.local/bin" >> "$GITHUB_PATH" + - name: Tests, C API, parity and checkasm + env: + ASM: ${{ matrix.asm }} + run: | + meson setup build -Denable_asm="$ASM" -Dtrim_dsp=false \ + -Dtestdata_tests=true -Dlibwebp=subproject -Dwuffs=disabled + if [ "$ASM" = true ]; then + meson test -C build --list | grep -q checkasm + fi + meson test -C build --print-errorlogs --num-processes 3 + + scripts: + runs-on: ubuntu-24.04 + timeout-minutes: 45 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: halidecx/wpd-test-data + ref: f8c31341db3ab4400f048a96e7b3736fed303b34 + path: wpd-test-data + persist-credentials: false + - uses: dtolnay/rust-toolchain@7e38f4b43b4db5c8dd498af069a4f6196df1d067 + with: + toolchain: stable + - name: Install tools + run: | + sudo apt-get update + sudo apt-get install -y pipx ninja-build nasm cmake webp + pipx install meson==1.12.1 + echo "$HOME/.local/bin" >> "$GITHUB_PATH" + - name: Build assembly and fallback tools + run: | + meson setup build -Dtrim_dsp=false -Dtestdata_tests=true \ + -Dlibwebp=subproject -Dwuffs=disabled + meson compile -C build -j 3 libwebpdec + meson setup build-noasm -Denable_asm=false \ + -Dlibwebp=disabled -Dwuffs=disabled + meson compile -C build-noasm -j 3 + - name: Existing correctness checks + run: | + ./scripts/testdata.sh + ./scripts/animcheck.sh + ./scripts/threadcheck.sh + ./scripts/md5check.sh ./build-noasm/wpd ./build/wpd + ./scripts/clicheck.sh ./build-noasm/wpd ./build/wpd + ./scripts/rac32.sh --print-errorlogs --num-processes 3 + + c-sanitizers: + runs-on: ubuntu-24.04 + timeout-minutes: 30 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: halidecx/wpd-test-data + ref: f8c31341db3ab4400f048a96e7b3736fed303b34 + path: wpd-test-data + persist-credentials: false + - uses: dtolnay/rust-toolchain@7e38f4b43b4db5c8dd498af069a4f6196df1d067 + with: + toolchain: stable + - name: Install tools + run: | + sudo apt-get update + sudo apt-get install -y pipx ninja-build nasm cmake + pipx install meson==1.12.1 + echo "$HOME/.local/bin" >> "$GITHUB_PATH" + - name: C harnesses and damaged-input smoke + run: | + ./scripts/sanitize.sh --print-errorlogs --num-processes 3 + ./scripts/fuzz.sh 8 wpd-test-data/odd_lossy.webp \ + wpd-test-data/odd_a_lossy.webp wpd-test-data/palette_rgb.webp \ + wpd-test-data/mixed_codecs.webp + + nightly: + runs-on: ubuntu-24.04 + timeout-minutes: 45 + strategy: + fail-fast: false + matrix: + check: [rustsan, tsan, miri, fuzz] + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: halidecx/wpd-test-data + ref: f8c31341db3ab4400f048a96e7b3736fed303b34 + path: wpd-test-data + persist-credentials: false + - uses: dtolnay/rust-toolchain@7e38f4b43b4db5c8dd498af069a4f6196df1d067 + with: + toolchain: nightly + components: rust-src,miri + - name: Install tools + run: | + sudo apt-get update + sudo apt-get install -y pipx ninja-build nasm cmake clang + pipx install meson==1.12.1 + echo "$HOME/.local/bin" >> "$GITHUB_PATH" + - name: Nightly check + env: + CHECK: ${{ matrix.check }} + run: | + case "$CHECK" in + rustsan) + meson setup build -Dlibwebp=disabled -Dwuffs=disabled + meson compile -C build -j 3 + ./scripts/rustsan.sh + ;; + tsan) ./scripts/tsan.sh ;; + miri) ./scripts/miri.sh --lib container::tests ;; + fuzz) + cargo install cargo-fuzz --version 0.13.2 --locked + ./scripts/fuzz-smoke.sh + ;; + esac + - name: Save fuzz failures + if: failure() && matrix.check == 'fuzz' + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: fuzz-failures + path: fuzz/artifacts + if-no-files-found: ignore diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..99139ca --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,31 @@ +# Changelog + +## Unreleased — 0.2.0 + +### Build and CI + +- Repair the end-to-end fuzz target's decoder options initializer so all four + coverage-guided targets build again. +- Correct the optional Wuffs dependency name in the Meson build. +- Add one GitHub Actions workflow for tests, rustfmt/clippy, fuzz target builds, + seeded smoke runs, and the existing correctness and sanitizer checks. +- Bound fuzz-harness pictures to one megapixel so mutated dimensions fit the + smoke run's memory budget without changing decoder limits. + +### libwebp compatibility + +- Reject corrupted first-chunk FourCCs instead of silently treating an extended + container as a simple image. +- Add opt-in libwebp still compatibility through Rust, C and `--libwebp-compat`, + retaining strict decoding by default. Match duplicate VP8X chunks, shortened + image chunks, damaged trailing chunks and simple-image final padding. +- Retain the complete compatible still payload until streaming EOF and preserve + animation container validation. +- Match portable libwebp VP8 transforms on non-conforming lossy streams, with + damaged-stream regressions. Save and restore each decoder's strict DSP tables + when toggling the mode; native libwebp CPU output can still differ. +- Accept a final alpha entropy symbol crossing EOF only when compatibility mode + has already produced the complete alpha plane. +- Add a directory comparison script for libwebp 1.6.0 `dwebp` and `anim_dump`, + distinguishing visible differences from transparent RGB. +- Fuzz strict and compatible decoding with public and damaged regression seeds. diff --git a/README.md b/README.md index 4da01c4..a2b1cac 100644 --- a/README.md +++ b/README.md @@ -90,6 +90,24 @@ Test data is maintained at into the `wpd/` root. `./scripts/testdata.sh` runs end-to-end assembly and fallback checks. +GitHub Actions runs assembly and fallback tests, checkasm, format/lint checks, +the correctness scripts, C and Rust sanitizers, and a container Miri smoke +check. The corpus revision is pinned in `.github/workflows/ci.yml`. CI compares +the assembly and fallback tools with `md5check.sh` and `clicheck.sh`; comparing +an older release still needs an explicit baseline binary. Timing scripts +(`bench.sh` and `cmpbench.sh`) remain manual because shared CI runners do not +provide stable performance measurements. + +`./scripts/fuzz-smoke.sh [seconds-per-target] [corpus-directory]` builds every +fuzz target and runs each for 15 seconds by default. It requires nightly Rust, +Python 3 and cargo-fuzz 0.13.2. It derives container and raw VP8/VP8L seeds from +the test corpus without changing its WebP files, and leaves generated seeds and +failure artifacts under `fuzz/`. Longer fuzzing and the full safe-core Miri +suite (`./scripts/miri.sh`) are useful local checks before releases. The +decoding harnesses bound pictures to 1,048,576 pixels so mutated dimensions fit +the smoke run's memory budget; larger pictures remain in ordinary corpus tests. +This is a harness limit, not a decoder limit. + For libwebp parity testing and benchmarking against alternative WebP decoders, build the optional third-party test binaries: @@ -127,3 +145,35 @@ This project would not be possible without: - [dav1d](https://www.videolan.org/projects/dav1d.html), the fastest open-source AV1 decoder - [rav1d](https://github.com/memorysafety/rav1d), a Rust rewrite of dav1d + +## libwebp compatibility + +Strict decoding remains the default. `--libwebp-compat` accepts damaged still +containers that libwebp's still decoder accepts, including shortened image +chunks, duplicate VP8X chunks, and damaged trailing chunks. It also selects +libwebp's portable C VP8 transform behaviour for non-conforming lossy streams. +It applies to pixel output and text info. Animation containers retain their +validation. + +Set Rust's `Options.libwebp_compat` before opening input. C callers set +`WPDDecoderOptions.libwebp_compat` to 1 and initialize `struct_size` with +`sizeof(WPDDecoderOptions)`; older options structs retain strict decoding. + +Compatible still streams produce pixels after `end_of_stream`, retaining +physical input bytes past RIFF's declared end when the codec needs them. Default +still decoding and animation decoding remain incremental. Library callers should +bound incoming bytes; the CLI's `--max-input` and the normal frame and output +limits still apply. + +Compare a directory with libwebp 1.6.0's `dwebp` and `anim_dump`: + +```sh +python3 scripts/webpcompare.py path/to/corpus --wpd build/wpd \ + --libwebp-compat --noasm +``` + +`--noasm` selects portable C for stills through `dwebp`; `anim_dump` remains +native. Libwebp's CPU-specific code can render malformed lossy streams +differently from its portable code, so this mode does not promise identical +pixels from every native libwebp build on non-conforming input. The comparison +reports visible differences separately from RGB differences under zero alpha. diff --git a/capi/src/decoder.rs b/capi/src/decoder.rs index e0ad8b0..4c6dd57 100644 --- a/capi/src/decoder.rs +++ b/capi/src/decoder.rs @@ -234,6 +234,7 @@ fn set_options( || !flag(options.use_cropping) || !flag(options.use_scaling) || !flag(options.flip) + || !flag(options.libwebp_compat) { return Err(decoder.fail("invalid decoder options", Error::InvalidArgument)); } @@ -299,6 +300,9 @@ entry!(fn wpd_decoder_set_options(decoder, options: *const WPDDecoderOptions) { local.frame_size_limit = unsafe { ptr::addr_of!((*options).frame_size_limit).read() }; } + if size >= WPDDecoderOptions::v4() { + local.libwebp_compat = unsafe { ptr::addr_of!((*options).libwebp_compat).read() }; + } reported(set_options(decoder, &local).map(|()| WPD_OK)) }); diff --git a/capi/src/options.rs b/capi/src/options.rs index cbd42c3..615cf26 100644 --- a/capi/src/options.rs +++ b/capi/src/options.rs @@ -24,6 +24,9 @@ pub struct WPDDecoderOptions { /// Takes the v2 struct's tail padding, for the reason `reserved` took v1's. pub reserved2: c_int, pub frame_size_limit: c_uint, + /// Takes the v3 struct's tail padding. + pub reserved3: c_int, + pub libwebp_compat: c_int, } /// What `sizeof` gave the v1 struct: it ended at `flip`, and padded out to the @@ -42,6 +45,11 @@ const V2_SIZE: usize = const _: () = assert!(V2_SIZE < WPDDecoderOptions::v3()); +const V3_SIZE: usize = + WPDDecoderOptions::v3().next_multiple_of(mem::align_of::()); + +const _: () = assert!(V3_SIZE < WPDDecoderOptions::v4()); + impl WPDDecoderOptions { pub(crate) const fn v1() -> usize { mem::offset_of!(WPDDecoderOptions, flip) + mem::size_of::() @@ -55,6 +63,10 @@ impl WPDDecoderOptions { mem::offset_of!(WPDDecoderOptions, frame_size_limit) + mem::size_of::() } + pub(crate) const fn v4() -> usize { + mem::offset_of!(WPDDecoderOptions, libwebp_compat) + mem::size_of::() + } + /// Legacy callers retain serial decoding and serial log callbacks. pub(crate) fn to_core(&self) -> Options { Options { @@ -79,6 +91,7 @@ impl WPDDecoderOptions { } else { 0 }, + libwebp_compat: self.struct_size >= Self::v4() && self.libwebp_compat != 0, } } } @@ -110,4 +123,17 @@ mod tests { options.struct_size = mem::size_of::(); assert_eq!(options.to_core().frame_size_limit, 64); } + + #[test] + fn older_options_keep_strict_decoding() { + let mut options: WPDDecoderOptions = unsafe { mem::zeroed() }; + + options.libwebp_compat = 1; + for size in [V1_SIZE, V2_SIZE, V3_SIZE] { + options.struct_size = size; + assert!(!options.to_core().libwebp_compat); + } + options.struct_size = mem::size_of::(); + assert!(options.to_core().libwebp_compat); + } } diff --git a/fuzz/budget.rs b/fuzz/budget.rs new file mode 100644 index 0000000..fe1e0c6 --- /dev/null +++ b/fuzz/budget.rs @@ -0,0 +1,13 @@ +pub const MAX_PIXELS: u32 = 1 << 20; + +pub fn fits(data: &[u8]) -> bool { + // Keep malformed headers in coverage; only a successfully read size can + // exceed the harness budget. Raw codecs have no decoder options limit. + wpd::api::info(data).map_or(true, |info| { + wpd::api::Options { + frame_size_limit: MAX_PIXELS, + ..Default::default() + } + .fits(info.width, info.height) + }) +} diff --git a/fuzz/fuzz_targets/container.rs b/fuzz/fuzz_targets/container.rs index ff1825b..85373cf 100644 --- a/fuzz/fuzz_targets/container.rs +++ b/fuzz/fuzz_targets/container.rs @@ -1,4 +1,3 @@ - #![no_main] use libfuzzer_sys::fuzz_target; diff --git a/fuzz/fuzz_targets/e2e.rs b/fuzz/fuzz_targets/e2e.rs index 9832126..f600302 100644 --- a/fuzz/fuzz_targets/e2e.rs +++ b/fuzz/fuzz_targets/e2e.rs @@ -11,6 +11,9 @@ use wpd_capi::decoder::{wpd_decode_into, WPDOutputBuffer}; use wpd_capi::frame::{WPDFrame, WPDOutputPlane}; use wpd_capi::options::WPDDecoderOptions; +#[path = "../budget.rs"] +mod budget; + const FORMATS: [Format; 16] = [ Format::Yuv420p, Format::Yuva420p, @@ -40,6 +43,7 @@ fn decode_options(data: &[u8]) -> (Options, bool) { let flags = byte(data, 1); let subframe = flags & 4 != 0; let mut options = Options { + frame_size_limit: budget::MAX_PIXELS, n_threads: [1, 2, 3, 8][usize::from(flags >> 6)], bypass_filtering: flags & 8 != 0, no_fancy_upsampling: flags & 16 != 0, @@ -123,6 +127,10 @@ fn decode_external(data: &[u8], options: Options) { flip: i32::from(options.flip), reserved: 0, n_threads: options.n_threads, + reserved2: 0, + frame_size_limit: options.frame_size_limit, + reserved3: 0, + libwebp_compat: i32::from(options.libwebp_compat), }; let mut frame = WPDFrame { struct_size: mem::size_of::(), @@ -153,12 +161,14 @@ fn decode_external(data: &[u8], options: Options) { } } -fuzz_target!(|data: &[u8]| { +fn exercise(data: &[u8], libwebp_compat: bool) { let Some(&first) = data.first() else { return; }; let format = FORMATS[first as usize % FORMATS.len()]; - let (options, subframe) = decode_options(data); + let (mut options, subframe) = decode_options(data); + + options.libwebp_compat = libwebp_compat; let mut serial = Decoder::new(); configure( @@ -240,4 +250,12 @@ fuzz_target!(|data: &[u8]| { break; } } +} + +fuzz_target!(|data: &[u8]| { + if !budget::fits(data) { + return; + } + exercise(data, false); + exercise(data, true); }); diff --git a/fuzz/fuzz_targets/vp8.rs b/fuzz/fuzz_targets/vp8.rs index dbe3ec7..f2794f6 100644 --- a/fuzz/fuzz_targets/vp8.rs +++ b/fuzz/fuzz_targets/vp8.rs @@ -1,10 +1,15 @@ - #![no_main] use libfuzzer_sys::fuzz_target; use wpd::vp8::Decoder; +#[path = "../budget.rs"] +mod budget; + fuzz_target!(|data: &[u8]| { + if !budget::fits(data) { + return; + } let mut decoder = Decoder::new(); let _ = decoder.decode_frame(data); diff --git a/fuzz/fuzz_targets/vp8l.rs b/fuzz/fuzz_targets/vp8l.rs index 48367bd..fc91499 100644 --- a/fuzz/fuzz_targets/vp8l.rs +++ b/fuzz/fuzz_targets/vp8l.rs @@ -1,14 +1,19 @@ - #![no_main] use libfuzzer_sys::fuzz_target; use wpd::vp8l::{AlphaDst, Decoder, Target}; +#[path = "../budget.rs"] +mod budget; + fuzz_target!(|data: &[u8]| { if data.len() < 2 { return; } let (head, payload) = data.split_at(2); + if !budget::fits(payload) { + return; + } let mut decoder = Decoder::new(); let alpha_chunk = head[0] & 1 != 0; diff --git a/include/wpd.h b/include/wpd.h index 7f60934..ff94859 100644 --- a/include/wpd.h +++ b/include/wpd.h @@ -260,10 +260,38 @@ typedef struct WPDDecoderOptions { * no such field, gets no limit. */ unsigned frame_size_limit; + /** Must be zero. Takes the tail padding after frame_size_limit. */ + int reserved3; + /** + * Match libwebp's still-container acceptance and portable lossy decoding + * on damaged input. 0 keeps strict decoding; 1 enables compatibility. + * Set before opening input. Animation containers retain their validation. + * Older callers retain strict decoding. + * Compatible still streams retain bytes after RIFF's declared end and + * produce their picture only after wpd_decoder_end_of_stream(). + */ + int libwebp_compat; } WPDDecoderOptions; #define WPD_DECODER_OPTIONS_INIT \ - {sizeof(WPDDecoderOptions), 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0} + {sizeof(WPDDecoderOptions), \ + 0, \ + 0, \ + 0, \ + 0, \ + 0, \ + 0, \ + 0, \ + 0, \ + 0, \ + 0, \ + 0, \ + 0, \ + 0, \ + 0, \ + 0, \ + 0, \ + 0} /** * Set processing options. Cropping precedes scaling. diff --git a/meson.build b/meson.build index f3eb390..66af0e2 100644 --- a/meson.build +++ b/meson.build @@ -517,7 +517,7 @@ custom_target( message('imagewebpdec: enabled (build explicitly with target imagewebpdec)') wuffs_dep = dependency( - '', + 'wuffs', fallback: ['wuffs', 'wuffs_dep'], required: get_option('wuffs'), ) diff --git a/scripts/fuzz-smoke.sh b/scripts/fuzz-smoke.sh new file mode 100755 index 0000000..de32d09 --- /dev/null +++ b/scripts/fuzz-smoke.sh @@ -0,0 +1,76 @@ +#!/bin/bash -eu + +SECONDS_PER_TARGET="${1:-15}" +CORPUS="${2:-wpd-test-data}" + +case "$SECONDS_PER_TARGET" in + ''|*[!0-9]*|0*) echo "fuzz-smoke.sh: seconds must be positive" >&2; exit 1 ;; +esac + +# Keep generated seeds separate from saved developer corpora and artifacts. +mkdir -p fuzz/corpus/ci/{container,e2e,vp8,vp8l} fuzz/artifacts +python3 - "$CORPUS" <<'PY' +from pathlib import Path +import sys + +corpus = Path(sys.argv[1]) +out = Path("fuzz/corpus/ci") +files = sorted(corpus.glob("*.webp")) +if not files: + sys.exit(f"no WebP files found in {corpus}") + +def chunks(data): + offset = 0 + while offset + 8 <= len(data): + tag = data[offset:offset + 4] + size = int.from_bytes(data[offset + 4:offset + 8], "little") + end = offset + 8 + size + if end > len(data): + break + payload = data[offset + 8:end] + if tag == b"ANMF": + yield from chunks(payload[16:]) + else: + yield tag, payload + offset = end + (size & 1) + +for path in files: + data = path.read_bytes() + # Large seeds make a smoke run spend its budget replaying images. + if len(data) > 65536: + continue + for target in ("container", "e2e"): + (out / target / path.stem).write_bytes(data) + for index, (tag, payload) in enumerate(chunks(data[12:])): + name = f"{path.stem}-{index}" + if tag == b"VP8 ": + (out / "vp8" / name).write_bytes(payload) + elif tag == b"VP8L": + # The target uses two leading bytes to select ARGB and threading. + for threads in (0, 1): + (out / "vp8l" / f"{name}-{threads}").write_bytes( + bytes((0, threads)) + payload) + +# Exercise the damaged transform regressions through raw and driver targets. +for path in sorted(Path("tests/data/vp8-compat").glob("*.vp8")): + payload = path.read_bytes() + name = f"compat-{path.stem}" + (out / "vp8" / name).write_bytes(payload) + chunk = b"VP8 " + len(payload).to_bytes(4, "little") + payload + chunk += bytes(len(payload) & 1) + data = b"RIFF" + (len(chunk) + 4).to_bytes(4, "little") + b"WEBP" + chunk + for target in ("container", "e2e"): + (out / target / name).write_bytes(data) + +for target in ("container", "e2e", "vp8", "vp8l"): + if not any((out / target).iterdir()): + sys.exit(f"no seeds prepared for {target}") +PY + +# With no target argument cargo-fuzz builds every declared target. +cargo +nightly fuzz build -O --debug-assertions +for target in container vp8 vp8l e2e; do + cargo +nightly fuzz run -O --debug-assertions "$target" \ + "fuzz/corpus/ci/$target" -- -max_total_time="$SECONDS_PER_TARGET" \ + -max_len=65536 -timeout=5 -rss_limit_mb=1024 -seed=1 +done diff --git a/scripts/webpcompare.py b/scripts/webpcompare.py new file mode 100755 index 0000000..39c0dd5 --- /dev/null +++ b/scripts/webpcompare.py @@ -0,0 +1,190 @@ +#!/usr/bin/env python3 +"""Compare stills and composited animations with libwebp 1.6.0. + +Emit one JSON record per input and a summary. RGB differences under zero alpha +are counted separately. Input files, including extensionless fuzz seeds, are +never changed. Decoded files live in a temporary directory under wpd-test-data. +""" + +import argparse +import hashlib +import json +import os +from pathlib import Path +import re +import subprocess +import tempfile + + +def run(command, timeout): + try: + p = subprocess.run(command, stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE, timeout=timeout) + return p.returncode, p.stderr[-4096:].decode("utf-8", "replace") + except subprocess.TimeoutExpired: + return "timeout", "wall-clock limit exceeded" + + +def check_version(tool, pattern): + p = subprocess.run([tool, "-version"], capture_output=True, text=True, + check=True, timeout=10) + version = (p.stdout + p.stderr).strip() + if not re.search(pattern, version): + raise ValueError(f"{tool}: expected libwebp 1.6.0, got {version!r}") + return version + + +def animated(data): + return (len(data) >= 30 and data[:4] == b"RIFF" and + data[8:16] == b"WEBPVP8X" and data[20] & 2 != 0) + + +def pams(path): + """Read concatenated PAM frames and normalize RGB to straight RGBA.""" + frames = [] + with path.open("rb") as source: + while True: + magic = source.readline(4) + if not magic: + return frames + if magic != b"P7\n": + raise ValueError("invalid PAM magic") + fields = {} + for _ in range(32): + line = source.readline(256) + if line == b"ENDHDR\n": + break + pair = line.split() + if len(pair) == 2: + fields[pair[0]] = pair[1] + else: + raise ValueError("invalid PAM header") + width, height, depth = [int(fields[key]) for key in + (b"WIDTH", b"HEIGHT", b"DEPTH")] + if width <= 0 or height <= 0 or depth not in (3, 4): + raise ValueError("invalid PAM dimensions or depth") + size = width * height * depth + if size > 2 ** 30 or fields.get(b"MAXVAL") != b"255": + raise ValueError("PAM exceeds comparison limit") + pixels = source.read(size) + if len(pixels) != size: + raise ValueError("short PAM frame") + if depth == 3: + rgba = bytearray(width * height * 4) + for channel in range(3): + rgba[channel::4] = pixels[channel::3] + rgba[3::4] = b"\xff" * (width * height) + pixels = bytes(rgba) + frames.append((width, height, pixels)) + + +def digest(frames): + h = hashlib.sha256() + for width, height, pixels in frames: + h.update(width.to_bytes(4, "little")) + h.update(height.to_bytes(4, "little")) + h.update(pixels) + return h.hexdigest() + + +def difference(wpd, reference): + if [(w, h) for w, h, _ in wpd] != [(w, h) for w, h, _ in reference]: + return "geometry", 0, 0 + visible = transparent = 0 + for (_, _, a), (_, _, b) in zip(wpd, reference): + if a == b: + continue + for i in range(0, len(a), 4): + if a[i:i + 4] == b[i:i + 4]: + continue + if a[i + 3] == b[i + 3] == 0: + transparent += 1 + else: + visible += 1 + return ("visible" if visible else "transparent" if transparent else "identical", + visible, transparent) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("directory", type=Path) + parser.add_argument("--wpd", default="./build/wpd") + parser.add_argument("--dwebp", default="dwebp") + parser.add_argument("--anim-dump", default="anim_dump") + parser.add_argument("--libwebp-compat", action="store_true") + parser.add_argument("--noasm", action="store_true", + help="use portable C for dwebp; anim_dump remains native") + parser.add_argument("--timeout", type=float, default=60) + parser.add_argument("--work-dir", type=Path, default=Path("wpd-test-data")) + args = parser.parse_args() + if not args.directory.is_dir() or args.timeout <= 0: + parser.error("a directory and a positive timeout are required") + args.wpd = os.path.abspath(args.wpd) + args.work_dir.mkdir(parents=True, exist_ok=True) + versions = { + "dwebp": check_version(args.dwebp, r"^1\.6\.0$"), + "anim_dump": check_version(args.anim_dump, r"Decoder version: 1\.6\.0\b"), + } + counts = {key: 0 for key in ("identical", "transparent", "visible", "geometry", + "wpd_only", "libwebp_only", "neither", "abnormal")} + files = sorted(p for p in args.directory.rglob("*") if p.is_file() and + ".git" not in p.relative_to(args.directory).parts) + for path in files: + data = path.read_bytes() + record = {"path": str(path), "sha256": hashlib.sha256(data).hexdigest(), + "animation": animated(data)} + with tempfile.TemporaryDirectory(prefix="webpcompare-", dir=args.work_dir) as td: + work = Path(td) + wpath = work / "wpd.pam" + wcmd = [args.wpd, "--fmt", "rgba", "--muxer", "pam", "--threads", "1"] + if args.libwebp_compat: + wcmd.append("--libwebp-compat") + wstatus, werror = run(wcmd + [str(path), str(wpath)], args.timeout) + if record["animation"]: + lcmd = [args.anim_dump, "-pam", "-folder", str(work), "-prefix", "ref"] + else: + lcmd = [args.dwebp, "-quiet", "-pam", "-o", str(work / "ref0.pam")] + if args.noasm and not record["animation"]: + lcmd.append("-noasm") + lstatus, lerror = run(lcmd + [str(path)], args.timeout) + record.update(wpd_status=wstatus, libwebp_status=lstatus) + if (not isinstance(wstatus, int) or not isinstance(lstatus, int) or + wstatus < 0 or lstatus < 0): + outcome = "abnormal" + elif wstatus == lstatus == 0: + try: + a = pams(wpath) + references = sorted(work.glob("ref*.pam"), key=lambda p: + int(re.search(r"(\d+)\.pam$", p.name)[1])) + b = [frame for p in references for frame in pams(p)] + if not a or not b: + raise ValueError("success without decoded frames") + outcome, visible, transparent = difference(a, b) + record.update(wpd_pixel_sha256=digest(a), + libwebp_pixel_sha256=digest(b), + visible_pixels=visible, transparent_pixels=transparent, + frames=len(a)) + except (ValueError, KeyError) as error: + outcome = "abnormal" + record["output_error"] = str(error) + else: + outcome = ("wpd_only" if wstatus == 0 else + "libwebp_only" if lstatus == 0 else "neither") + if wstatus != 0: + record["wpd_error"] = werror.strip() + if lstatus != 0: + record["libwebp_error"] = lerror.strip() + record["outcome"] = outcome + counts[outcome] += 1 + print(json.dumps(record), flush=True) + print(json.dumps({"summary": counts, "files": len(files), "versions": versions, + "libwebp_compat": args.libwebp_compat, "noasm": args.noasm})) + return int(not files or any(counts[key] for key in + ("visible", "geometry", "wpd_only", "libwebp_only", "abnormal"))) + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except (OSError, ValueError, subprocess.SubprocessError) as error: + raise SystemExit(str(error)) diff --git a/src/container.rs b/src/container.rs index eb31075..707477d 100644 --- a/src/container.rs +++ b/src/container.rs @@ -120,6 +120,7 @@ pub struct Info { #[derive(Default)] pub struct Scan { + libwebp_compat: bool, pos: usize, riff_end: u64, info: Info, @@ -220,6 +221,18 @@ impl Scan { }; } + pub(crate) fn set_libwebp_compat(&mut self, enabled: bool) { + self.libwebp_compat = enabled; + } + + fn compat_still(&self) -> bool { + self.libwebp_compat && self.vp8x_flags & VP8X_FLAG_ANIM == 0 + } + + pub(crate) fn is_animation_container(&self) -> bool { + self.vp8x_flags & VP8X_FLAG_ANIM != 0 + } + pub fn info(&self) -> &Info { &self.info } @@ -244,6 +257,13 @@ impl Scan { } } else { self.info.coding = Coding::Lossy; + // libwebp's header check compares the first partition length + // with the declared chunk size without subtracting the header. + let size = if self.compat_still() { + size.saturating_add(9) + } else { + size + }; if let Some((width, height)) = bitstream_size(tag, p, size) { self.info.width = width; self.info.height = height; @@ -529,6 +549,9 @@ impl Scan { return self.raw_headers(buf, partial); } self.riff_end = u64::from(rl32(buf, 4)) + 8; + if self.libwebp_compat && !(20..=0xffff_fffe).contains(&self.riff_end) { + return Err(Error::InvalidData); + } self.pos = 12; } @@ -544,16 +567,29 @@ impl Scan { let tag = rl32(buf, at); let size = rl32(buf, at + 4); + if self.pos == 12 && !matches!(tag, TAG_VP8X | TAG_VP8 | TAG_VP8L) { + log::error("RIFF must start with VP8, VP8L or VP8X"); + return Err(Error::InvalidData); + } if size == u32::MAX { - self.info.truncated = true; + self.info.truncated |= !self.compat_still() || self.info.images == 0; break; } - let padded = size as usize + (size & 1) as usize; + let mut padded = size as usize + (size & 1) as usize; + + if self.compat_still() + && !self.vp8x + && matches!(tag, TAG_VP8 | TAG_VP8L) + && self.info.end - (self.pos + 8) == size as usize + { + // The simple still parser checks the payload, not its pad. + padded = size as usize; + } if self.info.end - (self.pos + 8) < padded { let avail = self.info.end - (self.pos + 8); - self.info.truncated = true; + self.info.truncated |= !self.compat_still() || self.info.images == 0; if self.collect_frames && tag == TAG_ANMF { self.anmf(window(buf, at + 8, avail), false)?; } @@ -570,6 +606,15 @@ impl Scan { match tag { TAG_VP8X => { + // The still decoder treats later VP8X chunks as optional + // chunks; the demuxer validates them instead. + if self.compat_still() && self.vp8x { + self.pos += 8 + padded; + continue; + } + if self.compat_still() && self.info.images != 0 { + break; + } if self.vp8x || size != VP8X_CHUNK_SIZE { log::error("invalid VP8X chunk"); return Err(Error::InvalidData); @@ -578,7 +623,9 @@ impl Scan { let flags = byte(buf, at + 8); - if flags & !VP8X_FLAGS_VALID != 0 { + if (!self.libwebp_compat || flags & VP8X_FLAG_ANIM != 0) + && flags & !VP8X_FLAGS_VALID != 0 + { log::error_args(format_args!( "VP8X sets reserved flag bits (0x{flags:02x})" )); @@ -596,6 +643,10 @@ impl Scan { crate::error::check_image_size(self.info.width, self.info.height)?; } TAG_ALPH => { + if self.compat_still() && self.info.images != 0 { + self.pos += 8 + padded; + continue; + } self.still_chunk_allowed()?; if self.still_alpha_allowed() { self.info.has_alpha = true; @@ -603,6 +654,10 @@ impl Scan { } } TAG_ANIM => { + if self.compat_still() { + self.pos += 8 + padded; + continue; + } if size < ANIM_CHUNK_SIZE { log::error("ANIM chunk is too short"); return Err(Error::InvalidData); @@ -620,6 +675,10 @@ impl Scan { } } TAG_ANMF => { + if self.compat_still() { + self.pos += 8 + padded; + continue; + } if !self.anim_chunk { log::error("ANMF chunk before the ANIM header"); return Err(Error::InvalidData); @@ -641,14 +700,26 @@ impl Scan { self.still_chunk_allowed()?; let first = self.info.images == 0; + let have = if self.compat_still() { + base + buf.len() - (self.pos + 8) + } else { + size as usize + }; + + if first + && self.compat_still() + && have < if tag == TAG_VP8L { 5 } else { 10 } + { + // A shortened chunk can be complete before its actual + // codec header has arrived. Revisit it on append. + self.info.truncated = true; + partial_still = partial; + break; + } self.info.images = self.info.images.saturating_add(1); if first { - self.still_size( - tag, - window(buf, at + 8, size as usize), - size as usize, - )?; + self.still_size(tag, window(buf, at + 8, have), size as usize)?; } } _ => { @@ -751,6 +822,25 @@ mod tests { assert_eq!(get_info(b"not a webp file at all"), Err(Error::NotWebp)); } + #[test] + fn a_damaged_vp8x_tag_cannot_turn_an_extended_file_into_a_simple_one() { + let mut payload = chunk(b"VP9X", &[0x10, 0, 0, 0, 0, 0, 0, 0, 0, 0]); + + payload.extend(chunk(b"VP8L", &vp8l_header(1, 1, true))); + + let file = riff(&payload); + + assert_eq!(get_info(&file), Err(Error::InvalidData)); + + let mut scan = Scan::new(); + + assert_eq!( + scan.headers(&file[..19], 0, true, true), + Err(Error::Truncated) + ); + assert_eq!(scan.headers(&file, 0, true, true), Err(Error::InvalidData)); + } + #[test] fn oversized_vp8x_dimensions_are_refused_before_allocating() { let payload = chunk(b"VP8X", &[2, 0, 0, 0, 0, 64, 0, 0, 0, 0]); diff --git a/src/driver/lossy.rs b/src/driver/lossy.rs index 704df4b..bbe769e 100644 --- a/src/driver/lossy.rs +++ b/src/driver/lossy.rs @@ -62,6 +62,7 @@ struct Alpha<'p, 'i> { compression: i32, filter: i32, threads: usize, + libwebp_compat: bool, } fn decode_alpha(a: Alpha<'_, '_>) -> Result<()> { @@ -78,6 +79,7 @@ fn decode_alpha(a: Alpha<'_, '_>) -> Result<()> { compression, filter, threads, + libwebp_compat, } = a; let extent = width .checked_mul(height.max(0) as usize) @@ -98,6 +100,7 @@ fn decode_alpha(a: Alpha<'_, '_>) -> Result<()> { } else if compression == ALPHA_COMPRESSION_VP8L { vp8l.set_canvas(width as i32, height); vp8l.threads = threads; + vp8l.libwebp_compat = libwebp_compat; let rest = match filter { ALPHA_FILTER_HORIZONTAL => Some(fdsp.horizontal_unfilter), @@ -199,6 +202,7 @@ impl FrameSlot { compression: *alpha_compression, filter: *alpha_filter, threads: env.threads, + libwebp_compat: env.libwebp_compat, }, vp8.first_mut(), ) @@ -219,10 +223,12 @@ impl FrameSlot { ) -> Result<()> { { let bypass = env.bypass_filtering; + let compat = env.libwebp_compat; let chunk = env.input.chunk(offset, size); let vp8 = self.vp8_decoder()?; vp8.bypass_filtering = bypass; + vp8.libwebp_compat = compat; if vp8.frame_init(chunk, size, size)? == Status::NeedMore { return Err(Error::InvalidData); } @@ -310,6 +316,7 @@ impl<'a> Decoder<'a> { pub(crate) fn frame_settings(&self) -> super::slot::FrameSettings { super::slot::FrameSettings { bypass_filtering: self.filter_bypass(), + libwebp_compat: self.options.libwebp_compat, no_fancy_upsampling: self.options.no_fancy_upsampling, to_argb: self.frame_to_argb(), premultiply: self.frame_premultiply(), @@ -355,10 +362,12 @@ impl<'a> Decoder<'a> { ) -> Result { if !self.vp8_active { let bypass = self.filter_bypass(); + let compat = self.options.libwebp_compat; let Self { frame, input, .. } = self; let vp8 = frame.vp8_decoder()?; vp8.bypass_filtering = bypass; + vp8.libwebp_compat = compat; match vp8.frame_init(input.chunk(offset, avail), avail, size)? { Status::NeedMore => return Ok(false), diff --git a/src/driver/mod.rs b/src/driver/mod.rs index e1a9104..1f2ab6c 100644 --- a/src/driver/mod.rs +++ b/src/driver/mod.rs @@ -503,6 +503,7 @@ impl<'a> Decoder<'a> { fn rescan_headers(&mut self) -> Result<(), Error> { let base = self.input.discarded(); + self.scan.set_libwebp_compat(self.options.libwebp_compat); let walked = self .scan .headers(self.input.bytes(), base, self.streaming, true); @@ -621,6 +622,11 @@ impl<'a> Decoder<'a> { /// buffered, and a caller appending forever holds at most the file the /// header describes. A header split across appends is read from both. fn riff_room(&self, data: &[u8]) -> usize { + // The still codec in libwebp can consume bytes past RIFF's declared + // end. This mode retains them, subject to Input's byte limit. + if self.options.libwebp_compat && !self.scan.is_animation_container() { + return usize::MAX; + } let size = self.input.size(); let end = self.scan.riff_end().or_else(|| { if size >= 12 { @@ -716,6 +722,12 @@ impl Decoder<'_> { if bad_crop || bad_scale || options.n_threads < 0 { return Err(self.fail("invalid decoder options", Error::InvalidArgument)); } + if self.opened && options.libwebp_compat != self.options.libwebp_compat { + return Err(self.fail( + "compatibility must be set before opening input", + Error::InvalidArgument, + )); + } if self.anim_mode == ANIM_SUBFRAME && options.transforms() { return Err(self.fail( "cropping, scaling and flipping are defined against the canvas, \ @@ -1044,6 +1056,14 @@ impl Decoder<'_> { _ => {} } } + if decoder.options.libwebp_compat && decoder.still_done && !decoder.animation { + return Ok(false); + } + // A compatible still can need bytes outside its declared image or + // RIFF extent. Its final partition size is only known at EOF. + if decoder.options.libwebp_compat && !decoder.animation && !decoder.eos { + return Ok(false); + } if decoder.scanned().raw != Raw::No { return if decoder.still_done { Ok(false) @@ -1067,7 +1087,12 @@ impl Decoder<'_> { let size = size as usize; let padded_size = size + (size & 1); - if decoder.end - payload_pos < padded_size { + let missing_still_pad = decoder.options.libwebp_compat + && !decoder.animation + && decoder.end - payload_pos == size + && matches!(chunk_type, TAG_VP8 | TAG_VP8L); + + if decoder.end - payload_pos < padded_size && !missing_still_pad { if !decoder.eos { let avail = decoder.end - payload_pos; @@ -1106,7 +1131,7 @@ impl Decoder<'_> { Error::InvalidData, )); } - if decoder.alpha_pending { + if decoder.alpha_pending && !decoder.options.libwebp_compat { return Err(("duplicate ALPHA chunk", Error::InvalidData)); } decoder.alpha_pending = true; @@ -1123,6 +1148,13 @@ impl Decoder<'_> { if decoder.animation || decoder.still_done { continue; } + // libwebp uses all remaining bytes for the still codec, + // even when the image chunk declares a shorter payload. + let size = if decoder.options.libwebp_compat { + decoder.input.size() - payload_pos + } else { + size + }; let ret = if decoder.vp8_active { decoder.vp8_lossy_step(payload_pos, size, size).and_then( |done| done.then_some(()).ok_or(Error::InvalidData), @@ -1140,6 +1172,11 @@ impl Decoder<'_> { if decoder.animation || decoder.still_done { continue; } + let size = if decoder.options.libwebp_compat { + decoder.input.size() - payload_pos + } else { + size + }; if decoder.frame.vp8l.still_active() { decoder .lossless_step(payload_pos, size, size, true) @@ -1157,6 +1194,9 @@ impl Decoder<'_> { return decoder.emit_still_lossless(out); } TAG_ANMF => { + if decoder.options.libwebp_compat && !decoder.animation { + continue; + } if !decoder.animation || decoder.canvas_width == 0 || decoder.canvas_height == 0 @@ -1375,6 +1415,183 @@ mod tests { out } + fn compat() -> Decoder<'static> { + let mut decoder = Decoder::new(); + + decoder + .set_core_options(Options { + libwebp_compat: true, + ..Options::default() + }) + .unwrap(); + decoder + } + + #[test] + fn still_compatibility_ignores_a_broken_tail_and_uses_available_image_bytes() { + let mut broken_tail = riff_lossless(); + + broken_tail.extend_from_slice(b"JUNK\xff\xff\xff\xff"); + let len = broken_tail.len() as u32 - 8; + + broken_tail[4..8].copy_from_slice(&len.to_le_bytes()); + + let mut short_chunk = riff_lossless(); + + short_chunk[16..20].copy_from_slice(&6u32.to_le_bytes()); + + let mut strict = Decoder::new(); + + assert_eq!(strict.open(&broken_tail), Err(Error::Truncated)); + strict.open(&short_chunk).unwrap(); + assert!(strict.next_picture(&mut Handout::default()).is_err()); + + let mut short_riff = riff_lossless(); + + short_riff[4..8].copy_from_slice(&18u32.to_le_bytes()); + short_riff[16..20].copy_from_slice(&6u32.to_le_bytes()); + + for data in [broken_tail, short_chunk, short_riff] { + for mode in 0..3 { + let mut decoder = compat(); + let mut frames = 0; + + if mode == 0 { + decoder.open(&data).unwrap(); + } else { + decoder.open_stream().unwrap(); + for i in 1..=data.len() { + if mode == 1 { + decoder.append(&data[i - 1..i]).unwrap(); + } else { + decoder.update(&data[..i]).unwrap(); + } + frames += usize::from( + decoder.next_picture(&mut Handout::default()).unwrap(), + ); + } + decoder.end_of_stream().unwrap(); + } + frames += + usize::from(decoder.next_picture(&mut Handout::default()).unwrap()); + assert_eq!(frames, 1, "mode {mode}"); + assert!(!decoder.next_picture(&mut Handout::default()).unwrap()); + assert_eq!((decoder.canvas_width, decoder.canvas_height), (2, 2)); + } + } + } + + #[test] + fn still_compatibility_keeps_the_first_vp8x_and_the_riff_size_check() { + let mut payload = chunk(b"VP8X", &[0, 0, 0, 0, 1, 0, 0, 1, 0, 0]); + + payload.extend(chunk(b"VP8X", &[0xff; 10])); + payload.extend(chunk(b"VP8L", RAW_LOSSLESS)); + + let mut data = b"RIFF".to_vec(); + + data.extend_from_slice(&(payload.len() as u32 + 4).to_le_bytes()); + data.extend_from_slice(b"WEBP"); + data.extend(payload); + assert_eq!(Decoder::new().open(&data), Err(Error::InvalidData)); + + let mut decoder = compat(); + + decoder.open(&data).unwrap(); + assert!(decoder.next_picture(&mut Handout::default()).unwrap()); + assert!(!decoder.next_picture(&mut Handout::default()).unwrap()); + let len = data.len() as u32; + + data[4..8].copy_from_slice(&len.to_le_bytes()); + assert_eq!(decoder.open(&data), Err(Error::Truncated)); + } + + #[test] + fn container_compatibility_cannot_change_after_opening_input() { + let mut decoder = Decoder::new(); + + decoder.open(&riff_lossless()).unwrap(); + assert_eq!( + decoder.set_core_options(Options { + libwebp_compat: true, + ..Options::default() + }), + Err(Error::InvalidArgument) + ); + } + + #[test] + fn still_compatibility_waits_for_the_actual_header_of_an_empty_chunk() { + for size in [0u32, 1, 4] { + let mut data = riff_lossless(); + + data[16..20].copy_from_slice(&size.to_le_bytes()); + + for update in [false, true] { + let mut decoder = compat(); + + decoder.open_stream().unwrap(); + for i in 1..=data.len() { + if update { + decoder.update(&data[..i]).unwrap(); + } else { + decoder.append(&data[i - 1..i]).unwrap(); + } + assert!(!decoder.next_picture(&mut Handout::default()).unwrap()); + } + decoder.end_of_stream().unwrap(); + assert!(decoder.next_picture(&mut Handout::default()).unwrap()); + assert!(!decoder.next_picture(&mut Handout::default()).unwrap()); + } + } + } + + #[test] + fn still_compatibility_accepts_a_simple_images_missing_final_pad() { + let mut data = riff_lossless(); + + data.push(0); + let len = data.len() as u32 - 8; + + data[4..8].copy_from_slice(&len.to_le_bytes()); + data[16..20].copy_from_slice(&13u32.to_le_bytes()); + assert_eq!(Decoder::new().open(&data), Err(Error::Truncated)); + let mut decoder = compat(); + + decoder.open(&data).unwrap(); + assert!(decoder.next_picture(&mut Handout::default()).unwrap()); + decoder.open_stream().unwrap(); + for byte in &data { + decoder.append(std::slice::from_ref(byte)).unwrap(); + } + decoder.end_of_stream().unwrap(); + assert!(decoder.next_picture(&mut Handout::default()).unwrap()); + } + + #[test] + fn still_compatibility_skips_optional_anmf_and_keeps_animation_validation() { + let mut data = riff_lossless(); + let mut payload = chunk(b"VP8X", &[0, 0, 0, 0, 1, 0, 0, 1, 0, 0]); + + payload.extend(chunk(b"ANMF", &[0; 16])); + payload.extend_from_slice(&data[12..]); + data.truncate(12); + data[4..8].copy_from_slice(&(payload.len() as u32 + 4).to_le_bytes()); + data.extend(payload); + + let mut decoder = compat(); + + decoder.open(&data).unwrap(); + assert!(decoder.next_picture(&mut Handout::default()).unwrap()); + assert!(!decoder.next_picture(&mut Handout::default()).unwrap()); + for flags in [3, 0x42, 0x82] { + let mut data = animation(&chunk(b"VP8L", RAW_LOSSLESS), 2, 2); + + data[20] = flags; + assert_eq!(compat().open(&data), Err(Error::InvalidData)); + } + } + #[test] fn detached_update_buffers_suspend_decoding_and_resume_at_the_same_position() { let mut decoder = Decoder::new(); diff --git a/src/driver/slot.rs b/src/driver/slot.rs index 0b64d7e..880e26e 100644 --- a/src/driver/slot.rs +++ b/src/driver/slot.rs @@ -33,6 +33,7 @@ impl std::ops::Deref for FrameEnv<'_, '_> { #[derive(Clone, Copy, Default, PartialEq, Eq)] pub(crate) struct FrameSettings { pub(crate) bypass_filtering: bool, + pub(crate) libwebp_compat: bool, pub(crate) no_fancy_upsampling: bool, /// The output format alone decides the frame must become ARGB, whatever /// the frames before it did, so the conversion can happen off the walk. diff --git a/src/dsp/vp8.rs b/src/dsp/vp8.rs index ff58d4d..122da87 100644 --- a/src/dsp/vp8.rs +++ b/src/dsp/vp8.rs @@ -161,20 +161,32 @@ pub fn loop_filter_simple(buf: &mut [u8], stride: usize, flim: } } -pub fn luma_dc_wht(block: &mut [[i16; 16]; 16], dc: &mut [i16; 16]) { +fn transform_intermediate(value: i32) -> i32 { + if LIBWEBP { + value + } else { + i32::from(value as i16) + } +} + +fn luma_dc_wht_tmpl( + block: &mut [[i16; 16]; 16], + dc: &mut [i16; 16], +) { + let mut tmp = [0i32; 16]; for i in 0..4 { let d = |k: usize| i32::from(dc[k * 4 + i]); let (t0, t1) = (d(0) + d(3), d(1) + d(2)); let (t2, t3) = (d(1) - d(2), d(0) - d(3)); - dc[i] = (t0 + t1) as i16; - dc[4 + i] = (t3 + t2) as i16; - dc[8 + i] = (t0 - t1) as i16; - dc[12 + i] = (t3 - t2) as i16; + tmp[i] = transform_intermediate::(t0 + t1); + tmp[4 + i] = transform_intermediate::(t3 + t2); + tmp[8 + i] = transform_intermediate::(t0 - t1); + tmp[12 + i] = transform_intermediate::(t3 - t2); } for i in 0..4 { - let d = |k: usize| i32::from(dc[i * 4 + k]); + let d = |k: usize| tmp[i * 4 + k]; let (t0, t1) = (d(0) + d(3) + 3, d(1) + d(2)); let (t2, t3) = (d(1) - d(2), d(0) - d(3) + 3); @@ -187,6 +199,10 @@ pub fn luma_dc_wht(block: &mut [[i16; 16]; 16], dc: &mut [i16; 16]) { } } +pub fn luma_dc_wht(block: &mut [[i16; 16]; 16], dc: &mut [i16; 16]) { + luma_dc_wht_tmpl::(block, dc); +} + pub fn luma_dc_wht_dc(block: &mut [[i16; 16]; 16], dc: &mut [i16; 16]) { let val = ((i32::from(dc[0]) + 3) >> 3) as i16; @@ -197,15 +213,19 @@ pub fn luma_dc_wht_dc(block: &mut [[i16; 16]; 16], dc: &mut [i16; 16]) { } fn mul_20091(a: i32) -> i32 { - ((a * 20091) >> 16) + a + ((a.wrapping_mul(20091)) >> 16) + a } fn mul_35468(a: i32) -> i32 { - (a * 35468) >> 16 + (a.wrapping_mul(35468)) >> 16 } -pub fn idct_add(dst: &mut [u8], stride: usize, block: &mut [i16; 16]) { - let mut tmp = [0i16; 16]; +fn idct_add_tmpl( + dst: &mut [u8], + stride: usize, + block: &mut [i16; 16], +) { + let mut tmp = [0i32; 16]; for i in 0..4 { let b = |k: usize| i32::from(block[k * 4 + i]); @@ -218,14 +238,14 @@ pub fn idct_add(dst: &mut [u8], stride: usize, block: &mut [i16; 16]) { block[k * 4 + i] = 0; } - tmp[i * 4] = (t0 + t3) as i16; - tmp[i * 4 + 1] = (t1 + t2) as i16; - tmp[i * 4 + 2] = (t1 - t2) as i16; - tmp[i * 4 + 3] = (t0 - t3) as i16; + tmp[i * 4] = transform_intermediate::(t0 + t3); + tmp[i * 4 + 1] = transform_intermediate::(t1 + t2); + tmp[i * 4 + 2] = transform_intermediate::(t1 - t2); + tmp[i * 4 + 3] = transform_intermediate::(t0 - t3); } for i in 0..4 { - let t = |k: usize| i32::from(tmp[k * 4 + i]); + let t = |k: usize| tmp[k * 4 + i]; let t0 = t(0) + t(2); let t1 = t(0) - t(2); let t2 = mul_35468(t(1)) - mul_20091(t(3)); @@ -239,6 +259,10 @@ pub fn idct_add(dst: &mut [u8], stride: usize, block: &mut [i16; 16]) { } } +pub fn idct_add(dst: &mut [u8], stride: usize, block: &mut [i16; 16]) { + idct_add_tmpl::(dst, stride, block); +} + // LLVM 19 recognizes a saturating pack when the upper bound is clipped // first, whereas i16::clamp produces extra vector clamps. #[allow(clippy::manual_clamp)] @@ -331,6 +355,32 @@ mod tests { assert_eq!(block[0], 0); } + #[test] + fn full_width_transforms_handle_extreme_coefficients() { + for value in [i16::MIN, i16::MAX] { + let mut block = [value; 16]; + let mut dst = [128u8; 16]; + + idct_add_tmpl::(&mut dst, 4, &mut block); + assert_eq!(block, [0; 16]); + } + } + + #[test] + fn compatibility_dc_rounding_does_not_narrow_before_shifting() { + let mut dsp = Vp8Dsp::new(); + let mut blocks = [[0i16; 16]; 16]; + let mut dc = [0i16; 16]; + + dsp.libwebp_transforms(); + dc[0] = i16::MAX; + (dsp.luma_dc_wht_dc)(&mut blocks, &mut dc); + for block in &blocks { + assert_eq!(block[0], 4096); + } + assert_eq!(dc, [0; 16]); + } + #[test] fn the_transform_clears_its_coefficients() { let mut block = [7i16; 16]; @@ -354,6 +404,7 @@ pub type LfAllFn = fn(&mut [u8], usize, usize, i32, i32, i32, i32, u32); pub type LfUvAllFn = fn(&mut [u8], usize, &mut [u8], usize, usize, i32, i32, i32, i32, u32); +#[derive(Clone, Copy)] pub struct Vp8Dsp { pub luma_dc_wht: WhtFn, pub luma_dc_wht_dc: WhtFn, @@ -402,6 +453,14 @@ fn idct_add_c(p: &mut [u8], o: usize, s: usize, block: &mut [i16; 16]) { idct_add(&mut p[o..], s, block); } +fn wht_libwebp_c(block: &mut [[i16; 16]; 16], dc: &mut [i16; 16]) { + luma_dc_wht_tmpl::(block, dc); +} + +fn idct_add_libwebp_c(p: &mut [u8], o: usize, s: usize, block: &mut [i16; 16]) { + idct_add_tmpl::(&mut p[o..], s, block); +} + fn idct_dc_add_c(p: &mut [u8], o: usize, s: usize, block: &mut [i16; 16]) { idct_dc_add(&mut p[o..], s, block); } @@ -558,6 +617,19 @@ impl Vp8Dsp { } } + /// libwebp's portable C transforms keep the first pass in 32-bit + /// integers. RFC 6386 describes 16-bit intermediates; damaged + /// coefficients can make the two produce different pixels. Select the + /// same arithmetic with and without assembly. + pub(crate) fn libwebp_transforms(&mut self) { + self.luma_dc_wht = wht_libwebp_c; + self.luma_dc_wht_dc = wht_dc_c; + self.idct_add = idct_add_libwebp_c; + self.idct_dc_add = idct_dc_add_c; + self.idct_dc_add4y = idct_dc_add4y_c; + self.idct_dc_add4uv = idct_dc_add4uv_c; + } + pub fn new() -> Self { #[allow(unused_mut)] let mut table = Self::scalar(); diff --git a/src/options.rs b/src/options.rs index a1e9d35..3607b51 100644 --- a/src/options.rs +++ b/src/options.rs @@ -14,6 +14,12 @@ pub struct Options { /// decoded, so a caller facing untrusted input can bound the memory and /// time a few bytes of header may ask for; dav1d's `frame_size_limit`. pub frame_size_limit: u32, + /// Match libwebp's still-container acceptance and portable lossy decoding + /// on damaged input. Set before opening the input; animations retain + /// their container validation. The default keeps strict decoding. + /// Compatible still streams retain bytes past the declared RIFF end and + /// produce their picture after `end_of_stream`. + pub libwebp_compat: bool, } impl Options { diff --git a/src/vp8/mod.rs b/src/vp8/mod.rs index f255e13..353a345 100644 --- a/src/vp8/mod.rs +++ b/src/vp8/mod.rs @@ -291,6 +291,9 @@ pub struct Decoder { pub width: i32, pub height: i32, pub bypass_filtering: bool, + /// Match the full-width intermediates of libwebp's portable C transforms. + pub libwebp_compat: bool, + strict_dsp: Option<(Vp8Dsp, Vp8Dsp)>, mb_width: usize, mb_height: usize, @@ -1523,6 +1526,19 @@ impl Decoder { .inspect_err(|_| crate::log::error("Frame allocation failed"))?; } + if self.libwebp_compat { + if self.strict_dsp.is_none() { + // CPU detection or its global mask may have changed since + // this decoder was created. Restore its own tables later. + self.strict_dsp = Some((self.recon.dsp, self.coeffs.dsp)); + self.recon.dsp.libwebp_transforms(); + self.coeffs.dsp.libwebp_transforms(); + } + } else if let Some((recon, coeffs)) = self.strict_dsp.take() { + self.recon.dsp = recon; + self.coeffs.dsp = coeffs; + } + self.recon.deblock_filter = self.recon.filter.level != 0 && !self.bypass_filtering; diff --git a/src/vp8l/entropy.rs b/src/vp8l/entropy.rs index 24cc72b..23c6f0a 100644 --- a/src/vp8l/entropy.rs +++ b/src/vp8l/entropy.rs @@ -734,6 +734,7 @@ fn run(args: Args<'_, '_>) -> Result { + pub allow_final_overrun: bool, pub gb: &'a mut BitReader, pub buf: &'a [u8], pub pixels: &'a mut [u8], @@ -745,6 +746,7 @@ pub struct AlphaArgs<'a, 'e> { pub fn decode_alpha_pixels(args: AlphaArgs<'_, '_>) -> Result<()> { let AlphaArgs { + allow_final_overrun, gb, buf, pixels, @@ -815,7 +817,10 @@ pub fn decode_alpha_pixels(args: AlphaArgs<'_, '_>) -> Result<()> { return Err(Error::InvalidData); } } - if gb.is_eos(buf) { + // libwebp's paletted-alpha decoder accepts EOF at the final symbol + // once every pixel is written. The checks inside the loop still refuse + // an overrun before the plane is complete and invalid references. + if gb.is_eos(buf) && !allow_final_overrun { crate::log::error("alpha data runs past the end of the chunk"); return Err(Error::InvalidData); } @@ -864,6 +869,51 @@ mod tests { } } + #[test] + fn compatibility_allows_only_a_final_alpha_symbol_at_eof() { + let mut arena = Vec::new(); + let mut plan = super::super::huffman::Plan::default(); + let lengths = [1u8, 1]; + let mut sorted = [0u16; 2]; + + super::super::huffman::count_lengths(&mut plan, &lengths); + let reader = + super::super::huffman::build(&mut arena, &mut plan, &lengths, &mut sorted) + .unwrap(); + let group = HTreeGroup { + trees: [reader; HUFFMAN_CODES_PER_META_CODE], + ..HTreeGroup::default() + }; + // Sixty-four one-bit literals fit; the sixty-fifth uses the + // zero-padded end of the bit window. A sixty-sixth is still invalid. + let buf = [0u8; 8]; + for (allow, count, expected) in [ + (false, 64, Ok(())), + (false, 65, Err(Error::InvalidData)), + (true, 65, Ok(())), + (true, 66, Err(Error::InvalidData)), + ] { + let mut gb = BitReader::new(&buf); + let mut pixels = vec![0xff; count]; + assert_eq!( + decode_alpha_pixels(AlphaArgs { + allow_final_overrun: allow, + gb: &mut gb, + buf: &buf, + pixels: &mut pixels, + width: count, + groups: &[group], + arena: &arena, + entropy: None, + }), + expected + ); + if expected.is_ok() { + assert!(pixels.iter().all(|&p| p == 0)); + } + } + } + #[test] fn overlapping_copies_match_the_pixel_at_a_time_definition() { check_overlaps(std::array::from_fn(|i| i as u8)); diff --git a/src/vp8l/mod.rs b/src/vp8l/mod.rs index f80ff55..1ccbb79 100644 --- a/src/vp8l/mod.rs +++ b/src/vp8l/mod.rs @@ -289,6 +289,8 @@ pub struct Decoder { /// The threads a decode may use, counting the calling one. pub threads: usize, + /// Allow libwebp's final paletted-alpha symbol to finish at EOF. + pub libwebp_compat: bool, } impl Decoder { @@ -925,7 +927,11 @@ impl Decoder { { let Decoder { - gb, image, indices, .. + gb, + image, + indices, + libwebp_compat, + .. } = self; let (head, tail) = image.split_at_mut(ROLE_ENTROPY); let ent = &tail[0]; @@ -934,6 +940,7 @@ impl Decoder { gb, buf, pixels: &mut indices[..total], + allow_final_overrun: *libwebp_compat, width, groups: &head[ROLE_ARGB].groups, arena: &head[ROLE_ARGB].arena, diff --git a/tests/alpha_compat.rs b/tests/alpha_compat.rs new file mode 100644 index 0000000..1213a2f --- /dev/null +++ b/tests/alpha_compat.rs @@ -0,0 +1,71 @@ +use wpd::api::{Decoder, Options}; +use wpd::image::Format; + +type AlphaCase = (&'static [(usize, u8, u8)], u64); + +fn pixel_hash(decoder: &mut Decoder<'_>) -> u64 { + let picture = decoder.next_frame().unwrap().unwrap(); + picture + .rows_of(0) + .flatten() + .fold(0xcbf29ce484222325u64, |hash, &value| { + (hash ^ u64::from(value)).wrapping_mul(0x100000001b3) + }) +} + +#[test] +fn compatibility_matches_libwebp_at_the_end_of_a_paletted_alpha_plane() { + if cfg!(miri) { + return; + } + // Public project testdata, mutated only in its compressed alpha stream. + // Native libwebp 1.6.0 accepts the final symbol at EOF. The unchanged + // colour stream and container isolate alpha decoding from other policies. + let path = std::path::Path::new(env!("CARGO_MANIFEST_DIR")) + .join("wpd-test-data/odd_a_lossy.webp"); + let seed = std::fs::read(path).unwrap(); + let cases: &[AlphaCase] = &[ + (&[(231, 202, 200), (351, 128, 125)], 0x87ed788ad3cb8e12), + ( + &[(384, 53, 49), (385, 57, 56), (669, 191, 36)], + 0x01cdcded58bfb418, + ), + ]; + for &(patch, expected) in cases { + let mut bytes = seed.clone(); + for &(offset, old, new) in patch { + assert_eq!(bytes[offset], old); + bytes[offset] = new; + } + let mut strict = Decoder::new(); + strict.open(&bytes).unwrap(); + assert!(strict.next_frame().is_err()); + for n_threads in [1, 4] { + let mut decoder = Decoder::new(); + decoder + .set_options(Options { + libwebp_compat: true, + n_threads, + ..Options::default() + }) + .unwrap(); + decoder.set_format(Format::Rgba).unwrap(); + decoder.open(&bytes).unwrap(); + assert_eq!(pixel_hash(&mut decoder), expected); + } + let mut decoder = Decoder::new(); + decoder + .set_options(Options { + libwebp_compat: true, + ..Options::default() + }) + .unwrap(); + decoder.set_format(Format::Rgba).unwrap(); + decoder.open_stream().unwrap(); + for part in bytes.chunks(97) { + decoder.append(part).unwrap(); + } + decoder.end_of_stream().unwrap(); + assert_eq!(pixel_hash(&mut decoder), expected); + } +} diff --git a/tests/data/vp8-compat/05386ead209a72d4411d4d0eadd0f9ed6235730c.vp8 b/tests/data/vp8-compat/05386ead209a72d4411d4d0eadd0f9ed6235730c.vp8 new file mode 100644 index 0000000..55bc6d8 Binary files /dev/null and b/tests/data/vp8-compat/05386ead209a72d4411d4d0eadd0f9ed6235730c.vp8 differ diff --git a/tests/data/vp8-compat/1106e5ea9977c4e83826977b05a664ba311fd396.vp8 b/tests/data/vp8-compat/1106e5ea9977c4e83826977b05a664ba311fd396.vp8 new file mode 100644 index 0000000..29ed297 Binary files /dev/null and b/tests/data/vp8-compat/1106e5ea9977c4e83826977b05a664ba311fd396.vp8 differ diff --git a/tests/data/vp8-compat/2fa90d04ce17374a075194e72510319d4f12e5f1.vp8 b/tests/data/vp8-compat/2fa90d04ce17374a075194e72510319d4f12e5f1.vp8 new file mode 100644 index 0000000..6a662c0 Binary files /dev/null and b/tests/data/vp8-compat/2fa90d04ce17374a075194e72510319d4f12e5f1.vp8 differ diff --git a/tests/data/vp8-compat/31ad5278df83871478b68166dc93f71a9d6ca96a.vp8 b/tests/data/vp8-compat/31ad5278df83871478b68166dc93f71a9d6ca96a.vp8 new file mode 100644 index 0000000..90ad6d7 Binary files /dev/null and b/tests/data/vp8-compat/31ad5278df83871478b68166dc93f71a9d6ca96a.vp8 differ diff --git a/tests/data/vp8-compat/3d641847c558f1e8e204ae6b9ae7a39829df71e8.vp8 b/tests/data/vp8-compat/3d641847c558f1e8e204ae6b9ae7a39829df71e8.vp8 new file mode 100644 index 0000000..269d584 Binary files /dev/null and b/tests/data/vp8-compat/3d641847c558f1e8e204ae6b9ae7a39829df71e8.vp8 differ diff --git a/tests/data/vp8-compat/41232e41e66c25526a68befb54c530f39385239e.vp8 b/tests/data/vp8-compat/41232e41e66c25526a68befb54c530f39385239e.vp8 new file mode 100644 index 0000000..69e3bdd Binary files /dev/null and b/tests/data/vp8-compat/41232e41e66c25526a68befb54c530f39385239e.vp8 differ diff --git a/tests/data/vp8-compat/45c6d1e69830bad6ca4e19014559805c7309a07a.vp8 b/tests/data/vp8-compat/45c6d1e69830bad6ca4e19014559805c7309a07a.vp8 new file mode 100644 index 0000000..efb2353 Binary files /dev/null and b/tests/data/vp8-compat/45c6d1e69830bad6ca4e19014559805c7309a07a.vp8 differ diff --git a/tests/data/vp8-compat/README.md b/tests/data/vp8-compat/README.md new file mode 100644 index 0000000..e17af44 --- /dev/null +++ b/tests/data/vp8-compat/README.md @@ -0,0 +1,30 @@ +# VP8 transform-width regressions + +These are the VP8 payloads of the eleven damaged lossy files found by the +2026-10-03 differential fuzz run. The payload of +`ef1d46488254374abe677fc99c4a1b29bc6c8f21` is identical to +`05386ead209a72d4411d4d0eadd0f9ed6235730c`, so ten fixtures cover eleven cases. +Names are the original libFuzzer corpus identifiers. The tests wrap each payload +in a fresh, valid simple WebP container to isolate transform arithmetic. + +The 160x160 files came from `segment01.webp` and `segment02.webp` in the public +[libwebp test data](https://chromium.googlesource.com/webm/libwebp-test-data/+/06ddd96/). +The three 640x480 files came from a synthetic image generated with ImageMagick +(`-seed 1 -size 640x480 plasma:fractal -attenuate 0.3 +noise Gaussian`) and +encoded by cwebp 1.6.0 at quality 75. None came from private images. + +`tests/vp8_compat.rs` pins FNV-1a hashes of visible Y, U and V bytes decoded +with libwebp 1.6.0's portable C decoder. To reproduce an oracle, wrap a payload +in a RIFF `WEBP` / `VP8` chunk and run +`dwebp -noasm -yuv input.webp -o output.yuv`. The tests do not require a native +decoder or downloaded corpus. + +[RFC 6386 sections 14.3 and 14.4](https://www.rfc-editor.org/rfc/rfc6386.html#section-14.3) +describe signed 16-bit intermediate buffers. Libwebp's C transforms keep the +first pass in 32-bit integers. Seven cases exceed the DCT first-pass width, one +exceeds the WHT first-pass width, and three exceed only the DCT second-pass +width. The latter three agree with default wpd and libwebp C; libwebp's native +x86 SSE2 path differs. Its DCT second-pass additions wrap to 16 bits before +shifting. Consequently a damaged file can have CPU-dependent pixels within +libwebp itself. The opt-in mode targets portable C arithmetic and does not claim +to match every libwebp CPU implementation. diff --git a/tests/data/vp8-compat/d4e4675457e402210e2e72ea9e61f93c40a81a68.vp8 b/tests/data/vp8-compat/d4e4675457e402210e2e72ea9e61f93c40a81a68.vp8 new file mode 100644 index 0000000..a032e69 Binary files /dev/null and b/tests/data/vp8-compat/d4e4675457e402210e2e72ea9e61f93c40a81a68.vp8 differ diff --git a/tests/data/vp8-compat/fd4b9d6c836f6b8c5d6e6705e4e98261dfabb0be.vp8 b/tests/data/vp8-compat/fd4b9d6c836f6b8c5d6e6705e4e98261dfabb0be.vp8 new file mode 100644 index 0000000..67f1c6a Binary files /dev/null and b/tests/data/vp8-compat/fd4b9d6c836f6b8c5d6e6705e4e98261dfabb0be.vp8 differ diff --git a/tests/data/vp8-compat/fddf820775b3266dc39aeadc299d34df5c895299.vp8 b/tests/data/vp8-compat/fddf820775b3266dc39aeadc299d34df5c895299.vp8 new file mode 100644 index 0000000..b623ca0 Binary files /dev/null and b/tests/data/vp8-compat/fddf820775b3266dc39aeadc299d34df5c895299.vp8 differ diff --git a/tests/vp8_compat.rs b/tests/vp8_compat.rs new file mode 100644 index 0000000..84e21da --- /dev/null +++ b/tests/vp8_compat.rs @@ -0,0 +1,158 @@ +use wpd::api::{Decoder, Options}; +use wpd::image::Format; + +const CASES: &[(&[u8], u64)] = &[ + ( + include_bytes!("data/vp8-compat/05386ead209a72d4411d4d0eadd0f9ed6235730c.vp8"), + 0xd5985d6a806cbc3e, + ), + ( + include_bytes!("data/vp8-compat/1106e5ea9977c4e83826977b05a664ba311fd396.vp8"), + 0x606a790c9aa52794, + ), + ( + include_bytes!("data/vp8-compat/2fa90d04ce17374a075194e72510319d4f12e5f1.vp8"), + 0xddf085515fdc19fc, + ), + ( + include_bytes!("data/vp8-compat/31ad5278df83871478b68166dc93f71a9d6ca96a.vp8"), + 0x662636aaaac3ffdb, + ), + ( + include_bytes!("data/vp8-compat/3d641847c558f1e8e204ae6b9ae7a39829df71e8.vp8"), + 0xaaf3d69cff46af8d, + ), + ( + include_bytes!("data/vp8-compat/41232e41e66c25526a68befb54c530f39385239e.vp8"), + 0xc9114c262ae54373, + ), + ( + include_bytes!("data/vp8-compat/45c6d1e69830bad6ca4e19014559805c7309a07a.vp8"), + 0x66bf1476e887b1f4, + ), + ( + include_bytes!("data/vp8-compat/d4e4675457e402210e2e72ea9e61f93c40a81a68.vp8"), + 0x3abd600787eaf9fc, + ), + // ef1d46488254374abe677fc99c4a1b29bc6c8f21 has the same VP8 chunk as 05386. + ( + include_bytes!("data/vp8-compat/05386ead209a72d4411d4d0eadd0f9ed6235730c.vp8"), + 0xd5985d6a806cbc3e, + ), + ( + include_bytes!("data/vp8-compat/fd4b9d6c836f6b8c5d6e6705e4e98261dfabb0be.vp8"), + 0xa09d93c6d9f17892, + ), + ( + include_bytes!("data/vp8-compat/fddf820775b3266dc39aeadc299d34df5c895299.vp8"), + 0x7c9c719b14554032, + ), +]; + +fn webp(vp8: &[u8]) -> Vec { + let mut bytes = b"RIFF".to_vec(); + let padded = vp8.len() + (vp8.len() & 1); + bytes.extend_from_slice(&(padded as u32 + 12).to_le_bytes()); + bytes.extend_from_slice(b"WEBPVP8 "); + bytes.extend_from_slice(&(vp8.len() as u32).to_le_bytes()); + bytes.extend_from_slice(vp8); + bytes.resize(20 + padded, 0); + bytes +} + +// FNV-1a over visible Y, U and V rows, in dwebp -yuv order. These values +// were measured with libwebp 1.6.0's portable C decoder (dwebp -noasm). +fn pixels(decoder: &mut Decoder<'_>) -> u64 { + let picture = decoder.next_frame().unwrap().unwrap(); + assert_eq!(picture.format(), Format::Yuv420p); + let mut hash = 0xcbf29ce484222325u64; + for plane in 0..3 { + for row in picture.rows_of(plane) { + for &value in row { + hash = (hash ^ u64::from(value)).wrapping_mul(0x100000001b3); + } + } + } + hash +} + +#[test] +fn compatibility_matches_libwebp_c_on_every_damaged_lossy_case() { + for &(vp8, expected) in CASES { + let bytes = webp(vp8); + for n_threads in [1, 2, 4] { + let mut decoder = Decoder::new(); + decoder.set_format(Format::Yuv420p).unwrap(); + decoder + .set_options(Options { + libwebp_compat: true, + n_threads, + ..Options::default() + }) + .unwrap(); + decoder.open(&bytes).unwrap(); + assert_eq!(pixels(&mut decoder), expected); + } + } +} + +#[test] +fn incremental_compatibility_uses_the_same_transform_arithmetic() { + for &(vp8, expected) in CASES { + let bytes = webp(vp8); + let mut decoder = Decoder::new(); + decoder.set_format(Format::Yuv420p).unwrap(); + decoder + .set_options(Options { + libwebp_compat: true, + ..Options::default() + }) + .unwrap(); + decoder.open_stream().unwrap(); + for part in bytes.chunks(97) { + decoder.append(part).unwrap(); + } + decoder.end_of_stream().unwrap(); + assert_eq!(pixels(&mut decoder), expected); + } +} + +#[test] +fn compatibility_preserves_the_valid_corpus_pixels() { + if cfg!(miri) { + return; + } + let corpus = std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join("wpd-test-data"); + for entry in std::fs::read_dir(corpus).unwrap() { + let path = entry.unwrap().path(); + if !path.extension().is_some_and(|ext| ext == "webp") { + continue; + } + let bytes = std::fs::read(&path).unwrap(); + let mut hashes = Vec::new(); + for compat in [false, true] { + let mut decoder = Decoder::new(); + decoder + .set_options(Options { + libwebp_compat: compat, + n_threads: 4, + ..Options::default() + }) + .unwrap(); + decoder.set_format(Format::Rgba).unwrap(); + decoder.open(&bytes).unwrap(); + let mut frames = Vec::new(); + while let Some(picture) = decoder.next_frame().unwrap() { + let mut hash = 0xcbf29ce484222325u64; + for row in picture.rows_of(0) { + for &value in row { + hash = (hash ^ u64::from(value)).wrapping_mul(0x100000001b3); + } + } + frames.push(hash); + } + hashes.push(frames); + } + assert_eq!(hashes[0], hashes[1], "{}", path.display()); + } +} diff --git a/tests/vp8_mode_reset.rs b/tests/vp8_mode_reset.rs new file mode 100644 index 0000000..32af5bd --- /dev/null +++ b/tests/vp8_mode_reset.rs @@ -0,0 +1,52 @@ +use wpd::vp8::Decoder; + +fn visible_planes(decoder: &Decoder) -> Vec> { + (0..3) + .map(|p| { + let plane = decoder.picture.planes[p]; + let bytes = decoder.picture.plane(p); + let (width, height) = if p == 0 { + (decoder.width as usize, decoder.height as usize) + } else { + ( + (decoder.width as usize).div_ceil(2), + (decoder.height as usize).div_ceil(2), + ) + }; + (0..height) + .flat_map(|y| { + let at = plane.origin + y * plane.stride; + bytes[at..at + width].iter().copied() + }) + .collect() + }) + .collect() +} + +// This separate test binary starts before any API decoder initializes CPU +// detection, making the raw scalar-to-assembly transition reproducible. +#[test] +fn compatibility_restores_the_original_raw_decoder_tables() { + let vp8 = + include_bytes!("data/vp8-compat/05386ead209a72d4411d4d0eadd0f9ed6235730c.vp8"); + let mut decoder = Decoder::new(); + decoder.decode_frame(vp8).unwrap(); + let strict = visible_planes(&decoder); + + // Another API decoder publishes detected CPU flags while this raw + // decoder retains the tables it selected at construction. + let _other = wpd::api::Decoder::new(); + for _ in 0..2 { + decoder.libwebp_compat = true; + decoder.decode_frame(vp8).unwrap(); + assert_ne!(strict, visible_planes(&decoder)); + + decoder.libwebp_compat = false; + decoder.decode_frame(vp8).unwrap(); + for (p, (expected, actual)) in + strict.iter().zip(visible_planes(&decoder)).enumerate() + { + assert!(expected == &actual, "visible plane {p} changed after reset"); + } + } +} diff --git a/tools/tests/compat.rs b/tools/tests/compat.rs new file mode 100644 index 0000000..bef0880 --- /dev/null +++ b/tools/tests/compat.rs @@ -0,0 +1,79 @@ +#![forbid(unsafe_code)] + +use std::io::Write; +use std::process::{Command, Output, Stdio}; + +/* A synthetic two-pixel lossless red/green image, encoded with cwebp 1.6.0 + * from P6\n2 1\n255\n followed by ff0000 00ff00. */ +const STILL: &[u8] = &[ + 0x52, 0x49, 0x46, 0x46, 0x1e, 0, 0, 0, 0x57, 0x45, 0x42, 0x50, 0x56, 0x50, 0x38, + 0x4c, 0x11, 0, 0, 0, 0x2f, 1, 0, 0, 0, 0x0f, 0xb0, 0xff, 0xf3, 0x1f, 0xf3, 0x1f, + 0x15, 0x32, 0xa2, 0xff, 1, 0, +]; + +fn run(args: &[&str], input: &[u8]) -> Output { + let mut child = Command::new(env!("CARGO_BIN_EXE_wpd")) + .args(args) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .unwrap(); + + child.stdin.take().unwrap().write_all(input).unwrap(); + child.wait_with_output().unwrap() +} + +fn chunk(tag: &[u8; 4], payload: &[u8]) -> Vec { + let mut out = tag.to_vec(); + + out.extend_from_slice(&(payload.len() as u32).to_le_bytes()); + out.extend_from_slice(payload); + if payload.len() & 1 != 0 { + out.push(0); + } + out +} + +fn riff(payload: &[u8]) -> Vec { + let mut out = b"RIFF".to_vec(); + + out.extend_from_slice(&(payload.len() as u32 + 4).to_le_bytes()); + out.extend_from_slice(b"WEBP"); + out.extend_from_slice(payload); + out +} + +#[test] +fn compatibility_applies_to_info_output_muxers_and_streaming() { + let mut payload = chunk(b"VP8X", &[0, 0, 0, 0, 1, 0, 0, 0, 0, 0]); + + payload.extend(chunk(b"VP8X", &[0, 0, 0, 0, 1, 0, 0, 0, 0, 0])); + payload.extend_from_slice(&STILL[12..]); + payload.extend_from_slice(b"JUNK\xff\xff\xff\xfftail"); + let data = riff(&payload); + let reference = run(&["--muxer", "pam", "-", "-"], STILL); + + assert!(reference.status.success()); + assert!(!run(&["--info", "-"], &data).status.success()); + assert!(run(&["--libwebp-compat", "--info", "-"], &data) + .status + .success()); + for stream in ["1", "997"] { + let output = run( + &[ + "--libwebp-compat", + "--stream", + stream, + "--muxer", + "pam", + "-", + "-", + ], + &data, + ); + + assert!(output.status.success()); + assert_eq!(output.stdout, reference.stdout); + } +} diff --git a/tools/wpd.rs b/tools/wpd.rs index d54e2c5..0ed3f8f 100644 --- a/tools/wpd.rs +++ b/tools/wpd.rs @@ -99,6 +99,9 @@ const USAGE_TAIL: &str = concat!( " --info\n", " print canvas, animation, the frame table and per-frame\n", " timing to stdout\n", + " --libwebp-compat\n", + " match libwebp's acceptance and lossy decoding of damaged\n", + " still images; default is strict decoding\n", " --stream u32\n", " decode incrementally, appending this many bytes at a time,\n", " instead of opening the file whole\n", @@ -280,6 +283,7 @@ struct Options { scale: Option<(i32, i32)>, frame_size_limit: u32, info: bool, + libwebp_compat: bool, subframe: bool, muxer: Option, verify: Option, @@ -303,6 +307,7 @@ const OPTIONS: &[(&str, Option, bool)] = &[ ("muxer", None, true), ("verify", None, true), ("info", None, false), + ("libwebp-compat", None, false), ("loops", None, true), ("cpumask", None, true), ("subframe", None, false), @@ -374,6 +379,7 @@ fn set(o: &mut Options, name: &str, value: String) -> Result<(), &'static str> { } "info" => o.info = true, "subframe" => o.subframe = true, + "libwebp-compat" => o.libwebp_compat = true, _ => return Err(MISSING), } Ok(()) @@ -657,6 +663,7 @@ fn new_decoder( n_threads: i32, scale: Option<(i32, i32)>, frame_size_limit: u32, + libwebp_compat: bool, ) -> Option> { let mut decoder = Decoder::new(); @@ -665,6 +672,7 @@ fn new_decoder( n_threads, scale, frame_size_limit, + libwebp_compat, ..api::Options::default() }) .is_err() @@ -854,7 +862,20 @@ fn run( let mut out_format = opts.out_format; if opened && output.muxer != Muxer::Raw { - let Ok(image) = api::info(&data) else { + let info = if opts.libwebp_compat { + let mut decoder = Decoder::new(); + + decoder + .set_options(api::Options { + libwebp_compat: true, + ..Default::default() + }) + .and_then(|()| decoder.open(&data)) + .and_then(|()| decoder.info()) + } else { + api::info(&data) + }; + let Ok(image) = info else { eprintln!("{}: cannot read image header", input_name.to_string_lossy()); return ExitCode::FAILURE; }; @@ -890,6 +911,7 @@ fn run( opts.n_threads, opts.scale, opts.frame_size_limit, + opts.libwebp_compat, ) else { return ExitCode::FAILURE; }; @@ -905,6 +927,7 @@ fn run( opts.n_threads, opts.scale, opts.frame_size_limit, + opts.libwebp_compat, ) else { return ExitCode::FAILURE; };