From 8fe0b0110f55a8bbb42eec8069f8bf8ff978dcee Mon Sep 17 00:00:00 2001 From: Victor Kuznetsov Date: Wed, 5 Aug 2026 11:17:38 -0700 Subject: [PATCH 1/2] Make the video SynthID operating point measurable and hard to move silently The shipped profile was certified by one oracle row, but only noise_std was pinned: long_side and fps -- two thirds of what the verifier was actually shown -- could move with a green suite. The test now derives the pin from data/evaluations/video-synthid-oracle.csv, so a default without a certifying row fails. The certified profile is a perturbation-to-signal ratio, not a bare noise_std. sd-vae-ft-mse publishes no scaling_factor key, so 0.18215 comes from the AutoencoderKL class default under an upper-unbounded diffusers pin. The loader now gates that value, carries it on VideoVaeRuntime, and passes it into encode and decode so the validated value is the applied value. video_synthid_sweep.py loads through the same function: the harness producing the certified rows was the one path exempt from the gate it exists to feed. psnr_db is measured against the already-resized frame and before the encoder, so it cannot see the downscale, the decimation, or the codec, and no in-loop metric can. scripts/video_fidelity_probe.py scores the delivered file end to end, streaming the way the engine does and sharing its frame-selection rule rather than copying it -- a frame-count check cannot catch a rule that reorders frames without changing how many. The manifest gains source geometry, vae, track, verbatim verdict and session fields. The two 2026-07-31 rows keep them empty: they were never recorded and are not recoverable. Verdicts now have four states, because the verifier's unclear reading logged as not_detected is the silent regression the manifest exists to prevent. docs/video-synthid-quality-research.md records the research behind this: the noise axis is worth about 2 dB and is nearly exhausted, resolution is the real prize but is an uncertified destruction axis rather than a free win, and every proposed autoencoder swap was refuted. First local measurements included. Verified: engine output is byte-identical before and after the refactor on a locally built clip, at noise_std 0.00 and 0.15. Co-Authored-By: Claude Opus 5 --- .claude/rules/development.md | 21 + README.md | 8 + data/README.md | 37 ++ data/evaluations/video-synthid-oracle.csv | 6 +- docs/index.md | 1 + docs/known-limitations.md | 8 + docs/module-internals.md | 30 +- docs/synthid.md | 5 + docs/video-synthid-quality-research.md | 493 ++++++++++++++++++++ scripts/video_fidelity_probe.py | 216 +++++++++ scripts/video_synthid_sweep.py | 53 ++- src/remove_ai_watermarks/video_invisible.py | 29 +- src/remove_ai_watermarks/video_synthid.py | 4 + tests/test_video_invisible.py | 43 +- 14 files changed, 927 insertions(+), 27 deletions(-) create mode 100644 docs/video-synthid-quality-research.md create mode 100644 scripts/video_fidelity_probe.py diff --git a/.claude/rules/development.md b/.claude/rules/development.md index c58230b..64cb0ac 100644 --- a/.claude/rules/development.md +++ b/.claude/rules/development.md @@ -82,4 +82,25 @@ Before changing anything in the detection path, record the detectors' exact verd over a local sample first and diff them after. A refactor here is only correct if that record is byte-identical, and a green test suite does not establish that on its own. +## A certified operating point is data, not a constant + +The video SynthID default is only meaningful as a row in +`data/evaluations/video-synthid-oracle.csv`, so +`test_shipped_defaults_match_a_certified_manifest_row` derives the pin from that +manifest instead of restating literals. Pinning `noise_std` alone had let `long_side` +and `fps` -- two thirds of what the oracle was actually shown -- move with a green +suite. + +The certified profile is a perturbation-to-signal ratio, not a bare `noise_std`, so +the latent scaling factor is gated in `load_video_vae_runtime`, carried on +`VideoVaeRuntime`, and passed into encode and decode: the validated value and the +applied value are one measurement. Anything that produces oracle evidence loads +through that function. `video_synthid_sweep.py` hand-rolled the load and was the one +path exempt from the gate it exists to feed, which is exactly backwards. + +Prove a video-path refactor the same way the detection path is proven, and without +needing an oracle carrier: build a clip from a tracked fixture with ffmpeg, run the +engine before and after, and require an identical output sha256. Keep the generated +media outside the repository. + Environment setup, dependency recovery, CI behavior, and fixture policy: [`../../docs/development.md`](../../docs/development.md). diff --git a/README.md b/README.md index 175e889..03030ac 100644 --- a/README.md +++ b/README.md @@ -398,6 +398,14 @@ verifier verdict blank: uv run --extra video --extra diffusion python scripts/video_synthid_sweep.py input.mp4 -o sweep/ ``` +To score what an output actually cost, use `scripts/video_fidelity_probe.py`: +the engine's own PSNR is measured before the resize and the encode, so only the +probe sees the delivered picture. + +```bash +uv run --extra video python scripts/video_fidelity_probe.py input.mp4 input_clean.mp4 +``` + The control must still be SynthID-positive before a negative candidate can count as removal evidence. In the 2026-07-29 two-clip calibration, both matched controls were positive in Gemini's built-in SynthID verifier; the stronger diff --git a/data/README.md b/data/README.md index 682c087..4a4667a 100644 --- a/data/README.md +++ b/data/README.md @@ -38,3 +38,40 @@ data/ The source distribution excludes `data/`; the wheel contains only package runtime assets. + +## Video SynthID oracle manifest + +`evaluations/video-synthid-oracle.csv` is the only evidence that the shipped +video removal profile works, so it is also the source of truth for three shipped +defaults: `tests/test_video_invisible.py` asserts that `noise_std`, `long_side`, +and `fps` together match a row this manifest records as certified. Changing one +of those three without adding the row that certifies it fails the suite. `vae` is +deliberately outside that check because neither tracked row records one; add it +to the assertion in the same commit as the first row that does. + +| Column | Meaning | +| --- | --- | +| `date`, `source_url`, `source_sha256` | Identify the carrier. | +| `source_width`, `source_height`, `source_fps` | Carrier geometry. Without it the actual downscale factor of a row cannot be recovered later. | +| `duration_seconds`, `source_verdict` | Clip length submitted and the verifier's reading of the untouched carrier. | +| `vae`, `noise_std`, `long_side`, `fps`, `seed` | The full run configuration. | +| `output_sha256` | Identifies the exact submitted file. | +| `output_verdict` | One of `detected`, `not_detected`, `indeterminate`, `refused`. | +| `output_verdict_text` | The verifier's wording, verbatim. | +| `output_detected_range` | Time range the verifier reported for the output. | +| `track` | `visual`, `audio`, `both`, or empty. The verifier scores tracks separately and this path copies source audio unchanged. | +| `session_id` | Groups rows submitted in one oracle session, so per-session drift stays visible. | +| `stratum` | Content class of the carrier, for stratified certification. | +| `psnr_db`, `temporal_residual_ratio` | Fidelity measurements. Neither is a watermark verdict. | + +Record `indeterminate` when the verifier answers with its unclear state rather +than a negative: an unclear reading logged as `not_detected` is exactly the +silent regression this manifest exists to prevent. Leave a field empty when it +was not recorded, and never backfill it with a plausible value. + +`psnr_db` is measured against the already-resized frame and before the encode, +so it excludes the downscale, the decimation, and the codec. Rows stay +comparable to each other only while that definition holds. + +The two 2026-07-31 rows predate this schema; their empty fields were never +recorded and are not recoverable from the row. diff --git a/data/evaluations/video-synthid-oracle.csv b/data/evaluations/video-synthid-oracle.csv index e084e96..74b90f8 100644 --- a/data/evaluations/video-synthid-oracle.csv +++ b/data/evaluations/video-synthid-oracle.csv @@ -1,3 +1,3 @@ -date,source_url,source_sha256,duration_seconds,source_verdict,noise_std,long_side,fps,seed,output_sha256,output_verdict,psnr_db,temporal_residual_ratio -2026-07-31,https://storage.googleapis.com/gdm-deepmind-com-prod-public/media/media/veo__veo-3__off-road.mp4,79a552b9406a079682440c31f14d33a10ba8e1b8b2e96425f5de70f63350299d,8,detected_all_frames,0.10,512,12,0,079165105d4c56e1612091987c08c2627049423025f74c0d4e245fb47c2ff0e3,detected,26.2932,1.0072 -2026-07-31,https://storage.googleapis.com/gdm-deepmind-com-prod-public/media/media/veo__veo-3__off-road.mp4,79a552b9406a079682440c31f14d33a10ba8e1b8b2e96425f5de70f63350299d,8,detected_all_frames,0.15,512,12,0,1c4046bcfdead138353b4e2a73339ba227bb5e544878d80c5bc6cd8427c7b00e,not_detected,25.3911,1.0578 +date,source_url,source_sha256,source_width,source_height,source_fps,duration_seconds,source_verdict,vae,noise_std,long_side,fps,seed,output_sha256,output_verdict,output_verdict_text,output_detected_range,track,session_id,stratum,psnr_db,temporal_residual_ratio +2026-07-31,https://storage.googleapis.com/gdm-deepmind-com-prod-public/media/media/veo__veo-3__off-road.mp4,79a552b9406a079682440c31f14d33a10ba8e1b8b2e96425f5de70f63350299d,,,,8,detected_all_frames,,0.10,512,12,0,079165105d4c56e1612091987c08c2627049423025f74c0d4e245fb47c2ff0e3,detected,,,,,,26.2932,1.0072 +2026-07-31,https://storage.googleapis.com/gdm-deepmind-com-prod-public/media/media/veo__veo-3__off-road.mp4,79a552b9406a079682440c31f14d33a10ba8e1b8b2e96425f5de70f63350299d,,,,8,detected_all_frames,,0.15,512,12,0,1c4046bcfdead138353b4e2a73339ba227bb5e544878d80c5bc6cd8427c7b00e,not_detected,,,,,,25.3911,1.0578 diff --git a/docs/index.md b/docs/index.md index af36164..d73188b 100644 --- a/docs/index.md +++ b/docs/index.md @@ -37,4 +37,5 @@ The current behavior is defined by the code, tests, README, and user guides. - [Doubao reverse-alpha research](research-doubao-distillation.md) - [SynthID identity research](synthid-robust-identity-research.md) - [SynthID identity follow-up](synthid-robust-identity-research-2026-06-08.md) +- [Video SynthID quality research](video-synthid-quality-research.md) - [Text protection research](text-protection-research.md) diff --git a/docs/known-limitations.md b/docs/known-limitations.md index 6cafb0e..be8e3fb 100644 --- a/docs/known-limitations.md +++ b/docs/known-limitations.md @@ -118,6 +118,14 @@ rather than clip duration. Runtime still grows linearly with duration, and the separate multi-candidate research sweep deliberately retains its short sampled prefix so it can reuse identical latents across candidate strengths. +The reported `psnr_db` is measured against the already-resized frame and before +the H.264 encode, so it excludes the downscale, the frame decimation, and the +encoder. It measures the VAE round trip plus latent noise at the working geometry. +[`video-synthid-quality-research.md`](video-synthid-quality-research.md) records +what the manifest rows constrain, what a higher-resolution or higher-frame-rate +profile would need in order to be certified, and the audio-track question this +path has not yet answered. + ### Strength is content and seed dependent The two profiles resolve an unset strength differently, because different things diff --git a/docs/module-internals.md b/docs/module-internals.md index ed4a425..f7c9e30 100644 --- a/docs/module-internals.md +++ b/docs/module-internals.md @@ -252,7 +252,35 @@ quality measurements. Neither is a watermark detector. The high-level result reports completed removal without a separate verification-status flag. The companion `scripts/video_synthid_sweep.py` imports the same engine helpers to build a matched control and candidate grid, preventing research and shipped -regeneration paths from drifting. The full-clip oracle floor is +regeneration paths from drifting. + +The engine's `psnr_db` is measured against the already-resized frame and before +the encoder, so it scores the VAE round trip plus latent noise and cannot see +the downscale, the decimation, or the codec. No in-loop metric can: the +candidate frame is captured before it reaches the encoder pipe. +`scripts/video_fidelity_probe.py` covers the rest by decoding the delivered file +after muxing, upscaling it back to the source geometry, and scoring it against +the untouched source frames. It also reports the delivered file's bitrate, so a +fixed-crf bitrate rise cannot read as unchanged quality; because the mux copies +source audio verbatim, that figure is a container bitrate, not a video one. The +probe streams and accumulates the same way the engine does, so its peak memory +does not grow with clip length. It drives the source through the engine's own +`_iter_sampled_frames` at the source geometry rather than repeating the +selection rule: a frame-count check cannot catch a rule that reorders frames +without changing how many, so the rule itself has to be shared. + +`load_video_vae_runtime` asserts the default model's latent scaling factor +against `VIDEO_SYNTHID_VAE_SCALING_FACTOR` and warns that no certified profile +exists for any other model. The published `sd-vae-ft-mse` config carries no +`scaling_factor` key, so the value is a `diffusers` class default under an +upper-unbounded pin, and the certified profile is a perturbation-to-signal ratio +rather than a bare `noise_std`. A library bump that moved that default would +otherwise rescale every perturbation with a green suite. The validated factor is +carried on `VideoVaeRuntime` and passed into encode and decode, so the gated +value and the applied value are one measurement rather than three independent +reads. `scripts/video_synthid_sweep.py` loads through the same function: the +harness that produces the certified rows is the last place that should be exempt +from the gate. The full-clip oracle floor is `noise_std=0.15`: on the public eight-second Veo carrier, `0.10` remained detected while `0.15` did not. diff --git a/docs/synthid.md b/docs/synthid.md index 304e923..9dee465 100644 --- a/docs/synthid.md +++ b/docs/synthid.md @@ -372,6 +372,11 @@ itself. `0.15` is therefore the shipped default. The tracked manifest `data/evaluations/video-synthid-oracle.csv` records the public source URL, hashes, fidelity metrics, and verdicts without committing generated videos. +What that calibration does and does not constrain, the ranked experiment program +for trading less quality for the same removal, and the change candidates that were +refuted along the way are recorded in +[`video-synthid-quality-research.md`](video-synthid-quality-research.md). + The VAE perturbation follows the general regeneration-attack construction from Zhao et al. The video-specific control and temporal metric are local additions. VideoMarkBench motivates testing frame aggregation and matched perturbations, diff --git a/docs/video-synthid-quality-research.md b/docs/video-synthid-quality-research.md new file mode 100644 index 0000000..f1ac015 --- /dev/null +++ b/docs/video-synthid-quality-research.md @@ -0,0 +1,493 @@ +# Video SynthID quality research (2026-08-05) + +> Research archive. This page records experiments and decisions from the date +> above. It may mention prototypes or defaults that were later changed. Use the +> user guides and current source code for the supported interface. + +Cited research behind the question **"can the video SynthID path keep removing the +mark while giving up far less quality than the shipped 512 px / 12 fps profile?"** +Produced by a 13-agent workflow: 6 parallel scouts (pipeline ablation, SynthID +internals, attack literature, autoencoder landscape, perceptual masking, experiment +design), 11 change proposals from 3 independent design angles, and 3 adversarial +critics (signal theory, repository engineering, verifiability). No experiment was +run against the provider oracle for this page; every claim is labeled MEASURED, +REPORTED, or INFERRED. + +Repository claims below were re-verified against source after the workflow +returned: the PSNR reference frame, the audio stream copy, the missing `_fit_size` +clamp, the absent `enable_tiling` call, the crf asymmetry, the test pins, and the +absent `scaling_factor` key in the cached checkpoint config. + +## Context + +`remove_video_invisible` regenerates video pixels through `sd-vae-ft-mse`: frames +are decimated to 12 fps, resized to a 512 px long side, encoded to latents with +`latent_dist.mode()`, perturbed by one seeded spatial noise field shared across +every frame, decoded, and streamed to an H.264 encoder at crf 18. A separate +stream-copy mux re-adds source audio and strips metadata. The shipped +`noise_std=0.15` is certified by one oracle row on one carrier. + +The quality complaint is real: a 1080p source is delivered at roughly a quarter of +its linear resolution and half its frame rate. This page asks what part of that +cost is buying removal and what part is buying nothing. + +## What the two oracle rows do and do not prove + +`data/evaluations/video-synthid-oracle.csv`, one carrier (Veo 3 off-road, +sha256 `79a552b9...`), one seed, one geometry: + +| noise_std | long_side | fps | verdict | psnr_db | temporal_residual_ratio | +| --- | --- | --- | --- | --- | --- | +| 0.10 | 512 | 12 | detected | 26.2932 | 1.0072 | +| 0.15 | 512 | 12 | not_detected | 25.3911 | 1.0578 | + +They prove exactly one thing: the envelope of 512 px, 12 fps, crf 18, and a full +VAE round trip does **not** silence the oracle on its own. The verdict flip is +bought by the last 0.05 of latent noise. + +They cannot decompose that conjunction. No row varies `long_side` or `fps`, so the +contribution of the downscale is unmeasured. The frequent reading "the 512 px +downscale was probably doing the removal work" is not supported, and neither is its +opposite. + +## Measured facts from the repository + +- **The fidelity metric is blind to the expensive steps.** `_iter_sampled_frames` + applies `cv2.resize` at [`video_invisible.py:236`](../src/remove_ai_watermarks/video_invisible.py); + the same generator feeds the loop at `:401-407`; the accumulator at `:429-435` + zips those already-resized frames against `regenerated` captured at `:423`, which + is before `frame_pipe.write` at `:430`. `psnr_db` therefore excludes the + downscale, the decimation, and the encoder. It measures the VAE round trip plus + latent noise at the working geometry, nothing else. +- **Frame decimation cannot destroy the carrier.** `_iter_sampled_frames` + (`:224-238`) is pure subset selection: the only statement inside the threshold + test is `yield cv2.resize(...)`, with no arithmetic across frames. The + perturbation is equally time-blind: `shared_noise.expand(latents.shape[0], -1, -1, -1)` + at `:295` broadcasts one CHW field across the batch and never mixes across time. + Any fps effect on the verdict is a detection-probability effect, not carrier + destruction. +- **`_fit_size` has no clamp at 1.0.** `scale = long_side / max(width, height)` at + `:88` upscales any source whose long side is below 512, and both axes are floored + independently to a multiple of 8 at `:89-96`, which introduces an anamorphic + shift at scale 1.0 (1366x768 becomes 1360x768). +- **`enable_tiling` is never called** anywhere in `src/`, `scripts/`, or `tests/`. + `enable_slicing()` at `:135` already splits both encode and decode to single + frames, so activation VRAM is set by one frame and does not scale with + `--batch-size`. +- **crf asymmetry:** 18 on the invisible path (`:321`, `:386`) against 14 on the + visible path ([`video_visible.py:1343`](../src/remove_ai_watermarks/video_visible.py)) + through the same encoder. +- **No HDR guard on the invisible path.** `_HDR_TRANSFERS` (`video_visible.py:97`) + and the `component_depth > 8` rejection (`:1322`) exist only for visible removal. +- **Audio is byte-copied.** `-map 0:v:0`, `-map 1:a?`, `-c copy` at + [`video_encoding.py:413`](../src/remove_ai_watermarks/video_encoding.py). +- **Only `noise_std` is pinned.** `tests/test_video_invisible.py:285` asserts + `DEFAULT_VIDEO_SYNTHID_NOISE_STD == 0.15`. Neither `long_side` nor `fps` is + pinned anywhere, so changing either breaks no test. +- **`scaling_factor` 0.18215 is a class default, not a checkpoint fact.** The + cached `config.json` for `stabilityai/sd-vae-ft-mse` has no `scaling_factor` key + at all (verified locally: `_class_name`, `latent_channels`, `block_out_channels`, + `sample_size`, and block types only). The value comes from the `AutoencoderKL` + class default under `diffusers>=0.38.0` with no upper bound (`pyproject.toml:94`, + `uv.lock` resolves 0.39.0), while `maintain.sh` runs `uv-outdated`. A dependency + bump can move the certified operating point with a fully green test suite. +- **The manifest schema cannot record what a real program needs:** no source + geometry, no control row, no track column, no verbatim verdict, no indeterminate + state. + +## Measured facts from external sources + +- Gemini video verification quota: 10 checks per rolling 24 hours, up to 5 minutes + of video total, under 90 seconds and 100 MB per file. The verifier reports which + parts of the video carry the mark, and it has a **third** state beyond detected + and not detected: unclear, with documented causes including "not enough details + to watermark". +- The verifier scores **audio and visual tracks separately**. Google's published + example verdict reads as SynthID detected in the audio over a time range with no + SynthID detected in the visuals. +- SynthID-Image (arXiv:2510.09263) is a post-hoc, model-independent pixel-space + watermark: a separate encoder network stamps an already-decoded image rather than + being injected into the generator's latents. Its published payload figure is 136 + bits within a 512x512 image, and its product setup runs at 1536x1536. + +## Inferred, with the reasoning that makes them weak + +- **The noise axis is nearly exhausted.** Fitting `MSE = A + B * noise_std^2` to the + two measured rows gives A = 124.5 and B = 2815, so a pure round trip + (`noise_std = 0`) lands near 27.2 dB. The entire noise budget is worth at most + **+1.79 dB**; everything else is the autoencoder. This is a two-parameter fit to + two points with zero degrees of freedom, and the assumptions that MSE is + quadratic in `noise_std` and that reconstruction and noise errors are orthogonal + are untested. One local run at `noise_std=0` replaces it with a measurement. +- **Statistical power of the current certification.** With zero failures at n = 1, + the exact one-sided Clopper-Pearson bound `1 - 0.05^(1/n)` is 95%: the data are + compatible with removal failing almost always. n = 15 gives 18.1%, n = 30 gives + 9.5%. +- **Size of the geometry prize.** At a 1280x720 source, `_fit_size(1280, 720, 512)` + is `(512, 288)` (pinned in `tests/test_video_synthid_sweep.py:30`), so 6.25x of + the pixels are discarded by the downscale and another 2x by decimation from a + 24 fps source. Ladder rungs: 768 gives 2.25x the current area, 1024 gives 4.0x. + The carrier's own geometry is recorded nowhere, so this is conditional. +- **The direction of the resolution effect is disputed, and both sides are + inference.** Against raising it: the absolute frequency ceiling below which an f8 + VAE reconstructs faithfully is tied to the latent pitch, so 512x288 (a 64x36 + latent) preserves roughly up to 32 cycles per frame width while 1920 (a 240x135 + latent) preserves up to about 120. Raising resolution moves the carrier band out + of the regime the decoder synthesizes and into the regime it reproduces + faithfully, handing the detector more evidence. For raising it: Google's product + operating point is 1536x1536, so native is closer to the distribution the + watermark encoder targets. Note that the second argument cuts against the + proposal rather than for it. + +## First local measurements (2026-08-05) + +Run without the oracle on a locally built carrier: a 6-second 1280x720 24 fps +clip panning across `data/fixtures/provenance/doubao-1.png`, processed by the +shipped path at 512 px / 12 fps on MPS. Generated media stayed outside the +repository. + +| noise_std | engine `psnr_db` | end-to-end PSNR | end-to-end SSIM | bitrate | +| --- | --- | --- | --- | --- | +| 0.00 | 27.8049 | 27.6935 | 0.6955 | 2564 kbps | +| 0.15 | 25.8853 | 25.8541 | 0.6703 | 2716 kbps | + +Three readings, all MEASURED, none of them about SynthID: + +1. **The two-point fit's shape survives contact with independent content.** The + pure round trip lands at 27.80 dB against the 27.2 dB the carrier fit + predicted, and the whole noise budget costs 1.92 dB against the predicted + 1.79 dB. These are not the carrier's numbers, but the decomposition holds: + the autoencoder is the floor and the entire `noise_std` axis is worth about + 2 dB. +2. **On this content the downscale is nearly free in PSNR terms.** End-to-end + PSNR sits within 0.11 dB of the in-loop number, meaning the 512 px geometry + cost almost nothing next to the VAE damage. This clip is a pan over a smooth + generated image with little high-frequency detail, so it is the friendly + case: real camera texture should widen that gap. Run the probe across content + types before trusting any estimate of the geometry prize. +3. **`temporal_residual_ratio` is not meaningful on a near-static shot.** It read + 1.82 at `noise_std=0` and 2.47 at 0.15, far outside the [1.0072, 1.0578] band + ever observed before. A slow pan gives the source almost no motion residual, + so the `max(temporal_baseline, 1e-6)` denominator collapses and the ratio + inflates. This is the predicted defect, now observed rather than argued. + +## The single most important unknown + +**How much the 512 px downscale contributes to removal.** Every quality gain routes +through this question, and the critics moved it from "probably free" to "direction +unknown", which is exactly what makes it the highest-information experiment +available. + +## Ranked experiment program + +Ranked by information per oracle query. The budget is roughly 10 checks per 24 +hours (MEASURED), so the program is paced by calendar, not by money. + +### E0. The zero-oracle tier (0 submissions) - do this first + +Not an experiment on the oracle; the precondition that makes everything else +interpretable. + +- `ffprobe` the carrier and record `source_width`, `source_height`, `source_fps`. + The actual downscale factor of the existing rows is currently unrecoverable. +- Extend the manifest schema: variant/control, source geometry, `vae`, `track` + (audio|visual), verdict state in {DETECTED, NOT_DETECTED, INDETERMINATE, + REFUSED}, verbatim verdict text, detected time range for the output, + `session_id`, content stratum. +- **Measure the `noise_std = 0.0` round trip locally** on the same carrier. This + converts the inferred 27.2 dB ceiling into a measurement and bounds the whole + noise axis for one GPU pass and zero oracle cost. +- Freeze the current defaults' metrics on a fixed clip set, matching the + record-then-diff rule in `.claude/rules/development.md`. +- Determine whether the carrier has an audio track and whether it carries SynthID. + +### E1. Instrument validation: multiplexing and a session anchor (1-2 submissions) + +The verifier reports time ranges, so one file can carry several doses. A 24-second +file of `[control 8 s | certified 0.15 8 s | control 8 s]` should read detected on +the outer segments and not detected on the middle one, both halves already known +from the manifest. + +`encode_video_frames` requires matching dimensions and one frame rate, so +multiplexing works **only along the `noise_std` axis**, not across geometries: one +multiplexed file per geometric envelope. + +Separately, submitting the existing detected file first in each session costs 10% +of the quota, turns "the oracle may have changed" into a per-session gate, and +measures the flip rate on a byte-identical file, a quantity every plan silently +assumes is zero. + +### E2. Resolution ladder at fixed dose - the decisive axis (2 submissions per rung) + +A fixed value, not a ceiling. Rung 1 is `long_side = 768` with fps held at 12 so +exactly one destruction axis moves. Jumping straight to 1920 is a bad first rung: +a 3.75x jump makes a detected verdict uninformative about where the boundary lies. + +- **Matched control:** 768 / 12 / `noise_std = 0.10`, expected DETECTED. It is + strictly stronger than a plain re-encode control because it discharges the + envelope, the VAE round trip, and a nonzero dose at once, and it diffs directly + against the existing 512 row. +- **Candidate:** 768 / 12 / 0.15. +- **Positive:** the resolution axis is open; next rungs 1024 and then native with a + clamp, 2 submissions each. +- **Negative:** the downscale contributes to removal. This is the most valuable + available negative: it kills the native-resolution proposals outright and turns + the question into how much dose must be spent to buy resolution back, at the + known exchange rate B = 2815. +- Watch for a monotonicity violation: 768 / 0.10 reading NOT_DETECTED would + overturn the critics' spectral argument. + +### E3. Frame rate at fixed geometry and dose (2 submissions) + +Control 512 / 24 / 0.10 expected DETECTED; candidate 512 / 24 / 0.15 expected +not detected. The mechanism is provable from source, so the prior is high, but the +only counter-mechanism - per-frame count aggregation - is probabilistic and n = 1 +on an 8-second clip does not measure it. Certify on the longest clip that fits the +90 s / 100 MB limits, because the risk compounds with frame count. + +Ranked below E2 because the outcome is nearly predetermined and the prize (judder) +is smaller than the geometry prize. It is the safest bet if a guaranteed win is +wanted. + +### E4. `noise_std = 0.0` as an oracle probe (1 submission) - deliberately low rank + +The local half of this probe is the cheapest high-information measurement in the +program and already sits in E0 at zero oracle cost. The **oracle** half ranks low: +it buys no quality by construction, and its likely DETECTED outcome is nearly +deducible from the 0.10 row under monotonicity. Its one real value is that +NOT_DETECTED here would be non-monotone against 0.10 and would refute the +assumption underneath every ladder and bisection in this program. Worth one query, +after E2 and E3. + +### E5. Stratified certification of the surviving operating point (15-30 submissions) + +Only after E2 and E3 produce a winner. Zero detections across at least 15 +stratified carriers (at least 3 per stratum: face closeup, fast motion, flat sky, +dark night, text overlay) bounds the failure rate at 18.1%; 30 carriers bound it at +9.5%. Add the project's 1.5x margin convention, at least 2 seeds, at least 2 +sessions, and per-track verdict recording. + +For the flat-sky stratum, check the source reads detected first: Google documents a +"not enough details to watermark" state, so a clean reading there may say nothing +about removal. + +## Surviving change candidates + +**S1. Fix the metric's reference frame (0 oracle submissions).** Add +`source_psnr_db` (candidate upscaled back to native against the untouched source +frame) **and** a separate post-mux pass that decodes the delivered file. The +critics killed the original claim that this would expose crf 18: the candidate is +captured at `:423`, before `frame_pipe.write` at `:430`, so no in-loop metric can +see the codec. Only the post-mux pass covers resize, decimation, crf, and mux +together. Use `INTER_AREA` or a fixed analytic size for the upscale so the CPU +resize does not dominate the loop, and update both other consumers of the +generator in the same change (`read_sampled_frames` at `:188-209`, and +`scripts/video_synthid_sweep.py:147` where `np.stack(frames)` must keep receiving +resized frames). Do not redefine the existing `psnr_db`: the two manifest rows stay +comparable to each other only while that field keeps its exact current meaning. + +**S2. Frame rate as a fixed certified value (after E3).** +`DEFAULT_VIDEO_SYNTHID_FPS` 12.0 to 24.0, keeping `min(fps, source_fps)`. A +ceiling of 60.0 was **rejected**: it makes the delivered operating point a function +of the user's source, so one sample from a family gets certified while the rest +ship uncertified, and the risk direction is unfavorable. Port the VFR/PTS bridge +(`probe_video_timestamps`, `timestamped_input`, currently only in +`video_visible.py`): at native frame rate the output stops reading as a proxy, and +silent CFR-ification becomes a master-quality defect. Add `fps` and `long_side` +pins next to `tests/test_video_invisible.py:285`, tied to the certifying manifest +row. + +**S3. Clamp and align `_fit_size` (0 oracle submissions, with a caveat).** +`scale = min(1.0, long_side / max(width, height))` at `:88`; align the long side +and derive the short side from the true aspect, rounding to the nearest multiple of +8. The pinned `_fit_size(1280, 720, 512) == (512, 288)` survives. The caveat: the +clamp is an **uncertified operator change for a whole class of users**, since a +320x240 source is upscaled to 512x384 today and would run at 320x240 afterwards, +and neither has been tested. Ship it as its own documented change, not as a +drive-by fix. + +**S4. Extend the manifest schema and oracle protocol (0 submissions).** A +precondition for the whole program. Record the verbatim verdict text, because an +unclear state logged as not detected is exactly the silent regression the protocol +exists to prevent. + +**S5. A sigma normalization contract (0 submissions, applies to the current path).** +No VAE swap survived, but two elements are real risks today: `vae.config.scaling_factor` +cannot be trusted, so log `latents.std()` at `:276` and assert the effective +scaling factor at load; and any cross-model sigma transfer must match on the **RMS +of the decoded pixel perturbation**, not on latent units. + +**S6. Additive texture masking - contingency on an E2 negative only.** Not energy +preserving: the certified sigma stays as a floor everywhere and the mask only adds +on top in textured regions. The original energy-preserving form was refuted - the +mechanism normalizes and then clips, which breaks the claimed invariant by a +content-dependent amount, and the proposed unit test asserted the invariant before +the clip and would have stayed green. Energy preservation is also unsafe in +principle: the detector does not need a full frame, and the verifier reports +per-segment, so a flat region perturbed only at the floor is a crop carrying nearly +the full carrier. The salvaged version has **no standalone PSNR win**; its only +value is freeing distortion budget to spend on resolution if E2 shows the dose must +rise. It additionally needs motion compensation for the mask (flow is computed at +`:442`, after the decode at `:423`, so the loop must be restructured) and one +deliberately bad candidate so `temporal_residual_ratio` acquires a known failing +value: it has only ever been observed in [1.0072, 1.0578], so "the ratio looks +fine" is currently an unfalsified claim. + +## Refuted proposals - do not resurrect + +- **Carrier-scale, resolution-invariant noise field.** "Carrier scale" is set by + our VAE's latent pitch, not the carrier's. The motivating arithmetic treats a + field on the latent grid as white at pixel resolution and is wrong by exactly the + square of the scale factor: white noise on a 240-wide latent already spans + 0-120 cycles per frame, entirely inside the band the decoder reproduces. The + construction actually band-limits the perturbation from 0-120 to 0-32 cycles, + making it smoother - the first thing any spread-spectrum extractor's content + suppression removes - while moving its energy toward the peak of the contrast + sensitivity function, making it more visible. Bilinear upsampling is also + heteroscedastic, and global renormalization by `noise.std()` stamps a visible + amplitude lattice rather than fixing it. +- **Native resolution plus tiling plus a carrier-scale field.** Inherits the + arithmetic above, concedes that native processing destroys less carrier without + proposing a replacement mechanism, and bundles four destruction axes into one + candidate on a 4-query budget so a detected verdict is undecomposable. Its cost + model was also wrong by roughly 10x. The salvageable parts moved into S1 and S3. +- **`AutoencoderKLTemporalDecoder`.** Refuted independently by three critics. + Removal is a property of the encode-decode composition, so a strictly more + faithful decoder preserves more carrier at the same sigma. The class implements + neither slicing nor tiling, so `enable_slicing()` at `:135` raises + `NotImplementedError` and the documented bounded-memory property dies. Its + temporal receptive field needs windows of tens of frames, its license carries a + revenue restriction incompatible with an Apache-2.0 default, and its own + published table shows a **worse** FID (9.17 against 7.61). +- **Wan or another temporal video VAE.** Hard refusal in code: `AutoencoderKLWan._encode` + uses `iter_ = 1 + (num_frame - 1) // 4`, so at the shipped `batch_size = 4` only + frame 0 is encoded and a 4-frame batch decodes back to 1, which makes + `zip(..., strict=True)` at `:429` raise. The proposed remedy of carrying the + causal feature cache across chunks is impossible through the public API, since + `clear_cache()` runs on both entry and exit of `_encode` and `_decode`. + `float(vae.config.scaling_factor)` at `:270` and `:292` also raises, because Wan + exposes `latents_mean`/`latents_std` instead. Logically, feeding 4:1 temporal + compression frames 83 ms apart either reconstructs well (carrier preserved, no + removal gain) or hallucinates (no quality prize); the two claims are mutually + exclusive. +- **A 16-channel f8 VAE.** Self-refuted and confirmed by the critics. Quadrupling + latent channels roughly doubles L in Zhao's theorem, so sigma must roughly double + to hold removal, and the VAE frontier in the UnMarker results is monotone with no + published point where a more faithful autoencoder removes **more**. Predicted net + loss of 2-7.5 dB by its own arithmetic. Its config also carries + `scaling_factor = 0.2614`, so an unchanged `noise_std = 0.15` is about 30% weaker + in raw latent units. Only the normalization contract survived, as S5. +- **A per-channel luma/chroma probe.** The arms match in absolute latent units but + not in relative dose per channel, since `scaling_factor` normalizes the aggregate + latent, so the very reading the probe exists for is confounded with dose. The + chroma arm is additionally subsampled by the yuv420p encode, which no current + metric sees. Maximum prize under 1 dB by its own fit. +- **Post-processing with unsharp plus grain, as a shipped stage.** The + carrier-reimport argument is sound and was verified line by line: `unsharp_mask` + reads only its own argument ([`humanizer.py:79`](../src/remove_ai_watermarks/humanizer.py)), + and `adaptive_polish` touches the source only through one float. But reimport was + never the binding risk: + `cv2.addWeighted(img_f, 1.0 + amount, blurred, -amount, 0.0)` at + `_ADAPTIVE_MAX_UNSHARP = 1.0` multiplies the **surviving carrier residue** by up + to 2x as the terminal operation before encoding, and whether that re-arms a given + file depends on that file's unobservable margin. With no local decoder and only a + sampled oracle the stage is structurally uncertifiable, not merely expensive to + certify. The only rescue is moving sharpening **above** the VAE stage so the + removal operator stays terminal. Grain, which monotonically lowers detector SNR, + is the safe half and can be separated. + +## Measuring quality properly + +The current `psnr_db` cannot show the improvement this work exists to produce. +Minimum upgrade: + +1. **`source_psnr_db`** - candidate upscaled back to native against the untouched + source frame. Good for ranking configurations against a common reference, + dominated by unrecoverable high-frequency content, so not a measure of what the + VAE costs. +2. **A post-mux end-to-end pass** - decode the delivered file after + `mux_encoded_video` (`:456`) and compare against the source. The only measure + covering resize, decimation, crf, and mux together. Keep it out of the streaming + loop so `test_stream_batches_consumes_only_one_batch_ahead` stays valid. + Implemented as `scripts/video_fidelity_probe.py`, which streams, reports the + delivered file's bitrate, and shares the engine's frame-selection rule rather + than copying it. +3. **DISTS** (arXiv:2004.07728), built to tolerate texture resampling - exactly + what the VAE does to foliage, skin, and grass, and exactly what PSNR and LPIPS + punish even when the result is perceptually equivalent. It separates "the VAE + resampled the grass" from "the VAE destroyed an edge". +4. **VMAF** through `ffmpeg` `libvmaf`, cheapest to add, whose ADM/DLM feature + names this exact complaint. Its temporal term is only a mean absolute luma + difference between neighboring frames, so it is not a flicker detector: keep the + motion-compensated ratio. +5. **Encoded file size as a third axis.** Every current metric is taken before the + pipe, so a bitrate explosion currently reads as "quality did not suffer". This + matters most for any perturbation that varies frame to frame: at fixed crf it + raises bitrate rather than lowering quality. +6. **`temporal_residual_ratio` repairs**: add a p95 across frames next to the mean + so localized flicker stops being averaged away, add a long-horizon term against + the first frame, and replace `max(temporal_baseline, 1e-6)` with an explicit + undefined result on static shots. +7. **A no-reference metric** (DOVER or FAST-VQA) for the case where no + full-reference measure can compare a 512/12 output against a 1024/24 output on + one scale. Relative ranking within a sweep only; absolute values are + uncalibrated for this artifact class. + +Nothing local measures removal. `remove_video_invisible` checks `get_ai_metadata` +on the output, which is metadata, not pixels. Only a manifest row counts. + +## Risks + +**Error asymmetry is the governing constraint.** Shipping a leak means a user +receives a watermarked file believing it is clean, with no local pixel decoder to +catch it and no feedback path that would surface it. Staying conservative means a +user receives 512 px / 12 fps, a cost that is visible, bounded, and reversible with +an explicit flag. A symmetric test is therefore inadmissible, and the burden of +proof sits entirely on the new default. + +**n = 1.** The current default is certified by one success on one carrier with one +seed. Any quality change inherits that weakness and must not deepen it. + +**Content and seed dependence** is MEASURED on this project's image branch: +survivors switch by content type, and near the threshold the same input flipped +between runs on seed alone. + +**Oracle instability.** Three verdict states, not two. Per-segment reporting. +Separate audio and visual tracks - and the pipeline byte-copies audio, so a Veo 3 +clip with generated sound leaves with its audio SynthID intact, and the manifest +cannot even record which track the 2026-07-31 negative referred to. That is +potentially a shipped product hole, not only an experimental confound. Session +drift is currently unfalsifiable. + +**Observability narrows exactly where quality rises.** At fixed crf the bitrate +grows with pixels per second, so the maximum duration fitting under 100 MB falls as +fps and resolution rise. The configuration users actually receive becomes **less** +observable than the one it replaces. That is a permanent property, not an +inconvenience. + +**Ceilings instead of fixed values.** Any ceiling makes the delivered operating +point a function of the user's source, and with it the perturbation's cycles per +frame - the quantity carrying the certified margin. Ship fixed values. + +**Documentation and test surface.** Only `DEFAULT_VIDEO_SYNTHID_NOISE_STD` is +pinned. Moving `long_side` or `fps` touches hardcoded numbers in `README.md`, +`docs/known-limitations.md`, `docs/cli.md`, `docs/python-api.md`, +`docs/module-internals.md`, `docs/synthid.md`, and `docs/verification-plan.md`. +Drift-prone operational numbers should not live in seven places of prose; their +source of truth is the manifest. + +**Silent configuration drift.** `sd-vae-ft-mse/config.json` carries no +`scaling_factor`, so 0.18215 is a class default under an upper-unbounded +`diffusers>=0.38.0` while `maintain.sh` runs `uv-outdated`. A library bump can move +the certified operating point with a green suite. + +**Missing guards that get worse at native resolution.** There is no HDR or >8-bit +rejection and no VFR/PTS bridge on the invisible path. A 512 px output plainly +reads as a proxy; a native-resolution output reads as a master, and the cost of +silently flattening a 10-bit PQ source to 8-bit SDR rises accordingly. + +**Do not bundle axes.** Raising `long_side`, raising `fps`, and lowering crf all +reduce total destruction and all require recertification. The metric reference-frame +fix is the only exception, because it changes the measurement rather than the +pixels. Do not put a crf 18 to 14 change in the same candidate as a resolution +change. diff --git a/scripts/video_fidelity_probe.py b/scripts/video_fidelity_probe.py new file mode 100644 index 0000000..3e173ff --- /dev/null +++ b/scripts/video_fidelity_probe.py @@ -0,0 +1,216 @@ +"""End-to-end fidelity of a DELIVERED video against its source. + +The engine's own ``psnr_db`` is measured against the already-resized frame and +before the encoder, so it reports the VAE round trip plus latent noise and +nothing else. The downscale, the frame decimation and the H.264 encode -- the +three steps that actually cost the user picture -- are invisible to it, and no +metric computed inside the streaming loop can see them, because the candidate +frame is captured before it is written to the encoder pipe. + +This script measures what the engine cannot: it decodes the delivered file after +muxing, upscales each frame back to the source geometry, and scores it against +the untouched source frame it came from. That reference is deliberately harsh - +detail the downscale destroyed is unrecoverable, so the absolute number is +dominated by content the pipeline never had a chance to keep. Use it to RANK +configurations against a shared reference, not to attribute loss to one stage. + +It streams, for the same reason the engine does: holding a 1080p clip plus its +upscaled candidate and flow maps in memory runs to gigabytes and grows with clip +length. Peak here is a handful of frames regardless of duration. + +It is a research tool, not part of the shipped path, and it measures fidelity +only. Nothing here is a watermark verdict: only a provider-oracle row is. + + uv run python scripts/video_fidelity_probe.py source.mp4 out-512.mp4 out-768.mp4 +""" + +from __future__ import annotations + +import json +import logging +import math +import sys +from pathlib import Path +from typing import TYPE_CHECKING, Any + +import click +import cv2 +import numpy as np + +from remove_ai_watermarks.video_invisible import _iter_sampled_frames, _probe_video +from remove_ai_watermarks.video_temporal import _backward_map, _motion_residual + +sys.path.insert(0, str(Path(__file__).parent)) + +from invisible_quality_audit import _ssim # reuse, do not reimplement a fourth SSIM + +if TYPE_CHECKING: + from collections.abc import Iterator + + from numpy.typing import NDArray + +log = logging.getLogger(__name__) + + +def _delivered_geometry(path: Path) -> tuple[int, int, float]: + capture = cv2.VideoCapture(str(path)) + if not capture.isOpened(): + raise click.ClickException(f"Could not open video: {path}") + try: + width = round(capture.get(cv2.CAP_PROP_FRAME_WIDTH)) + height = round(capture.get(cv2.CAP_PROP_FRAME_HEIGHT)) + fps = float(capture.get(cv2.CAP_PROP_FPS)) + finally: + capture.release() + if width <= 0 or height <= 0 or fps <= 0.0: + raise click.ClickException(f"Video has no usable geometry or frame rate: {path}") + return width, height, fps + + +def _iter_frames(path: Path) -> Iterator[NDArray[Any]]: + capture = cv2.VideoCapture(str(path)) + if not capture.isOpened(): + raise click.ClickException(f"Could not open video: {path}") + try: + while True: + ok, frame = capture.read() + if not ok: + return + yield frame + finally: + capture.release() + + +def _measure( + source: Path, + delivered: Path, + *, + source_geometry: tuple[int, int, float], + duration: float | None, +) -> dict[str, Any]: + source_width, source_height, source_fps = source_geometry + width, height, fps = _delivered_geometry(delivered) + + # Drive the source through the engine's own sampler at the source's geometry, + # so the frame the probe compares against is the frame the engine regenerated. + # Importing the rule is what keeps the pairing correct: a count check cannot + # catch a selection rule that reorders frames without changing how many. + reference_frames = _iter_sampled_frames( + source, + source_fps=source_fps, + duration=duration, + effective_fps=fps, + size=(source_width, source_height), + ) + + squared_error = 0.0 + pixel_count = 0 + ssim_scores: list[float] = [] + temporal_baseline = 0.0 + temporal_candidate = 0.0 + previous_gray: NDArray[Any] | None = None + previous_reference_f32: NDArray[Any] | None = None + previous_candidate_f32: NDArray[Any] | None = None + frame_count = 0 + needs_upscale = (width, height) != (source_width, source_height) + + reference_iter = iter(reference_frames) + delivered_iter = _iter_frames(delivered) + while True: + reference = next(reference_iter, None) + delivered_frame = next(delivered_iter, None) + if reference is None and delivered_frame is None: + break + if reference is None or delivered_frame is None: + raise click.ClickException( + f"{delivered.name} and the sampled source ran out at different points after " + f"{frame_count} frames. Pass --duration to match the prefix this output was " + f"produced from, or check that it came from {source.name}." + ) + candidate = ( + cv2.resize(delivered_frame, (source_width, source_height), interpolation=cv2.INTER_LANCZOS4) + if needs_upscale + else delivered_frame + ) + reference_f32 = reference.astype(np.float32) + candidate_f32 = candidate.astype(np.float32) + difference = reference_f32 - candidate_f32 + squared_error += float(np.sum(difference * difference, dtype=np.float64)) + pixel_count += reference.size + ssim_scores.append( + _ssim(cv2.cvtColor(reference, cv2.COLOR_BGR2GRAY), cv2.cvtColor(candidate, cv2.COLOR_BGR2GRAY)) + ) + + current_gray = cv2.cvtColor(reference, cv2.COLOR_BGR2GRAY) + if previous_gray is not None and previous_reference_f32 is not None and previous_candidate_f32 is not None: + frame_maps = _backward_map(current_gray, previous_gray) + temporal_baseline += _motion_residual(reference_f32, previous_reference_f32, frame_maps) + temporal_candidate += _motion_residual(candidate_f32, previous_candidate_f32, frame_maps) + previous_gray = current_gray + previous_reference_f32 = reference_f32 + previous_candidate_f32 = candidate_f32 + frame_count += 1 + + if frame_count < 2: + raise click.ClickException(f"{delivered.name} paired fewer than two frames against {source.name}") + + mse = squared_error / pixel_count + size_bytes = delivered.stat().st_size + return { + "file": delivered.name, + "source": source.name, + "source_width": source_width, + "source_height": source_height, + "source_fps": round(source_fps, 4), + "width": width, + "height": height, + "fps": round(fps, 4), + "frames": frame_count, + "pixel_ratio": round((width * height) / (source_width * source_height), 4), + "source_psnr_db": math.inf if mse == 0.0 else round(20.0 * math.log10(255.0 / math.sqrt(mse)), 4), + "source_ssim": round(float(np.mean(ssim_scores)), 4), + "temporal_residual_ratio": round(temporal_candidate / max(temporal_baseline, 1e-6), 4), + "size_bytes": size_bytes, + # The mux copies the source audio track verbatim, so this is a container + # bitrate. It still ranks candidates at fixed crf; it is not a video bitrate. + "file_bitrate_kbps": round(size_bytes * 8.0 / (frame_count / fps) / 1000.0, 1), + } + + +@click.command() +@click.argument("source", type=click.Path(exists=True, dir_okay=False, path_type=Path)) +@click.argument("delivered", nargs=-1, required=True, type=click.Path(exists=True, dir_okay=False, path_type=Path)) +@click.option( + "--duration", + type=click.FloatRange(min=0.1), + default=None, + help="Score only the first N seconds of the source, matching a trimmed sweep candidate.", +) +@click.option("--json-out", type=click.Path(dir_okay=False, path_type=Path), help="Also write the rows as JSON.") +def main(source: Path, delivered: tuple[Path, ...], duration: float | None, json_out: Path | None) -> None: + """Score each DELIVERED file against SOURCE end to end.""" + logging.basicConfig(level=logging.INFO, format="%(message)s") + source_geometry = _probe_video(source) + log.info("Source %s: %dx%d at %.4f fps", source.name, *source_geometry) + + rows = [_measure(source, path, source_geometry=source_geometry, duration=duration) for path in delivered] + for row in rows: + log.info( + "%s: %dx%d at %s fps, %.1f%% of source pixels, PSNR %s dB, SSIM %s, temporal %s, %s kbps", + row["file"], + row["width"], + row["height"], + row["fps"], + row["pixel_ratio"] * 100.0, + row["source_psnr_db"], + row["source_ssim"], + row["temporal_residual_ratio"], + row["file_bitrate_kbps"], + ) + if json_out is not None: + json_out.write_text(json.dumps(rows, indent=2), encoding="utf-8") + log.info("Wrote %s", json_out) + + +if __name__ == "__main__": + main() diff --git a/scripts/video_synthid_sweep.py b/scripts/video_synthid_sweep.py index 4348a7e..c5f109a 100644 --- a/scripts/video_synthid_sweep.py +++ b/scripts/video_synthid_sweep.py @@ -39,17 +39,17 @@ from pathlib import Path from typing import TYPE_CHECKING import click -import cv2 import numpy as np from remove_ai_watermarks.video_invisible import ( _decode_frame_latents, _encode_frame_latents, _fit_size, - _pick_device, + _probe_video, _shared_latent_noise, build_temporal_reference, encode_video_frames, + load_video_vae_runtime, paired_psnr, read_sampled_frames, temporal_residual_ratio, @@ -87,9 +87,22 @@ def _sha256(path: Path) -> str: def _write_manifest(output_dir: Path, rows: Sequence[dict[str, str]]) -> Path: path = output_dir / "sweep.csv" + # Mirrors the tracked manifest's run-configuration fields (data/README.md) so a + # row carries into data/evaluations/video-synthid-oracle.csv without hand + # reconstruction. The two 2026-07-31 rows show the cost of not doing this: their + # geometry was never recorded and cannot be recovered from the row. fieldnames = [ "variant", + "source_sha256", + "source_width", + "source_height", + "source_fps", + "duration_seconds", + "vae", "noise_std", + "long_side", + "fps", + "seed", "psnr_db", "temporal_residual_ratio", "file", @@ -133,23 +146,27 @@ def main( ) -> None: """Generate VAE video candidates from the prefix of SOURCE.""" logging.basicConfig(level=logging.INFO, format="%(message)s") - import torch - from diffusers import AutoencoderKL - levels = _parse_noise_levels(noise_levels) - capture = cv2.VideoCapture(str(source)) - if not capture.isOpened(): - raise click.ClickException(f"Could not open video: {source}") - width = round(capture.get(cv2.CAP_PROP_FRAME_WIDTH)) - height = round(capture.get(cv2.CAP_PROP_FRAME_HEIGHT)) - capture.release() + width, height, source_fps = _probe_video(source) size = _fit_size(width, height, long_side) frames, effective_fps = read_sampled_frames(source, duration=duration, output_fps=fps, size=size) + run = { + "source_sha256": _sha256(source), + "source_width": str(width), + "source_height": str(height), + "source_fps": f"{source_fps:.4f}", + "duration_seconds": f"{duration:.4f}", + "vae": model, + "long_side": str(long_side), + "fps": f"{effective_fps:.4f}", + "seed": str(seed), + } output_dir.mkdir(parents=True, exist_ok=True) control_path = output_dir / "control.mp4" encode_video_frames(frames, source, control_path, fps=effective_fps) rows: list[dict[str, str]] = [ { + **run, "variant": "control", "noise_std": "", "psnr_db": "inf", @@ -160,12 +177,11 @@ def main( } ] - resolved_device = _pick_device(device) - dtype = torch.float16 if resolved_device == "cuda" else torch.float32 - log.info("Loading %s on %s", model, resolved_device) - vae = AutoencoderKL.from_pretrained(model, torch_dtype=dtype).to(resolved_device) - vae.eval() - vae.enable_slicing() + # Load through the engine's own loader rather than repeating it here: the + # scaling-factor gate lives there, and the harness that produces the certified + # rows is the last place that should be exempt from it. + runtime = load_video_vae_runtime(model=model, device=device) + vae, resolved_device = runtime.vae, runtime.resolved_device log.info("Encoding source frames") latent_batches = _encode_frame_latents( @@ -173,6 +189,7 @@ def main( vae=vae, device=resolved_device, batch_size=batch_size, + scaling_factor=runtime.scaling_factor, ) first_latents = latent_batches[0] shared_noise = _shared_latent_noise( @@ -190,6 +207,7 @@ def main( vae=vae, noise_std=level, shared_noise=shared_noise, + scaling_factor=runtime.scaling_factor, ) output_path = output_dir / f"vae-noise-{level:.4f}.mp4" encode_video_frames( @@ -202,6 +220,7 @@ def main( temporal_ratio = temporal_residual_ratio(regenerated, temporal_maps, temporal_baseline) rows.append( { + **run, "variant": "vae", "noise_std": f"{level:.4f}", "psnr_db": f"{psnr:.4f}", diff --git a/src/remove_ai_watermarks/video_invisible.py b/src/remove_ai_watermarks/video_invisible.py index 7fef3a4..c1bc830 100644 --- a/src/remove_ai_watermarks/video_invisible.py +++ b/src/remove_ai_watermarks/video_invisible.py @@ -33,6 +33,7 @@ from remove_ai_watermarks.video_synthid import ( DEFAULT_VIDEO_SYNTHID_NOISE_STD, DEFAULT_VIDEO_SYNTHID_VAE, VIDEO_SYNTHID_LATENT_MULTIPLE, + VIDEO_SYNTHID_VAE_SCALING_FACTOR, ) from remove_ai_watermarks.video_temporal import ( _backward_map, @@ -70,6 +71,9 @@ class VideoVaeRuntime: requested_device: str resolved_device: str vae: Any + # The gate that validates this factor and the encode/decode calls that apply it + # must read one value, not three independent reads of the same attribute. + scaling_factor: float def is_available() -> bool: @@ -133,11 +137,22 @@ def load_video_vae_runtime( vae = AutoencoderKL.from_pretrained(model, torch_dtype=dtype).to(resolved_device) vae.eval() vae.enable_slicing() + scaling_factor = float(vae.config.scaling_factor) + log.info("Latent scaling factor %.5f", scaling_factor) + if model != DEFAULT_VIDEO_SYNTHID_VAE: + log.warning("No oracle-certified profile exists for %s; the shipped noise_std is not calibrated for it", model) + elif scaling_factor != VIDEO_SYNTHID_VAE_SCALING_FACTOR: + raise RuntimeError( + f"{model} loaded with latent scaling factor {scaling_factor}, but the certified " + f"profile is defined against {VIDEO_SYNTHID_VAE_SCALING_FACTOR}. The perturbation " + "would be rescaled and the output would no longer match any certified row." + ) return VideoVaeRuntime( model=model, requested_device=device, resolved_device=resolved_device, vae=vae, + scaling_factor=scaling_factor, ) @@ -262,12 +277,12 @@ def _encode_frame_latents( vae: Any, device: str, batch_size: int, + scaling_factor: float, ) -> list[Any]: """Encode source frames once so every candidate can reuse identical latents.""" import torch latent_batches: list[Any] = [] - scaling_factor = float(vae.config.scaling_factor) with torch.inference_mode(): for batch in _frame_batches(frames, batch_size): rgb = np.stack([frame[:, :, ::-1] for frame in batch]) @@ -284,12 +299,12 @@ def _decode_frame_latents( vae: Any, noise_std: float, shared_noise: Any, + scaling_factor: float, ) -> list[np.ndarray]: """Decode cached latents with one perturbation shared across time.""" import torch output: list[np.ndarray] = [] - scaling_factor = float(vae.config.scaling_factor) with torch.inference_mode(): for latents in latent_batches: perturbed = latents + noise_std * shared_noise.expand(latents.shape[0], -1, -1, -1) @@ -411,9 +426,18 @@ def regenerate_video_candidate( vae=vae, device=resolved_device, batch_size=batch_size, + scaling_factor=runtime.scaling_factor, ) latents = latent_batches[0] if shared_noise is None: + # Removal strength is the ratio of the perturbation to this spread, + # not noise_std alone: it is the only local quantity that makes two + # models' doses comparable. + log.info( + "First latent batch spread %.4f against noise_std %.4f", + float(latents.float().std()), + noise_std, + ) shared_noise = _shared_latent_noise( latents.shape[1:], seed=seed, @@ -425,6 +449,7 @@ def regenerate_video_candidate( vae=vae, noise_std=noise_std, shared_noise=shared_noise, + scaling_factor=runtime.scaling_factor, ) for reference, candidate in zip(frames, regenerated, strict=True): frame_pipe.write(candidate.tobytes()) diff --git a/src/remove_ai_watermarks/video_synthid.py b/src/remove_ai_watermarks/video_synthid.py index e262a2c..81a4ae3 100644 --- a/src/remove_ai_watermarks/video_synthid.py +++ b/src/remove_ai_watermarks/video_synthid.py @@ -1,6 +1,10 @@ """Shared configuration for oracle-certified video SynthID removal.""" DEFAULT_VIDEO_SYNTHID_VAE = "stabilityai/sd-vae-ft-mse" +# The certified profile is a perturbation-to-signal ratio, so it is pinned against +# this latent scaling factor as much as against noise_std. Rationale and the drift +# it guards against: docs/module-internals.md. +VIDEO_SYNTHID_VAE_SCALING_FACTOR = 0.18215 DEFAULT_VIDEO_SYNTHID_NOISE_STD = 0.15 DEFAULT_VIDEO_SYNTHID_LONG_SIDE = 512 DEFAULT_VIDEO_SYNTHID_FPS = 12.0 diff --git a/tests/test_video_invisible.py b/tests/test_video_invisible.py index 47a49d0..0f35223 100644 --- a/tests/test_video_invisible.py +++ b/tests/test_video_invisible.py @@ -2,20 +2,27 @@ from __future__ import annotations +import csv import sys import threading +from pathlib import Path from types import SimpleNamespace from typing import TYPE_CHECKING, cast import pytest from remove_ai_watermarks import optional_deps, video_encoding, video_invisible -from remove_ai_watermarks.video_synthid import DEFAULT_VIDEO_SYNTHID_NOISE_STD +from remove_ai_watermarks.video_synthid import ( + DEFAULT_VIDEO_SYNTHID_FPS, + DEFAULT_VIDEO_SYNTHID_LONG_SIDE, + DEFAULT_VIDEO_SYNTHID_NOISE_STD, +) if TYPE_CHECKING: - from pathlib import Path from typing import BinaryIO +ORACLE_MANIFEST = Path(__file__).resolve().parents[1] / "data" / "evaluations" / "video-synthid-oracle.csv" + def test_encoder_redirects_large_stderr_while_frames_are_streaming( tmp_path: Path, @@ -281,8 +288,36 @@ def test_probe_video_timestamps_uses_best_effort_pts( assert video_encoding.probe_video_timestamps(source) == (0.0, 0.041667) -def test_default_noise_matches_full_clip_oracle_floor() -> None: - assert DEFAULT_VIDEO_SYNTHID_NOISE_STD == 0.15 +def test_shipped_defaults_match_a_certified_manifest_row() -> None: + """The shipped operating point must be one the provider oracle actually cleared. + + Pinning the constant alone was not enough. Only ``noise_std`` was asserted, so + ``long_side`` and ``fps`` could move to an uncertified geometry with a green + suite -- and they are two thirds of what the oracle was shown. Reading the + manifest ties all three to the evidence: raising the resolution or the frame + rate now fails here until a ``not_detected`` row exists for that exact triple. + + The tuple stops at three fields because the model is a fourth thing the oracle + was shown and neither tracked row records it. That omission is data-driven: add + ``vae`` here in the same commit as the first row that records one. + """ + with ORACLE_MANIFEST.open(newline="", encoding="utf-8") as stream: + certified = { + (float(row["noise_std"]), int(row["long_side"]), float(row["fps"])) + for row in csv.DictReader(stream) + # The manifest deliberately leaves unrecorded fields empty, so a row + # missing part of its configuration certifies no triple and is skipped + # rather than crashing the parse. + if row["output_verdict"] == "not_detected" + and all(row[field] for field in ("noise_std", "long_side", "fps")) + } + + assert certified, f"{ORACLE_MANIFEST.name} records no fully configured certified row" + shipped = (DEFAULT_VIDEO_SYNTHID_NOISE_STD, DEFAULT_VIDEO_SYNTHID_LONG_SIDE, DEFAULT_VIDEO_SYNTHID_FPS) + assert shipped in certified, ( + f"shipped (noise_std, long_side, fps)={shipped} has no certified row in " + f"{ORACLE_MANIFEST.name}; certified: {sorted(certified)}" + ) def test_stream_batches_consumes_only_one_batch_ahead() -> None: From 613b3493582249d12d9f53714e21231638894b01 Mon Sep 17 00:00:00 2001 From: Victor Kuznetsov Date: Wed, 5 Aug 2026 22:24:49 -0700 Subject: [PATCH 2/2] Cover the fidelity probe, and run it where ffmpeg actually exists The probe shipped with no tests while every sibling research script has them, and it needs neither a model nor a GPU, so nothing justified the gap. Its whole output is a ranking and a mispaired comparison still prints a plausible number, so the pairing tests carry more weight than the metric ones. The centerpiece builds a correctly phased and a wrong-phased delivery of the same clip. Both have six frames, so the frame-count check the probe used to rely on passes for both; the misaligned one scores near 9 dB against 40+. A separate assertion pins that the probe binds the engine's sampler rather than a copy, which is the contract a same-phase private copy would otherwise satisfy silently. This file now holds the only constraint on that sampler's phase. CI ran none of this. The test matrix installs no ffmpeg, and the job that does install it ran one class from another file, so five of these tests would have skipped on every runner. The video-e2e job now collects this file too. Recorded in .claude/rules/development.md: mutating the sampler's >= to > proves nothing, because the 1e-9 epsilon makes both branches agree on exact equality. The phase or the period is what has to move. That inert mutation passed a green suite here before the real one failed five tests. Dropped the downscale test's PSNR bound: solid colors survive both the downscale and the upscale exactly, so any score there pins the local ffmpeg's chroma rounding rather than the probe. Geometry is the assertion that means something. Co-Authored-By: Claude Opus 5 --- .claude/rules/development.md | 8 + .github/workflows/test.yml | 1 + docs/module-internals.md | 7 + tests/test_video_fidelity_probe.py | 262 +++++++++++++++++++++++++++++ 4 files changed, 278 insertions(+) create mode 100644 tests/test_video_fidelity_probe.py diff --git a/.claude/rules/development.md b/.claude/rules/development.md index 3f9af56..c2a95aa 100644 --- a/.claude/rules/development.md +++ b/.claude/rules/development.md @@ -119,4 +119,12 @@ needing an oracle carrier: build a clip from a tracked fixture with ffmpeg, run engine before and after, and require an identical output sha256. Keep the generated media outside the repository. +Frame sampling is compared as `timestamp + 1e-9 >= next_sample_time`, so mutating +that `>=` to `>` changes nothing and a green suite proves nothing. Mutate the phase +or the period instead -- starting the accumulator half a period late shifts the +selection by one frame while keeping the count identical, which is the drift a +frame-count check cannot see and what +`test_pairing_follows_the_engine_sampling_rule_not_just_the_frame_count` exists to +catch. + Environment setup, dependency recovery, CI behavior, and fixture policy: [`../../docs/development.md`](../../docs/development.md). diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index a10d52f..b51c172 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -83,3 +83,4 @@ jobs: run: >- uv run pytest -vv -o faulthandler_timeout=60 tests/test_video.py::TestVideoVisibleFullClip + tests/test_video_fidelity_probe.py diff --git a/docs/module-internals.md b/docs/module-internals.md index 8d2a235..31d48ec 100644 --- a/docs/module-internals.md +++ b/docs/module-internals.md @@ -348,6 +348,13 @@ Regression coverage: properties, metadata stripping, and a large-`mdat` metadata case that rejects any full-source `read_bytes()` call. CI installs ffmpeg explicitly for this test so the integration gate cannot silently skip. +- [`test_video_fidelity_probe.py`](../tests/test_video_fidelity_probe.py), which + builds its clips from solid-color frames whose index is recoverable from the + pixels, so a decimation that keeps the frame count but shifts the phase scores + far worse instead of passing unnoticed. It also asserts that the probe binds the + engine's sampler rather than a copy of it, and it holds the suite's only + constraint on that sampler's phase. Nothing here needs a model, but it does need + ffmpeg, so CI runs this file in the same job that installs it. ## Metadata and provenance diff --git a/tests/test_video_fidelity_probe.py b/tests/test_video_fidelity_probe.py new file mode 100644 index 0000000..a976e6c --- /dev/null +++ b/tests/test_video_fidelity_probe.py @@ -0,0 +1,262 @@ +"""Regression tests for the end-to-end video fidelity probe. + +The probe needs no model and no GPU, so every behavior below is covered without a +download. Its whole output is a ranking, and a mispaired comparison still prints a +plausible number, so the pairing tests matter more than the metric ones. +""" + +from __future__ import annotations + +import importlib.util +import json +import shutil +import subprocess +import sys +from pathlib import Path +from typing import TYPE_CHECKING + +import click +import click.testing +import numpy as np +import pytest + +from remove_ai_watermarks import video_invisible + +if TYPE_CHECKING: + from collections.abc import Iterator + from types import ModuleType + +_SCRIPT = Path(__file__).parent.parent / "scripts" / "video_fidelity_probe.py" + + +@pytest.fixture(scope="module") +def probe() -> Iterator[ModuleType]: + # The script inserts scripts/ on sys.path to reach its shared SSIM helper and + # never removes it, which would otherwise outlive this module and defeat the + # restore that test_fidelity_matching.py performs for the same directory. + original_path = list(sys.path) + spec = importlib.util.spec_from_file_location("video_fidelity_probe", _SCRIPT) + assert spec is not None + assert spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + try: + yield module + finally: + sys.path[:] = original_path + + +def _ffmpeg() -> str: + ffmpeg = shutil.which("ffmpeg") + if ffmpeg is None: + pytest.skip("the fidelity probe tests need ffmpeg to build their clips") + return ffmpeg + + +def _distinct_frames(count: int, *, width: int = 64, height: int = 48) -> list[np.ndarray]: + """Solid-color frames whose index is recoverable from the pixels. + + Distinguishable frames are the point: a pairing bug against a gradient or a + static scene still scores well, which is how a reordering drift hides. + """ + frames: list[np.ndarray] = [] + for index in range(count): + # Odd multipliers, so each channel alone is injective for count <= 256. The + # per-step deltas are what set the misalignment penalty the thresholds below + # rely on; a constant offset would change neither and is left out. + color = ((index * 23) % 256, (index * 71) % 256, (index * 137) % 256) + frames.append(np.full((height, width, 3), color, dtype=np.uint8)) + return frames + + +def _write_clip(path: Path, frames: list[np.ndarray], *, fps: float, ffmpeg: str) -> None: + height, width = frames[0].shape[:2] + command = [ + ffmpeg, + "-y", + "-loglevel", + "error", + "-f", + "rawvideo", + "-pix_fmt", + "bgr24", + "-s:v", + f"{width}x{height}", + "-r", + f"{fps:.12g}", + "-i", + "pipe:0", + "-c:v", + "libx264", + "-preset", + "ultrafast", + "-crf", + "0", + "-pix_fmt", + "yuv444p", + str(path), + ] + subprocess.run( # noqa: S603 + command, + input=b"".join(frame.tobytes() for frame in frames), + capture_output=True, + check=True, + ) + + +def test_the_probe_binds_the_engine_sampler_rather_than_its_own_copy(probe: ModuleType) -> None: + """The "shared, not copied" contract, asserted at the seam that carries it. + + The behavioral test below pins the phase the sampler happens to select, which a + private copy with the same phase would also satisfy. Identity is the only thing + that fails the moment someone reimplements the rule here. It needs no ffmpeg, so + it survives on runners where the rest of this file skips. + """ + assert probe._iter_sampled_frames is video_invisible._iter_sampled_frames + assert probe._probe_video is video_invisible._probe_video + + +def test_identical_source_and_delivery_score_as_untouched(probe: ModuleType, tmp_path: Path) -> None: + ffmpeg = _ffmpeg() + clip = tmp_path / "clip.mp4" + _write_clip(clip, _distinct_frames(8), fps=12.0, ffmpeg=ffmpeg) + + row = probe._measure(clip, clip, source_geometry=probe._probe_video(clip), duration=None) + + assert row["source_psnr_db"] == float("inf") + assert row["source_ssim"] == pytest.approx(1.0) + assert row["pixel_ratio"] == 1.0 + assert row["frames"] == 8 + + +def test_pairing_follows_the_engine_sampling_rule_not_just_the_frame_count( + probe: ModuleType, + tmp_path: Path, +) -> None: + """The guard the frame-count check could not provide. + + Decimating 12 fps to 6 fps selects source frames 0, 2, 4, ... A rule that + selected 1, 3, 5, ... instead returns exactly as many frames, so a count check + passes while every comparison is against the wrong frame. Starting the sampler's + accumulator half a period late produces exactly that, and this is the only test + in the suite that constrains the sampler's phase at all. + + The clips encode losslessly, so the aligned pairing is exact and clears any + floor; the misaligned one compares frames a full colour step apart and lands + near 9 dB, which is what leaves the 25 dB ceiling a wide moat rather than a + tuned threshold. + """ + ffmpeg = _ffmpeg() + source = tmp_path / "source.mp4" + frames = _distinct_frames(12) + _write_clip(source, frames, fps=12.0, ffmpeg=ffmpeg) + + aligned = tmp_path / "aligned.mp4" + _write_clip(aligned, frames[0::2], fps=6.0, ffmpeg=ffmpeg) + shifted = tmp_path / "shifted.mp4" + _write_clip(shifted, frames[1::2], fps=6.0, ffmpeg=ffmpeg) + + geometry = probe._probe_video(source) + aligned_row = probe._measure(source, aligned, source_geometry=geometry, duration=None) + shifted_row = probe._measure(source, shifted, source_geometry=geometry, duration=None) + + assert aligned_row["frames"] == shifted_row["frames"] == 6 + assert aligned_row["source_psnr_db"] > 40.0 + assert shifted_row["source_psnr_db"] < 25.0 + + +def test_a_delivery_shorter_than_the_source_names_duration_rather_than_drift( + probe: ModuleType, + tmp_path: Path, +) -> None: + """A short delivery is a duration prefix far more often than a real drift. + + The sweep harness trims its candidates by default, so pointing the probe at one + is the common case rather than the pathological one. It has to refuse instead of + scoring a truncated pairing, and the refusal has to name the fix. + """ + ffmpeg = _ffmpeg() + source = tmp_path / "source.mp4" + frames = _distinct_frames(12) + _write_clip(source, frames, fps=12.0, ffmpeg=ffmpeg) + trimmed = tmp_path / "trimmed.mp4" + _write_clip(trimmed, frames[:6], fps=12.0, ffmpeg=ffmpeg) + + geometry = probe._probe_video(source) + with pytest.raises(click.ClickException, match="--duration"): + probe._measure(source, trimmed, source_geometry=geometry, duration=None) + + row = probe._measure(source, trimmed, source_geometry=geometry, duration=0.5) + + assert row["frames"] == 6 + assert row["source_psnr_db"] > 40.0 + + +def test_a_downscaled_delivery_is_upscaled_back_to_the_source_geometry( + probe: ModuleType, + tmp_path: Path, +) -> None: + ffmpeg = _ffmpeg() + source = tmp_path / "source.mp4" + frames = _distinct_frames(6, width=64, height=48) + _write_clip(source, frames, fps=12.0, ffmpeg=ffmpeg) + delivered = tmp_path / "small.mp4" + _write_clip(delivered, _distinct_frames(6, width=32, height=24), fps=12.0, ffmpeg=ffmpeg) + + row = probe._measure(source, delivered, source_geometry=probe._probe_video(source), duration=None) + + assert (row["width"], row["height"]) == (32, 24) + assert row["pixel_ratio"] == 0.25 + # Geometry is the whole assertion. Solid colors survive both the downscale and + # the upscale exactly, so any score here would pin the local ffmpeg's chroma + # rounding rather than probe behavior. A missing upscale fails this test through + # the shape mismatch in the metric, not through a number. + assert row["frames"] == 6 + + +def test_geometry_read_rejects_a_file_that_is_not_a_video(probe: ModuleType, tmp_path: Path) -> None: + broken = tmp_path / "broken.mp4" + broken.write_bytes(b"not a video") + + with pytest.raises(click.ClickException, match="Could not open video"): + probe._delivered_geometry(broken) + + +def test_a_single_frame_delivery_is_rejected_before_the_temporal_metric( + probe: ModuleType, + tmp_path: Path, +) -> None: + ffmpeg = _ffmpeg() + clip = tmp_path / "one.mp4" + _write_clip(clip, _distinct_frames(1), fps=12.0, ffmpeg=ffmpeg) + + with pytest.raises(click.ClickException, match="fewer than two frames"): + probe._measure(clip, clip, source_geometry=probe._probe_video(clip), duration=None) + + +def test_the_command_writes_one_json_row_per_argument_in_order(probe: ModuleType, tmp_path: Path) -> None: + """`main` is the only code the helper tests do not reach, and it has to round-trip. + + The rows exist to be read by something else, so the file has to parse. An + identical source and delivery would serialize PSNR as bare ``Infinity``, which + `json.loads` accepts but a strict reader does not; distinct deliveries keep the + numbers finite and that stays out of the tool's normal output. + """ + ffmpeg = _ffmpeg() + source = tmp_path / "source.mp4" + _write_clip(source, _distinct_frames(6), fps=12.0, ffmpeg=ffmpeg) + small = tmp_path / "small.mp4" + _write_clip(small, _distinct_frames(6, width=32, height=24), fps=12.0, ffmpeg=ffmpeg) + smaller = tmp_path / "smaller.mp4" + _write_clip(smaller, _distinct_frames(6, width=16, height=16), fps=12.0, ffmpeg=ffmpeg) + report = tmp_path / "rows.json" + + result = click.testing.CliRunner().invoke( + probe.main, + [str(source), str(small), str(smaller), "--json-out", str(report)], + ) + + assert result.exit_code == 0, result.output + rows = json.loads(report.read_text(encoding="utf-8")) + assert [row["file"] for row in rows] == ["small.mp4", "smaller.mp4"] + assert all(row["source"] == "source.mp4" for row in rows)