Files
remove-ai-watermarks/tests/test_noai.py
T
Victor KuznetsovandClaude Opus 4.8 a4c901ff39 fix: metadata-strip parity, input robustness, and detection/clash coverage
Bug fixes (each with a regression test):
- metadata strip parity across every marker placement: IPTC digitalSourceType
  in XMP, the Samsung post-EOI trailer, the China TC260 AIGC block in EXIF
  UserComment, a bare AIGC block in a non-standard APP segment, and the ISOBMFF
  EXIF path (AIGC + xAI) are all now stripped -- anything a scanner flags, the
  strip reaches
- Samsung genAIType detected when its trailer sits past the 512 KB scan window
  (file-tail read on large photos)
- crashes on edge inputs: Gemini detector on images with a short side < 16px,
  footprint_mask on a zero-size ndarray, the humanizer on chromatic_shift >=
  width, and the CLI on unreadable/corrupt/empty input (clean error, not a
  traceback)
- WebP written losslessly (cv2 quality 101), not lossy at 100
- the IPTC digitalSourceType algorithmicMedia (procedural, not trained on
  sampled data) is no longer flagged as AI-generated, so clean procedural
  content is not scrubbed
- c2pa source-type: compositeWithTrainedAlgorithmicMedia is checked before the
  bare algorithmicMedia token, so an AI-enhanced composite is not misclassified

Detection:
- integrity-clash coverage now normalizes ByteDance / Canva / ElevenLabs /
  Black Forest Labs, so a transplanted manifest next to an independent
  conflicting stamp is caught; the generic China TC260 AIGC label is attributed
  to a co-present TC260 vendor, so a legit Doubao image (its own C2PA + TC260
  label) does not clash (corpus-validated: 0 new clashes on 5000 carriers)

CLI:
- batch exits non-zero (with a warning) when any image errors or a GPU-missing
  SynthID scrub is skipped, and copies the input through so the output dir stays
  complete -- it used to always exit 0 and could silently drop files

Perf:
- GeminiEngine reused as a process-wide singleton with a precomputed template
  ladder: -24% on the identify sparkle path, detection byte-identical

Internal: one shared _ai_exif_targets rule set feeds both EXIF scrubbers so
their coverage cannot drift; docs synced; maintain.sh hardened so the uv-secure
internal teardown crash no longer aborts the gate (still fails on a real finding).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-13 10:49:24 +03:00

567 lines
24 KiB
Python

"""Tests for vendored noai submodules: constants, extractor, c2pa, plus the
consolidated metadata strip (formerly noai.cleaner)."""
from __future__ import annotations
import struct
from pathlib import Path
import pytest
from remove_ai_watermarks.metadata import (
remove_ai_metadata as noai_remove_ai_metadata,
)
from remove_ai_watermarks.noai.c2pa import (
_parse_c2pa_chunk,
cbor_text_after,
extract_c2pa_chunk,
extract_c2pa_info,
has_c2pa_metadata,
inject_c2pa_chunk,
synthid_verdict,
)
from remove_ai_watermarks.noai.constants import (
AI_KEYWORDS,
AI_METADATA_KEYS,
C2PA_CHUNK_TYPE,
PNG_SIGNATURE,
SUPPORTED_FORMATS,
)
from remove_ai_watermarks.noai.extractor import (
extract_ai_metadata,
extract_metadata,
get_ai_metadata_summary,
has_ai_metadata,
)
from remove_ai_watermarks.noai.isobmff import (
blank_ai_exif_tokens,
is_isobmff,
strip_c2pa_boxes,
)
# ── Constants ───────────────────────────────────────────────────────
class TestConstants:
"""Verify constant integrity."""
def test_supported_formats_include_png(self):
assert ".png" in SUPPORTED_FORMATS
def test_supported_formats_include_jpg(self):
assert ".jpg" in SUPPORTED_FORMATS
def test_supported_formats_include_heic_avif(self):
# HEIC/AVIF are first-class on the pixel path now (read+write via pillow-heif),
# so batch discovers them and the CLI does not warn.
assert {".heic", ".heif", ".avif"} <= SUPPORTED_FORMATS
def test_supported_formats_exclude_jpeg_xl(self):
# JPEG-XL stays metadata/strip-only -- no pixel decoder without pillow-jxl.
assert ".jxl" not in SUPPORTED_FORMATS
def test_ai_metadata_keys_not_empty(self):
assert len(AI_METADATA_KEYS) > 0
def test_ai_keywords_not_empty(self):
assert len(AI_KEYWORDS) > 0
def test_png_signature_bytes(self):
assert PNG_SIGNATURE == b"\x89PNG\r\n\x1a\n"
def test_c2pa_chunk_type(self):
assert C2PA_CHUNK_TYPE == b"caBX"
# ── Extractor ───────────────────────────────────────────────────────
class TestExtractor:
"""Tests for noai.extractor functions."""
def test_extract_metadata_returns_dict(self, tmp_clean_png):
meta = extract_metadata(tmp_clean_png)
assert isinstance(meta, dict)
def test_extract_metadata_gets_standard_keys(self, tmp_clean_png):
meta = extract_metadata(tmp_clean_png)
assert "Author" in meta
def test_extract_ai_metadata_from_ai_image(self, tmp_png_with_ai_metadata):
meta = extract_ai_metadata(tmp_png_with_ai_metadata)
assert "parameters" in meta
def test_extract_ai_metadata_from_clean_image(self, tmp_clean_png):
meta = extract_ai_metadata(tmp_clean_png)
assert len(meta) == 0
def test_has_ai_metadata_detects(self, tmp_png_with_ai_metadata):
assert has_ai_metadata(tmp_png_with_ai_metadata)
def test_has_ai_metadata_clean(self, tmp_clean_png):
assert not has_ai_metadata(tmp_clean_png)
def test_summary_with_ai(self, tmp_png_with_ai_metadata):
summary = get_ai_metadata_summary(tmp_png_with_ai_metadata)
assert "AI Image Metadata" in summary
def test_summary_clean(self, tmp_clean_png):
summary = get_ai_metadata_summary(tmp_clean_png)
assert "No AI metadata" in summary
# ── Cleaner ─────────────────────────────────────────────────────────
class TestCleaner:
"""Metadata stripping via the single, consolidated ``metadata.remove_ai_metadata``
(the legacy ``noai.cleaner`` duplicate was retired)."""
def test_remove_ai_metadata(self, tmp_png_with_ai_metadata, tmp_path):
output = tmp_path / "cleaned.png"
noai_remove_ai_metadata(tmp_png_with_ai_metadata, output)
assert output.exists()
# Verify AI metadata removed
meta = extract_ai_metadata(output)
assert "parameters" not in meta
def test_has_ai_content(self, tmp_png_with_ai_metadata):
assert has_ai_metadata(tmp_png_with_ai_metadata)
# ── C2PA ────────────────────────────────────────────────────────────
class TestC2PA:
"""Tests for C2PA detection on regular (non-C2PA) images."""
def test_no_c2pa_on_regular_png(self, tmp_clean_png):
assert not has_c2pa_metadata(tmp_clean_png)
def test_no_c2pa_on_jpeg(self, tmp_jpeg_path):
assert not has_c2pa_metadata(tmp_jpeg_path)
def test_extract_c2pa_none_on_regular(self, tmp_clean_png):
assert extract_c2pa_chunk(tmp_clean_png) is None
def test_extract_c2pa_info_empty(self, tmp_clean_png):
info = extract_c2pa_info(tmp_clean_png)
assert info == {}
def test_c2pa_returns_false_for_non_png(self, tmp_jpeg_path):
assert not has_c2pa_metadata(tmp_jpeg_path)
SAMPLES_DIR = Path(__file__).resolve().parent.parent / "data" / "samples"
@pytest.mark.skipif(not SAMPLES_DIR.exists(), reason="data/samples not present")
class TestC2PARealSamples:
"""Parser behavior on real committed C2PA images."""
def test_detects_c2pa_in_openai_png(self):
assert has_c2pa_metadata(SAMPLES_DIR / "chatgpt-1.png")
def test_extract_info_openai_fields(self):
info = extract_c2pa_info(SAMPLES_DIR / "chatgpt-1.png")
assert info["has_c2pa"] is True
assert "OpenAI" in info["issuer"]
assert "c2pa_manifest" in info # "C2PA manifest (N bytes)"
assert "trainedAlgorithmicMedia" in info["source_type"]
# CBOR-clean claim generator, no regex artifacts (e.g. "fGPT-4o").
assert info["claim_generator"]
assert not info["claim_generator"].startswith("f")
assert "synthid_watermark" in info
def test_extract_info_adobe_has_no_synthid(self):
info = extract_c2pa_info(SAMPLES_DIR / "firefly-1.png")
assert "Adobe" in info["issuer"]
assert "synthid_watermark" not in info
def test_extract_chunk_returns_bytes(self):
chunk = extract_c2pa_chunk(SAMPLES_DIR / "chatgpt-1.png")
assert chunk is not None
assert chunk[4:8] == b"caBX" # chunk type in the 8-byte header
def test_inject_round_trip(self, tmp_clean_png, tmp_path):
"""Extract a real C2PA chunk, inject into a clean PNG, re-detect."""
chunk = extract_c2pa_chunk(SAMPLES_DIR / "chatgpt-1.png")
out = tmp_path / "injected.png"
inject_c2pa_chunk(tmp_clean_png, out, chunk)
assert has_c2pa_metadata(out)
assert "OpenAI" in extract_c2pa_info(out)["issuer"]
def test_extract_info_flux_jpeg_via_reader(self):
"""Real committed JPEG-with-C2PA fixture: the non-PNG reader path works."""
info = extract_c2pa_info(SAMPLES_DIR / "flux-1.jpg")
assert info["has_c2pa"] is True
assert info["c2pa_manifest"].startswith("C2PA manifest store") # reader, not chunk
assert "Black Forest Labs" in info["issuer"]
assert "trainedAlgorithmicMedia" in info["source_type"]
def test_extract_info_uses_reader_store(self):
"""The c2pa-python reader path: structured (not heuristic) extraction."""
from remove_ai_watermarks.noai import c2pa
assert c2pa.reader_available()
info = extract_c2pa_info(SAMPLES_DIR / "chatgpt-1.png")
# The store-JSON label proves the reader path served this, not the
# caBX-chunk fallback ("C2PA manifest (...)").
assert info["c2pa_manifest"].startswith("C2PA manifest store")
# Structured claim generator is exact, not a CBOR-scanned best-effort.
assert info["claim_generator"] == "ChatGPT"
def test_fallback_to_png_parser_when_reader_unavailable(self, monkeypatch):
"""With the reader disabled, the hand-rolled PNG parser still works."""
from remove_ai_watermarks.noai import c2pa
monkeypatch.setattr(c2pa, "_C2PA_READER_AVAILABLE", False)
info = extract_c2pa_info(SAMPLES_DIR / "chatgpt-1.png")
assert info["c2pa_manifest"].startswith("C2PA manifest (") # chunk path
assert "OpenAI" in info["issuer"]
assert "trainedAlgorithmicMedia" in info["source_type"]
assert "synthid_watermark" in info
class TestC2PAInjectValidation:
def test_inject_rejects_non_png(self, tmp_path):
with pytest.raises(ValueError, match="only supported for PNG"):
inject_c2pa_chunk(tmp_path / "in.jpg", tmp_path / "out.png", b"")
# ── CBOR text extraction (parser internals) ─────────────────────────
class TestCborTextAfter:
"""cbor_text_after handles the three CBOR text-string length prefixes."""
def test_direct_length(self):
# major-type 3, direct length (0x60 + len). "abc" -> 0x63.
payload = b"name" + bytes([0x63]) + b"abc"
assert cbor_text_after(payload, b"name") == "abc"
def test_one_byte_length(self):
s = b"x" * 30
payload = b"name" + bytes([0x78, 30]) + s
assert cbor_text_after(payload, b"name") == "x" * 30
def test_two_byte_length(self):
s = b"y" * 300
payload = b"name" + bytes([0x79]) + struct.pack(">H", 300) + s
assert cbor_text_after(payload, b"name") == "y" * 300
def test_key_not_found_returns_none(self):
assert cbor_text_after(b"nothing here", b"name") is None
def test_key_at_end_returns_none(self):
assert cbor_text_after(b"prefixname", b"name") is None
def test_invalid_head_returns_none(self):
# 0x00 is not a text-string head.
assert cbor_text_after(b"name" + bytes([0x00]) + b"abc", b"name") is None
def test_latin1_fallback_on_invalid_utf8(self):
payload = b"name" + bytes([0x61]) + b"\xff" # len 1, invalid utf-8
assert cbor_text_after(payload, b"name") is not None
class TestSynthIDVerdict:
def test_format(self):
assert synthid_verdict("OpenAI") == "likely present (OpenAI embeds SynthID with C2PA)"
def test_multiple_vendors(self):
assert "Google LLC, OpenAI" in synthid_verdict("Google LLC, OpenAI")
class TestParseChunkGuards:
"""_parse_c2pa_chunk rejects non-printable claim_generator garbage.
On some manifests (observed: Microsoft Designer) the first ``name`` key
precedes a binary hash field, not the generator string. The clean issuer +
SynthID verdict must still come through.
"""
def test_clean_generator_kept(self):
# "name" + CBOR text-string (head 0x69 = 0x60+9) "gpt-image"
chunk = b"...name" + bytes([0x69]) + b"gpt-image" + b"OpenAI trainedAlgorithmicMedia"
info: dict = {}
_parse_c2pa_chunk(chunk, info)
assert info["claim_generator"] == "gpt-image"
assert "OpenAI" in info["issuer"]
assert "synthid_watermark" in info # OpenAI + trainedAlgorithmicMedia
def test_nonprintable_generator_dropped(self):
# "name" + CBOR string (head 0x64 = len 4) with a control byte -> garbage
chunk = b"...name" + bytes([0x64]) + b"\x81abc" + b"OpenAI trainedAlgorithmicMedia"
info: dict = {}
_parse_c2pa_chunk(chunk, info)
assert "claim_generator" not in info # control-char garbage rejected
assert "OpenAI" in info["issuer"] # issuer byte-search still robust
class TestC2PADigitalSourceType:
"""The three IPTC digitalSourceType variants drive the AI verdict.
Only *trained* and *composite-with-trained* mean AI-generated (and so imply
a SynthID proxy for a SynthID vendor); plain ``algorithmicMedia`` is
procedural (not trained) and must NOT be flagged as AI.
"""
def test_plain_algorithmic_media_not_flagged_ai(self):
chunk = b"...name" + bytes([0x69]) + b"some-tool" + b" OpenAI algorithmicMedia"
info: dict = {}
_parse_c2pa_chunk(chunk, info)
assert info["source_type"] == "algorithmicMedia"
assert "synthid_watermark" not in info # procedural, not AI-generated
def test_composite_with_trained_is_ai_and_synthid(self):
chunk = b"...name" + bytes([0x69]) + b"some-tool" + b" OpenAI compositeWithTrainedAlgorithmicMedia"
info: dict = {}
_parse_c2pa_chunk(chunk, info)
assert "compositeWithTrainedAlgorithmicMedia" in info["source_type"]
assert "synthid_watermark" in info # AI-enhanced + OpenAI issuer
def test_composite_and_bare_algorithmic_cooccur_is_ai(self):
"""Regression: a manifest carrying BOTH ``compositeWithTrainedAlgorithmicMedia``
(AI-enhanced) and a bare procedural ``algorithmicMedia`` token must classify as
AI-enhanced. Before the reorder the bare-token elif fired first and returned
non-AI, dropping the composite AI signal (a false negative)."""
from remove_ai_watermarks.noai.c2pa import _populate_registry_fields
info: dict = {}
_populate_registry_fields(b"x compositeWithTrainedAlgorithmicMedia x algorithmicMedia x", info)
assert info.get("ai_source_kind") == "enhanced"
assert "compositeWithTrainedAlgorithmicMedia" in info["source_type"]
# ── ISOBMFF (AVIF / HEIF / JPEG-XL container stripping) ──────────────
FTYP = b"\x00\x00\x00\x18ftypavif\x00\x00\x00\x00avifmif1" # 24-byte ftyp box
class TestISOBMFF:
def test_is_isobmff_true(self):
assert is_isobmff(FTYP)
def test_is_isobmff_false_for_png(self):
assert not is_isobmff(b"\x89PNG\r\n\x1a\n\x00\x00")
def test_is_isobmff_false_for_short(self):
assert not is_isobmff(b"abc")
def test_strips_jpegxl_jumb_box(self):
"""JPEG-XL stores JUMBF in a ``jumb`` box, always stripped."""
jumb = struct.pack(">I", 8 + 5) + b"jumb" + b"hello"
cleaned, stripped = strip_c2pa_boxes(FTYP + jumb)
assert stripped == 1
assert cleaned == FTYP
def test_keeps_non_c2pa_box_with_64bit_size(self):
"""size==1 means a 64-bit largesize follows; non-C2PA box is kept."""
payload = b"\x00" * 8
box = b"\x00\x00\x00\x01" + b"free" + struct.pack(">Q", 16 + len(payload)) + payload
cleaned, stripped = strip_c2pa_boxes(FTYP + box)
assert stripped == 0
assert cleaned == FTYP + box
def test_malformed_box_does_not_crash(self):
# A box claiming size 4 (< 8-byte header) must terminate iteration safely.
cleaned, stripped = strip_c2pa_boxes(FTYP + b"\x00\x00\x00\x04XXXX")
assert stripped == 0
assert cleaned.startswith(FTYP)
def test_size_zero_box_runs_to_eof(self):
# size32==0 means the box extends to EOF; a non-C2PA box round-trips.
box = struct.pack(">I", 0) + b"free" + b"\x00\x00\x00\x00"
cleaned, stripped = strip_c2pa_boxes(FTYP + box)
assert stripped == 0
assert cleaned == FTYP + box
def test_truncated_largesize_terminates_safely(self):
# size32==1 promises a 64-bit largesize, but the box ends after 8 bytes;
# iteration must stop rather than read the missing largesize past EOF.
# The walk halts before EOF, so the fail-safe returns the input unchanged
# (emitting only FTYP would silently truncate the file).
data = FTYP + b"\x00\x00\x00\x01uuid"
cleaned, stripped = strip_c2pa_boxes(data)
assert stripped == 0
assert cleaned == data
@staticmethod
def _avif_with_exif(exif_0th: dict) -> bytes:
"""A fake AVIF (ftyp + mdat) whose mdat carries an EXIF TIFF block, as a
HEIF/AVIF ``Exif`` meta-box item stores it (bytes in mdat)."""
import piexif
blob = piexif.dump({"0th": exif_0th})
mdat = struct.pack(">I", 8 + len(blob)) + b"mdat" + blob
return FTYP + mdat
def test_blank_ai_token_in_exif_item(self):
import piexif
data = self._avif_with_exif({piexif.ImageIFD.Software: b"DALL-E", piexif.ImageIFD.Make: b"Canon"})
out, blanked = blank_ai_exif_tokens(data)
assert blanked == 1
assert len(out) == len(data) # same length -> box sizes / iloc stay valid
assert b"DALL-E" not in out # AI token destroyed
assert b"Canon" in out # camera tag preserved
# The TIFF structure still parses, with the AI value blanked and Make kept.
blob = out[out.index(b"Exif\x00\x00") + 6 :]
ifd = piexif.load(blob)["0th"]
assert ifd[piexif.ImageIFD.Software].strip() == b""
assert ifd[piexif.ImageIFD.Make] == b"Canon"
def test_blank_aigc_block_in_exif(self):
"""Parity with the JPEG path: the China TC260 ``{"AIGC":{...}}`` block in EXIF
ImageDescription must be blanked on the ISOBMFF path too -- ``blank_ai_exif_tokens``
is the ONLY EXIF scrubber for HEIC/AVIF (``_scrub_ai_exif`` never runs there)."""
import piexif
aigc = b'{"AIGC":{"Label":"1","ContentProducer":"00119144030008867405X210002","ProduceID":"abc"}}'
data = self._avif_with_exif({piexif.ImageIFD.ImageDescription: aigc, piexif.ImageIFD.Make: b"Canon"})
out, blanked = blank_ai_exif_tokens(data)
assert blanked >= 1
assert len(out) == len(data) # same length -> box sizes / iloc stay valid
assert b'"AIGC"' not in out # TC260 block destroyed
assert b"Canon" in out # camera tag preserved
def test_blank_xai_signature_pair_in_exif(self):
"""Parity: the xAI/Grok ``Signature:`` blob + UUID ``Artist`` pair in EXIF is
dropped together on the ISOBMFF path too."""
import piexif
sig = b"Signature: " + b"A" * 80
art = b"12345678-1234-1234-1234-123456789012"
data = self._avif_with_exif({piexif.ImageIFD.ImageDescription: sig, piexif.ImageIFD.Artist: art})
out, blanked = blank_ai_exif_tokens(data)
assert blanked == 2 # both the signature and the UUID artist
assert len(out) == len(data)
assert b"Signature: AAAA" not in out
def test_blank_leaves_clean_exif_untouched(self):
import piexif
data = self._avif_with_exif({piexif.ImageIFD.Software: b"Adobe Photoshop", piexif.ImageIFD.Make: b"NIKON"})
out, blanked = blank_ai_exif_tokens(data)
assert blanked == 0
assert out == data # no AI token -> byte-for-byte unchanged
def test_blank_no_exif_is_noop(self):
out, blanked = blank_ai_exif_tokens(FTYP + b"\x00\x00\x00\x0cmdat" + b"pixels!!")
assert blanked == 0
assert out == FTYP + b"\x00\x00\x00\x0cmdat" + b"pixels!!"
class TestIterTopLevelBoxes:
"""The box walker's three size encodings and its underflow/overflow guards."""
def test_64bit_largesize(self):
from remove_ai_watermarks.noai.isobmff import _iter_top_level_boxes
# size32 == 1 -> a 64-bit largesize follows the type; total box length = 24.
box = struct.pack(">I", 1) + b"uuid" + struct.pack(">Q", 24) + b"payload!"
boxes = list(_iter_top_level_boxes(box))
assert len(boxes) == 1
start, end, btype, payload_off = boxes[0]
assert (start, end, btype, payload_off) == (0, 24, b"uuid", 16)
def test_size0_runs_to_eof(self):
from remove_ai_watermarks.noai.isobmff import _iter_top_level_boxes
box = struct.pack(">I", 0) + b"mdat" + b"tail-to-eof"
boxes = list(_iter_top_level_boxes(box))
assert len(boxes) == 1
start, end, btype, payload_off = boxes[0]
assert (start, end, btype, payload_off) == (0, len(box), b"mdat", 8)
def test_underflow_size_stops_safely(self):
from remove_ai_watermarks.noai.isobmff import _iter_top_level_boxes
# size (4) < the 8-byte header -> the guard returns without yielding a box.
assert list(_iter_top_level_boxes(struct.pack(">I", 4) + b"ftyp" + b"more")) == []
def test_overflow_size_stops_safely(self):
from remove_ai_watermarks.noai.isobmff import _iter_top_level_boxes
# size claims 999 but the buffer is far shorter -> guard returns, no partial box.
assert list(_iter_top_level_boxes(struct.pack(">I", 999) + b"uuid" + b"x")) == []
class TestBlankAiXmpPackets:
"""XMP-packet blanking: same-length overwrite only for AI-marked packets, and only
when the packet is fully delimited."""
AIMARK = b"trainedAlgorithmicMedia"
def test_ai_packet_blanked_same_length(self):
from remove_ai_watermarks.noai.isobmff import blank_ai_xmp_packets
packet = b'<?xpacket begin="x"?><x:xmpmeta>' + self.AIMARK + b'</x:xmpmeta><?xpacket end="w"?>'
data = b"boxhdr" + packet + b"tail"
out, n = blank_ai_xmp_packets(data)
assert n == 1
assert len(out) == len(data) # same length -> iloc offsets stay valid
assert self.AIMARK not in out
assert b"boxhdr" in out
assert b"tail" in out
def test_clean_packet_left_intact(self):
from remove_ai_watermarks.noai.isobmff import blank_ai_xmp_packets
packet = b'<?xpacket begin="x"?><x:xmpmeta>plain copyright</x:xmpmeta><?xpacket end="w"?>'
out, n = blank_ai_xmp_packets(packet)
assert n == 0
assert out == packet
def test_missing_end_delimiter_not_blanked(self):
from remove_ai_watermarks.noai.isobmff import blank_ai_xmp_packets
# No <?xpacket end?> -> the packet regex cannot match, so it is left unchanged.
data = b'<?xpacket begin="x"?><x:xmpmeta>' + self.AIMARK + b"</x:xmpmeta>"
out, n = blank_ai_xmp_packets(data)
assert n == 0
assert out == data
class TestC2paBufferScans:
"""The shared buffer-scan helpers (used by both the PNG caBX parser and the
format-agnostic binary scan). Data-driven off the registries so they stay valid
as vendors are added."""
def test_soft_binding_vendors_in(self):
from remove_ai_watermarks.noai.c2pa import C2PA_SOFT_BINDINGS, soft_binding_vendors_in
sig, name = next(iter(C2PA_SOFT_BINDINGS.items()))
assert name in soft_binding_vendors_in(b"...manifest..." + sig + b"...tail...")
assert soft_binding_vendors_in(b"") == []
assert soft_binding_vendors_in(b"no soft-binding assertion here") == []
def test_synthid_vendors_in_requires_synthid_issuer(self):
from remove_ai_watermarks.noai.c2pa import C2PA_ISSUERS, SYNTHID_C2PA_ISSUERS, synthid_vendors_in
syn_sig = next(s for s in C2PA_ISSUERS if s in SYNTHID_C2PA_ISSUERS)
non_sig = next(s for s in C2PA_ISSUERS if s not in SYNTHID_C2PA_ISSUERS)
assert C2PA_ISSUERS[syn_sig] in synthid_vendors_in(b"x" + syn_sig + b"x")
# an issuer that does NOT pair SynthID with C2PA must not be reported as one
assert C2PA_ISSUERS[non_sig] not in synthid_vendors_in(b"x" + non_sig + b"x")
def test_synthid_verdict_format(self):
from remove_ai_watermarks.noai.c2pa import synthid_verdict
assert synthid_verdict("Google LLC") == "likely present (Google LLC embeds SynthID with C2PA)"
class TestC2PAInvalidSignature:
"""A .png file that is not actually PNG-signed must read as clean, not crash."""
def test_has_c2pa_false_for_non_png_bytes(self, tmp_path: Path):
fake = tmp_path / "fake.png"
fake.write_bytes(b"\xff\xd8\xff\xe0 not a png at all, just garbage bytes")
assert has_c2pa_metadata(fake) is False
def test_extract_chunk_none_for_non_png_bytes(self, tmp_path: Path):
fake = tmp_path / "fake.png"
fake.write_bytes(b"\xff\xd8\xff\xe0 not a png at all, just garbage bytes")
assert extract_c2pa_chunk(fake) is None