Merge remote-tracking branch 'origin/main' into research/video-synthid-quality-groundwork

This commit is contained in:
Victor Kuznetsov
2026-08-05 21:46:19 -07:00
32 changed files with 3669 additions and 1241 deletions
+1 -1
View File
@@ -32,7 +32,7 @@ _os.environ.setdefault("TRANSFORMERS_VERBOSITY", "error")
_warnings.filterwarnings("ignore", message=r".*ImageProcessorFast.*")
__version__ = "0.25.0"
__version__ = "0.26.0"
__all__ = [
"BatchSummary",
+18 -2
View File
@@ -29,7 +29,9 @@ if TYPE_CHECKING:
from typing import BinaryIO
_C2paReader: Any = None
_C2paError: Any = None
with contextlib.suppress(Exception):
from c2pa import C2paError as _C2paError # pyright: ignore[reportMissingTypeStubs]
from c2pa import Reader as _C2paReader # pyright: ignore[reportMissingTypeStubs]
_C2PA_READER_AVAILABLE = _C2paReader is not None
@@ -48,10 +50,22 @@ def reader_available() -> bool:
def _manifest_json_uncached(path: str) -> str | None:
"""The manifest store as JSON, or None when this file has no readable manifest.
Two outcomes are routine and stay at debug: a file with no manifest (``try_create``
returns None) and a container the reader does not support. ANY other failure is
logged at warning, because the caller cannot tell the difference from the return
value and the consequence is severe: the verdict silently falls back to the raw
byte scan and can lose a high-confidence signal. The log line preserves the
diagnostic context needed to investigate an intermittent reader failure.
"""
try:
reader = _C2paReader.try_create(path)
except _C2paError.NotSupported as error:
logger.debug("C2PA reader does not support %s: %s", path, error)
return None
except Exception as error:
logger.debug("C2PA reader rejected %s: %s", path, error)
logger.warning("C2PA reader failed to open %s: %s: %s", path, type(error).__name__, error)
return None
if reader is None:
return None
@@ -59,7 +73,9 @@ def _manifest_json_uncached(path: str) -> str | None:
with reader:
return cast("str", reader.json())
except Exception as error:
logger.debug("C2PA reader could not serialize %s: %s", path, error)
# The reader opened the file, so a manifest is there; failing to serialize it
# is never routine.
logger.warning("C2PA reader could not serialize %s: %s: %s", path, type(error).__name__, error)
return None
@@ -20,6 +20,9 @@ AI_KEYWORDS = _tokens(
PNG_SIGNATURE = b"\x89PNG\r\n\x1a\n"
C2PA_CHUNK_TYPE = b"caBX"
PNG_METADATA_CHUNKS = frozenset({b"tEXt", b"iTXt", b"zTXt", b"eXIf", b"iCCP"})
RIFF_METADATA_CHUNKS = frozenset({b"EXIF", b"XMP ", b"ICCP", b"C2PA"})
RIFF_CODED_IMAGE_CHUNKS = frozenset({b"VP8 ", b"VP8L", b"ALPH", b"ANMF"})
C2PA_SIGNATURES = tuple(
token.encode() for token in _tokens("c2pa|C2PA|jumb|jumd|JUMBF|jumbf|cbor|contentcreds|digid|assertions|manifest")
)
@@ -64,7 +64,7 @@ _AI_LABEL_MARKERS: tuple[bytes, ...] = AIGC_MARKERS + IPTC_AI_MARKERS + IPTC_AI_
# blanked in place (see ``blank_ai_xmp_packets``).
_XMP_PACKET_RE = re.compile(rb"<\?xpacket begin=.*?<\?xpacket end=[^>]*?\?>", re.DOTALL)
_STREAM_COPY_BYTES = 1024 * 1024
_STREAM_SCAN_BYTES = 4 * 1024 * 1024
STREAM_SCAN_BYTES = 4 * 1024 * 1024
# TC260-PG-20257A stores an MP4/MOV label as an ``AIGC`` key in
@@ -133,7 +133,7 @@ def _read_box_header(
return end, box_type, payload_off
def _iter_file_boxes(
def iter_file_boxes(
stream: BinaryIO,
start: int,
end: int,
@@ -193,17 +193,17 @@ def _tc260_aigc_regions(
Each tuple is ``(key_start, key_end, value_start, value_end, value)``.
"""
regions: list[tuple[int, int, int, int, bytes]] = []
for _moov_start, moov_end, moov_type, moov_payload in _iter_file_boxes(stream, 0, file_size):
for _moov_start, moov_end, moov_type, moov_payload in iter_file_boxes(stream, 0, file_size):
if moov_type != b"moov":
continue
for _udta_start, udta_end, udta_type, udta_payload in _iter_file_boxes(
for _udta_start, udta_end, udta_type, udta_payload in iter_file_boxes(
stream,
moov_payload,
moov_end,
):
if udta_type != b"udta":
continue
for _meta_start, meta_end, meta_type, meta_payload in _iter_file_boxes(
for _meta_start, meta_end, meta_type, meta_payload in iter_file_boxes(
stream,
udta_payload,
udta_end,
@@ -212,7 +212,7 @@ def _tc260_aigc_regions(
continue
keys: dict[int, tuple[int, int]] = {}
ilst_boxes: list[tuple[int, int]] = []
for _child_start, child_end, child_type, child_payload in _iter_file_boxes(
for _child_start, child_end, child_type, child_payload in iter_file_boxes(
stream,
meta_payload + 4,
meta_end,
@@ -224,7 +224,7 @@ def _tc260_aigc_regions(
if not keys:
continue
for ilst_payload, ilst_end in ilst_boxes:
for _item_start, item_end, item_type, item_payload in _iter_file_boxes(
for _item_start, item_end, item_type, item_payload in iter_file_boxes(
stream,
ilst_payload,
ilst_end,
@@ -233,7 +233,7 @@ def _tc260_aigc_regions(
key_span = keys.get(index)
if key_span is None:
continue
for _data_start, data_end, data_type, data_payload in _iter_file_boxes(
for _data_start, data_end, data_type, data_payload in iter_file_boxes(
stream,
item_payload,
item_end,
@@ -425,7 +425,7 @@ def strip_isobmff_media_file(
source: str | Path,
output: str | Path,
*,
max_box_scan: int = _STREAM_SCAN_BYTES,
max_box_scan: int = STREAM_SCAN_BYTES,
) -> tuple[int, int]:
"""Stream-copy an MP4/MOV/M4A while removing supported AI metadata.
@@ -0,0 +1,16 @@
"""Shared validation for versioned JSON transport contracts."""
from collections.abc import Collection
def require_schema_version(
value: object,
*,
contract: str,
supported: Collection[int],
) -> int:
"""Return an explicitly supported integer schema version or raise."""
if type(value) is not int or value not in supported:
versions = ", ".join(str(version) for version in sorted(supported))
raise ValueError(f"Unsupported {contract} schema: {value!r}; supported versions: {versions}")
return value
@@ -0,0 +1,861 @@
"""Collect JSON-safe metadata and container forensics for one media file.
The collector is deliberately evidence-only: it preserves raw EXIF, IPTC, C2PA,
container metadata, encoder structure, hashes, timestamps, and bounded binary
payloads without deciding whether the content is AI-generated. Provenance verdicts
and pixel statistics are separate library stages.
"""
import base64
import contextlib
import hashlib
import io
import json
import os
import plistlib
import re
import struct
import zlib
from pathlib import Path
from typing import Any, cast
import piexif
from PIL import Image
from PIL.IptcImagePlugin import getiptcinfo
from remove_ai_watermarks import image_io
from remove_ai_watermarks._internal.constants import (
PNG_METADATA_CHUNKS,
RIFF_CODED_IMAGE_CHUNKS,
RIFF_METADATA_CHUNKS,
)
from remove_ai_watermarks._internal.isobmff import (
C2PA_BOX_TYPES,
STREAM_SCAN_BYTES,
iter_file_boxes,
)
from remove_ai_watermarks._internal.schema import require_schema_version
from remove_ai_watermarks.metadata import QUICK_SCAN_BYTES
from remove_ai_watermarks.metadata_record import HEAD_WINDOW
__all__ = [
"FORENSIC_METADATA_RECORD_TYPE",
"FORENSIC_METADATA_SCHEMA_VERSION",
"SUPPORTED_EXTENSIONS",
"collect_forensic_metadata",
]
SUPPORTED_EXTENSIONS = {
".png",
".jpg",
".jpeg",
".webp",
".heic",
".heif",
".avif",
".tif",
".tiff",
".bmp",
".gif",
# video/px containers: no pixel decode, but C2PA reads them (Sora/Veo
# carry C2PA manifests) and the byte scans still apply
".mp4",
".mov",
".m4v",
".jxl",
}
FORENSIC_METADATA_SCHEMA_VERSION = 1
FORENSIC_METADATA_RECORD_TYPE = "forensic_metadata"
_B64_CAP = 1 << 20 # 1 MB safety ceiling per embedded blob
_TEXT_CAP = 1 << 20 # decoded PNG text ceiling per chunk
# Preserve enough top-level ISOBMFF uuid/jumb payload data for downstream
# provenance algorithms without requiring them to reopen the source file.
_PROVENANCE_B64_CAP = STREAM_SCAN_BYTES
_RAW_SCAN_HEAD = HEAD_WINDOW
_RAW_SCAN_TAIL = QUICK_SCAN_BYTES
def _safe_str(v: Any) -> str:
try:
return str(v)
except Exception:
return repr(v)
def _b64(b: bytes, *, cap: int = _B64_CAP) -> str:
"""Legacy base64 value, with an explicit marker when the payload is capped."""
encoded = base64.b64encode(b[:cap]).decode("ascii")
return encoded + f"...TRUNCATED({len(b)} bytes total)" if len(b) > cap else encoded
def _decode_exif_value(v: Any) -> Any:
"""Make a piexif value JSON-safe; bytes are kept in full as hex."""
if isinstance(v, bytes):
if len(v) <= 64:
try:
return v.decode("utf-8", "strict")
except (UnicodeDecodeError, ValueError):
return f"hex:{v.hex()}"
return f"hex:{v.hex()}"
if isinstance(v, tuple | list):
sequence = cast("list[Any] | tuple[Any, ...]", v)
return [_decode_exif_value(item) for item in sequence]
return v
def read_full_exif(
path: Path, exif_blob: bytes | None = None, data: bytes | None = None
) -> tuple[dict[str, Any], bytes | None]:
"""All EXIF IFDs with decoded tag names (piexif, no re-encode), plus the
raw embedded-thumbnail bytes for the caller's own thumbnail forensics.
``exif_blob`` is the PIL-exposed EXIF blob (PNG/WebP/HEIC path) so the
caller's single Image.open is not repeated here. ``data`` is the
already-read file bytes so piexif does not re-read the file."""
try:
exif: dict[str, Any] = piexif.load(data) if data is not None else piexif.load(str(path))
except Exception:
if not exif_blob:
return {}, None
try:
exif = piexif.load(exif_blob)
except Exception as exc:
return {"error": _safe_str(exc)}, None
out: dict[str, Any] = {}
thumbnail: bytes | None = None
for ifd, tags in exif.items():
if ifd == "thumbnail":
thumbnail = tags if isinstance(tags, bytes) else None
out["thumbnail"] = f"{len(tags)} bytes" if isinstance(tags, bytes) else None
continue
if not isinstance(tags, dict):
continue
all_tag_names = cast("dict[str, dict[int, dict[str, Any]]]", getattr(piexif, "TAGS", {}))
tag_names = all_tag_names.get(ifd, {})
decoded: dict[str, Any] = {}
for tag, value in cast("dict[int, Any]", tags).items():
name = str(tag_names.get(tag, {}).get("name", f"tag_{tag}"))
if name == "MakerNote" and isinstance(value, bytes):
# full hex, no cap: measured on real uploads, Apple is ~2 KB
# but Canon reaches 28 KB and Sony 38 KB (AF data, serials,
# embedded previews) -- a cap would silently drop exactly the
# camera-original evidence this scan exists to preserve
decoded[name] = f"hex:{value.hex()}"
else:
decoded[name] = _decode_exif_value(value)
out[ifd] = decoded
return out, thumbnail
def _png_text_decode(ctype: str, body: bytes) -> str:
"""Decode a tEXt/zTXt/iTXt chunk, inflating zlib where used.
The compressed forms are where ComfyUI / Automatic1111 hide the
generation workflow and prompt, so skipping the inflate would drop
the strongest AI-provenance text a PNG can carry."""
if ctype == "tEXt":
suffix = b"...TRUNCATED" if len(body) > _TEXT_CAP else b""
return (body[:_TEXT_CAP] + suffix).decode("utf-8", "replace")
if ctype == "zTXt":
nul = body.find(b"\x00")
if nul == -1:
return body[:_TEXT_CAP].decode("utf-8", "replace")
keyword = body[:nul].decode("latin-1", "replace")
# body[nul+1] = compression method (0 = zlib)
try:
inflater = zlib.decompressobj()
decoded = inflater.decompress(body[nul + 2 :], _TEXT_CAP + 1)
suffix = "...TRUNCATED" if len(decoded) > _TEXT_CAP else ""
text = decoded[:_TEXT_CAP].decode("utf-8", "replace") + suffix
except zlib.error:
text = body[:_TEXT_CAP].decode("utf-8", "replace")
return f"{keyword}\x00{text}"
# iTXt: keyword\0 compflag(1) compmethod(1) lang\0 translated\0 text
parts = body.split(b"\x00", 1)
if len(parts) < 2:
return body[:_TEXT_CAP].decode("utf-8", "replace")
keyword = parts[0].decode("latin-1", "replace")
rest = parts[1]
if len(rest) < 2:
return body[:_TEXT_CAP].decode("utf-8", "replace")
compflag = rest[0]
tail = rest[2:]
for _ in range(2): # skip language tag and translated keyword
nul = tail.find(b"\x00")
if nul == -1:
return body[:_TEXT_CAP].decode("utf-8", "replace")
tail = tail[nul + 1 :]
if compflag:
with contextlib.suppress(zlib.error):
inflater = zlib.decompressobj()
tail = inflater.decompress(tail, _TEXT_CAP + 1)
if len(tail) > _TEXT_CAP:
tail = tail[:_TEXT_CAP] + b"...TRUNCATED"
return f"{keyword}\x00{tail.decode('utf-8', 'replace')}"
def read_png_chunks(data: bytes) -> tuple[list[dict[str, Any]], bytes]:
"""Every PNG chunk in order (type, length; text chunks decoded and
inflated, binary chunks as base64) plus the post-IEND trailer bytes."""
chunks: list[dict[str, Any]] = []
post_iend = b""
try:
pos = 8
while pos + 12 <= len(data):
length = struct.unpack(">I", data[pos : pos + 4])[0]
ctype = data[pos + 4 : pos + 8].decode("latin-1")
body = data[pos + 8 : pos + 8 + length]
entry: dict[str, Any] = {"type": ctype, "length": length}
if ctype in ("tEXt", "zTXt", "iTXt"):
entry["text"] = _png_text_decode(ctype, body)
if entry["text"].startswith("XML:com.adobe.xmp"):
entry["kind"] = "xmp"
elif ctype == "tIME" and length == 7:
y, mo, d, h, mi, s = struct.unpack(">HBBBBB", body)
entry["time"] = f"{y:04d}-{mo:02d}-{d:02d}T{h:02d}:{mi:02d}:{s:02d}Z"
elif ctype == "gAMA" and length == 4:
entry["gamma"] = struct.unpack(">I", body)[0] / 100000
elif ctype == "sRGB" and length == 1:
entry["rendering_intent"] = body[0]
elif ctype == "iCCP":
nul = body.find(b"\x00")
if nul > 0:
entry["profile_name"] = body[:nul].decode("latin-1", "replace")
entry["base64"] = _b64(body)
elif ctype == "iDOT":
# present in iOS/macOS screenshots
entry["apple_screenshot_marker"] = True
elif ctype in ("IHDR", "IDAT"):
pass # pixel-data / header chunks: length is signal enough
elif length:
entry["base64"] = _b64(body)
chunks.append(entry)
pos += 12 + length
if ctype == "IEND":
post_iend = data[pos:]
break
except Exception as exc:
chunks.append({"error": _safe_str(exc)})
return chunks, post_iend
def _set_jpeg_trailer(result: dict[str, Any], data: bytes, eoi: int) -> None:
"""Preserve bytes after JPEG EOI for Samsung Galaxy AI detection."""
trailer = data[eoi + 2 :]
result["post_eoi_bytes"] = len(trailer)
if trailer:
result["post_eoi_base64"] = _b64(trailer)
def read_jpeg_segments(data: bytes) -> dict[str, Any]:
"""Every JPEG APP segment in order, plus post-EOI trailer size.
XMP APP1 segments are kept as full text; every other segment body is
kept as full base64 (1 MB ceiling per segment).
"""
result: dict[str, Any] = {"segments": [], "post_eoi_bytes": 0}
try:
pos = 2
while pos + 4 <= len(data):
if data[pos] != 0xFF:
break
marker = data[pos + 1]
if marker == 0xD9: # EOI
_set_jpeg_trailer(result, data, pos)
break
if marker == 0xDA: # SOS: entropy-coded data follows
eoi = data.rfind(b"\xff\xd9")
if eoi != -1:
_set_jpeg_trailer(result, data, eoi)
break
if not (0xE0 <= marker <= 0xEF):
length = struct.unpack(">H", data[pos + 2 : pos + 4])[0]
pos += 2 + length
continue
length = struct.unpack(">H", data[pos + 2 : pos + 4])[0]
body = data[pos + 4 : pos + 2 + length]
name = f"APP{marker - 0xE0}"
entry: dict[str, Any] = {"marker": name, "length": length}
# Adobe JPEG XMP APP1 magic (namespace URI in the packet, not a request).
if body.startswith(b"http://ns.adobe.com/xap/1.0/\x00"): # NOSONAR
entry["kind"] = "xmp"
entry["text"] = body[29:].decode("utf-8", "replace")
elif name == "APP2" and body.startswith(b"MPF\x00"):
# Multi-Picture Format: Ultra HDR gain map, Samsung dual shot
entry["kind"] = "mpf"
entry["base64"] = _b64(body)
elif name == "APP2" and body.startswith(b"ICC_PROFILE"):
entry["kind"] = "icc"
entry["base64"] = _b64(body)
elif name == "APP2" and body.startswith(b"FPXR"):
entry["kind"] = "flashpix"
entry["base64"] = _b64(body)
elif name == "APP11":
entry["kind"] = "c2pa_or_jumbf"
# the parsed manifest is in c2pa_store, but the raw JUMBF
# also carries assertion thumbnails the JSON may omit
entry["base64"] = _b64(body)
elif body.startswith(b"Exif\x00\x00"):
entry["kind"] = "exif"
entry["base64"] = _b64(body)
elif body.startswith(b"Photoshop 3.0\x00"):
entry["kind"] = "iptc_iim"
entry["base64"] = _b64(body)
else:
entry["base64"] = _b64(body)
result["segments"].append(entry)
pos += 2 + length
except Exception as exc:
result["error"] = _safe_str(exc)
return result
def read_pil_info(path: Path) -> tuple[dict[str, Any], dict[str, Any], bytes | None]:
"""One Image.open serving all PIL-derived data: container basics,
img.info passthrough (XMP, comments), the IPTC-IIM dataset, and the
raw EXIF blob (for the caller's piexif parse on PNG/WebP/HEIC)."""
out: dict[str, Any] = {}
iptc: dict[str, Any] = {}
exif_blob: bytes | None = None
try:
with Image.open(path) as img:
out["format"] = img.format
out["mode"] = img.mode
out["width"], out["height"] = img.size
out["n_frames"] = getattr(img, "n_frames", 1)
dpi = img.info.get("dpi")
if dpi:
out["dpi"] = [round(float(d), 2) for d in dpi]
icc = img.info.get("icc_profile")
if icc:
out["icc_profile"] = {
"length": len(icc),
# header: profile class, color space, PCS (bytes 12-24)
"header_hex": icc[12:24].hex() if len(icc) >= 24 else "",
"base64": _b64(icc),
}
blob = img.info.get("exif")
if isinstance(blob, bytes):
exif_blob = blob
try:
info = getiptcinfo(img)
except Exception:
info = None
if info:
iptc = {f"{k[0]}:{k[1]}": _decode_exif_value(v) for k, v in info.items()}
for key, value in img.info.items():
if key in ("icc_profile", "exif", "dpi"):
continue
if isinstance(value, bytes):
try:
out[f"info:{key}"] = value.decode("utf-8", "strict")
except (UnicodeDecodeError, ValueError):
out[f"info:{key}"] = f"base64:{_b64(value)}"
else:
out[f"info:{key}"] = _safe_str(value)
except Exception as exc:
out["error"] = _safe_str(exc)
return out, iptc, exif_blob
def read_c2pa_store(path: Path) -> dict[str, Any]:
"""Full C2PA manifest store through the package's cached reader."""
from remove_ai_watermarks._internal.c2pa import read_manifest_store_json
raw = read_manifest_store_json(path)
if raw is None:
return {}
try:
value: Any = json.loads(raw)
return (
cast("dict[str, Any]", value)
if isinstance(value, dict)
else {"error": "C2PA manifest store is not an object"}
)
except (TypeError, ValueError) as exc:
return {"error": _safe_str(exc)}
def sniff_format(head: bytes) -> str:
if head.startswith(b"\x89PNG"):
return "png"
if head.startswith(b"\xff\xd8"):
return "jpeg"
if head.startswith(b"RIFF") and head[8:12] == b"WEBP":
return "webp"
if head[:6] in (b"GIF87a", b"GIF89a"):
return "gif"
if head.startswith(b"BM"):
return "bmp"
if head.startswith((b"II*\x00", b"MM\x00*")):
return "tiff"
if head[4:8] == b"ftyp":
return f"isobmff:{head[8:12].decode('latin-1', 'replace')}"
return f"unknown:{head[:16].hex()}"
# --- JPEG encoder structure (metadata layer) ---
def _jpeg_forensics_bytes(data: bytes) -> dict[str, Any]:
"""Structure-level JPEG forensics: DQT tables (encoder fingerprint),
SOF type (baseline/progressive) + chroma subsampling, DHT Huffman
tables (custom = optimizing encoder), per-scan spectral selection
(progressive scan script), JFIF/Adobe app markers, COM, DRI."""
out: dict[str, Any] = {}
try:
if not data.startswith(b"\xff\xd8"):
return out
pos = 2
scans: list[dict[str, int]] = []
dqt: dict[str, list[int]] = {}
dht: list[str] = []
comments: list[str] = []
while pos + 4 <= len(data):
if data[pos] != 0xFF:
break
marker = data[pos + 1]
if marker in (0xD8, 0x01) or 0xD0 <= marker <= 0xD7:
pos += 2
continue
if marker == 0xD9:
break
length = struct.unpack(">H", data[pos + 2 : pos + 4])[0]
body = data[pos + 4 : pos + 2 + length]
if marker == 0xDB: # DQT
off = 0
while off < len(body):
tid = body[off] & 0x0F
prec = body[off] >> 4
n = 128 if prec else 64
vals = list(body[off + 1 : off + 1 + n])
if prec: # 16-bit entries
vals = [struct.unpack(">H", bytes(vals[i : i + 2]))[0] for i in range(0, len(vals) - 1, 2)]
dqt[str(tid)] = vals[:64]
off += 1 + n
elif marker == 0xC4: # DHT: custom tables mean an optimizing encoder
dht.append(body.hex())
elif marker == 0xDD and len(body) >= 2: # DRI
out["restart_interval"] = struct.unpack(">H", body[:2])[0]
elif marker == 0xE0 and body.startswith(b"JFIF\x00") and len(body) >= 12:
out["jfif"] = {
"version": f"{body[5]}.{body[6]}",
"density_units": body[7],
"x_density": struct.unpack(">H", body[8:10])[0],
"y_density": struct.unpack(">H", body[10:12])[0],
}
elif marker == 0xEE and body.startswith(b"Adobe") and len(body) >= 12:
out["adobe_transform"] = body[11]
elif marker in (0xC0, 0xC1, 0xC2) and len(body) >= 6:
out["progressive"] = marker == 0xC2
out["precision_bits"] = body[0]
out["sof_height"] = struct.unpack(">H", body[1:3])[0]
out["sof_width"] = struct.unpack(">H", body[3:5])[0]
comps: list[dict[str, int]] = []
for i in range(body[5]):
c = body[6 + i * 3 : 9 + i * 3]
if len(c) == 3:
comps.append({"h": c[1] >> 4, "v": c[1] & 0x0F, "tq": c[2]})
if len(comps) >= 3:
lum = comps[0]
subs = {1: "4:4:4", 2: "4:2:2"}.get(lum["h"] * lum["v"])
out["subsampling"] = subs or f"{lum['h']}x{lum['v']}"
elif marker == 0xFE: # COM
comments.append(body.decode("utf-8", "replace")[:2000])
elif marker == 0xDA:
# SOS spectral selection: the progressive scan script
# differs across libjpeg / mozjpeg / Photoshop
if len(body) >= 3:
ns = body[0]
tail = body[1 + ns * 2 :]
if len(tail) >= 3:
scans.append({"ss": tail[0], "se": tail[1], "ah": tail[2] >> 4, "al": tail[2] & 0x0F})
# skip entropy-coded data to the next marker
end = data.find(b"\xff\xd9", pos)
nxt = data.find(b"\xff", pos + 2)
while nxt != -1 and nxt + 1 < len(data) and data[nxt + 1] == 0x00:
nxt = data.find(b"\xff", nxt + 2)
if nxt == -1 or (end != -1 and nxt >= end):
break
pos = nxt
continue
pos += 2 + length
if dqt:
out["quant_tables"] = dqt
if dht:
out["huffman_tables_hex"] = dht
if comments:
out["comments"] = comments
if scans:
out["scan_count"] = len(scans)
out["scan_script"] = scans
except Exception as exc:
out["error"] = _safe_str(exc)
return out
def read_webp_chunks(data: bytes) -> list[dict[str, Any]]:
"""WebP RIFF chunk inventory (VP8X/VP8/VP8L/EXIF/XMP/ICCP/ANIM...)."""
chunks: list[dict[str, Any]] = []
try:
pos = 12
declared_end = 8 + struct.unpack("<I", data[4:8])[0] if len(data) >= 12 else len(data)
container_end = min(len(data), declared_end)
while pos + 8 <= container_end:
chunk_type = data[pos : pos + 4]
ctype = chunk_type.decode("latin-1")
length = struct.unpack("<I", data[pos + 4 : pos + 8])[0]
body = data[pos + 8 : min(pos + 8 + length, container_end)]
entry: dict[str, Any] = {"type": ctype, "length": length}
if ctype == "XMP ":
entry["kind"] = "xmp"
entry["text"] = body.decode("utf-8", "replace")
elif chunk_type in RIFF_CODED_IMAGE_CHUNKS:
pass # pixel-data chunks: length is signal enough
elif length:
entry["base64"] = _b64(body)
chunks.append(entry)
pos += 8 + length + (length & 1) # chunks are 2-byte aligned
except Exception as exc:
chunks.append({"error": _safe_str(exc)})
return chunks
def read_webp_late_metadata_path(path: Path, window: int = _RAW_SCAN_HEAD) -> list[dict[str, Any]]:
"""Stream metadata chunks after ``window`` while seeking over coded frames."""
chunks: list[dict[str, Any]] = []
try:
file_size = path.stat().st_size
with open(path, "rb") as handle:
header = handle.read(12)
if len(header) < 12 or not header.startswith(b"RIFF") or header[8:12] != b"WEBP":
return chunks
container_end = min(file_size, 8 + struct.unpack("<I", header[4:8])[0])
position = 12
while position + 8 <= container_end:
handle.seek(position)
chunk_header = handle.read(8)
if len(chunk_header) < 8:
break
chunk_type = chunk_header[:4]
(length,) = struct.unpack("<I", chunk_header[4:8])
start = position + 8
safe_length = max(0, min(length, container_end - start))
if chunk_type in RIFF_METADATA_CHUNKS and start >= window:
handle.seek(start)
body = handle.read(min(safe_length, _B64_CAP))
entry: dict[str, Any] = {
"type": chunk_type.decode("latin-1"),
"length": length,
"base64": _b64(body),
}
if len(body) < safe_length:
entry["truncated"] = True
chunks.append(entry)
position = start + safe_length + (safe_length & 1)
except (OSError, struct.error) as exc:
chunks.append({"error": _safe_str(exc)})
return chunks
def sha256_of(data: bytes) -> str:
return hashlib.sha256(data).hexdigest()
def xattr_where_from(path: Path) -> list[str]:
"""macOS download-source URLs (kMDItemWhereFroms), empty elsewhere."""
try:
getter = cast("Any", os.getxattr) # pyright: ignore[reportAttributeAccessIssue, reportUnknownMemberType]
raw = cast("bytes", getter(path, "com.apple.metadata:kMDItemWhereFroms"))
value = plistlib.loads(raw)
values = cast("list[Any]", value) if isinstance(value, list) else [value]
return [str(item) for item in values]
except (AttributeError, OSError, ValueError):
return []
def xattr_quarantine(path: Path) -> str | None:
"""macOS quarantine string: flags; timestamp; downloading agent (Safari,
Telegram, Chrome...). Presence alone means 'came from the internet'."""
try:
getter = cast("Any", os.getxattr) # pyright: ignore[reportAttributeAccessIssue, reportUnknownMemberType]
raw = cast("bytes", getter(path, "com.apple.quarantine"))
return raw.decode("utf-8", "replace")[:500]
except (AttributeError, OSError):
return None
def read_isobmff_inventory(data: bytes) -> dict[str, Any]:
"""HEIC/AVIF/MOV box inventory: top-level boxes plus the meta item
types (Exif, mime=XMP, auxl depth/gain-map, aae Apple-edits plist,
irot derived images). Strong phone-provenance signal."""
out: dict[str, Any] = {}
try:
stream = io.BytesIO(data)
def boxes(start: int, end: int) -> list[tuple[str, int, int]]:
return [
(box_type.decode("latin-1"), payload_offset, box_end)
for _, box_end, box_type, payload_offset in iter_file_boxes(stream, start, end)
]
top = boxes(0, len(data))
out["boxes"] = [t for t, _, _ in top]
provenance_boxes: list[dict[str, Any]] = []
for t, s, e in top:
if t.encode("latin-1") in C2PA_BOX_TYPES:
provenance_boxes.append(
{"type": t, "length": e - s, "base64": _b64(data[s:e], cap=_PROVENANCE_B64_CAP)}
)
if t == "moov":
for ct, cs, ce in boxes(s, e):
if ct == "mvhd" and ce - cs >= 24:
# full box + creation/modification times (1904 epoch)
version = data[cs]
base = cs + 4
creation = struct.unpack(">I", data[base : base + 4])[0] if version == 0 else None
if creation:
out["mvhd_creation_time"] = creation - 2082844800
elif t == "meta":
# full box: 4 bytes version/flags, then child boxes
for ct, cs, ce in boxes(s + 4, e):
if ct == "iinf":
# full box + entry count, then infe entries
count = struct.unpack(">H", data[cs + 4 : cs + 6])[0]
out["meta_item_count"] = count
item_types: list[str] = []
for it, is_, ie in boxes(cs + 6, ce):
if it == "infe" and ie - is_ >= 8:
# infe full box: version(1)+flags(3), then
# v2: item_ID(2)+protection(2)+item_type(4)
# v3: item_ID(4)+protection(2)+item_type(4)
version = data[is_]
off = is_ + 4 + (4 if version == 3 else 2) + 2
if off + 4 <= ie:
item_types.append(data[off : off + 4].decode("latin-1", "replace"))
if item_types:
out["meta_item_types"] = sorted(set(item_types))
elif ct == "iprp":
out["has_iprp"] = True
for pt, ps, pe in boxes(cs, ce):
if pt == "ipco":
props = [t for t, _, _ in boxes(ps, pe)]
out["ipco_properties"] = props
# auxC holds the auxiliary image type URN
for box_type, qs, qe in boxes(ps, pe):
if box_type == "auxC":
out["auxc_types"] = (
data[qs + 4 : qe].split(b"\x00")[0].decode("latin-1", "replace")
)
elif ct == "iref":
out["has_iref"] = True
if provenance_boxes:
out["provenance_boxes"] = provenance_boxes
# QuickTime metadata keys (©mak/©mod/©swr) for the MOV side of
# Live Photos: tolerant printable-string grab after each atom
qt: dict[str, str] = {}
for atom, key in ((b"\xa9mak", "make"), (b"\xa9mod", "model"), (b"\xa9swr", "software")):
idx = data.find(atom)
if idx != -1:
m = re.search(rb"[ -~]{4,80}", data[idx + 4 : idx + 200])
if m:
qt[key] = m.group(0).decode("ascii", "replace")
if qt:
out["quicktime"] = qt
except Exception as exc:
out["error"] = _safe_str(exc)
return out
def read_isobmff_provenance_path(path: Path) -> dict[str, Any]:
"""Stream top-level ISOBMFF boxes and preserve provenance payloads.
This is the large-file counterpart to :func:`read_isobmff_inventory`.
It seeks over media payloads instead of loading them into memory.
"""
out: dict[str, Any] = {"boxes": []}
provenance_boxes: list[dict[str, Any]] = []
collected = 0
try:
file_size = path.stat().st_size
with open(path, "rb") as f:
for _, box_end, box_type_raw, payload_offset in iter_file_boxes(f, 0, file_size):
box_type = box_type_raw.decode("latin-1")
out["boxes"].append(box_type)
payload_length = box_end - payload_offset
if box_type_raw in C2PA_BOX_TYPES and collected < _PROVENANCE_B64_CAP:
to_read = min(payload_length, _PROVENANCE_B64_CAP - collected)
f.seek(payload_offset)
payload = f.read(to_read)
entry: dict[str, Any] = {
"type": box_type,
"length": payload_length,
"base64": _b64(payload, cap=_PROVENANCE_B64_CAP),
}
if to_read < payload_length:
entry["truncated"] = True
provenance_boxes.append(entry)
collected += len(payload)
except (OSError, struct.error) as exc:
out["error"] = _safe_str(exc)
if provenance_boxes:
out["provenance_boxes"] = provenance_boxes
return out
def read_png_late_metadata_path(path: Path, window: int = _RAW_SCAN_HEAD) -> list[dict[str, Any]]:
"""Stream PNG metadata chunks whose payload starts after ``window``."""
chunks: list[dict[str, Any]] = []
try:
file_size = path.stat().st_size
with open(path, "rb") as f:
if f.read(8) != b"\x89PNG\r\n\x1a\n":
return chunks
pos = 8
while pos + 12 <= file_size:
f.seek(pos)
header = f.read(8)
if len(header) < 8:
break
length, chunk_type = struct.unpack(">I4s", header)
data_start = pos + 8
safe_length = max(0, min(length, file_size - data_start))
if chunk_type in PNG_METADATA_CHUNKS and data_start >= window:
body = f.read(min(safe_length, _B64_CAP))
entry: dict[str, Any] = {
"type": chunk_type.decode("latin-1"),
"length": length,
"base64": _b64(body),
}
if len(body) < safe_length:
entry["truncated"] = True
chunks.append(entry)
pos = data_start + safe_length + 4
if chunk_type == b"IEND":
break
except (OSError, struct.error) as exc:
chunks.append({"error": _safe_str(exc)})
return chunks
def apple_live_photo_id(head: bytes) -> str | None:
"""Apple Live Photo content identifier (links the still to its MOV).
The UUID sits in the Apple MakerNote (tag 17) of the still and in the
MOV metadata; a raw head scan finds it in either container."""
# the UUID string sits next to "content.identifier" in the MOV, but in
# the STILL it is a bare UUID inside the Apple MakerNote (whose header
# is "Apple iOS"), so gate on either marker
if b"content.identifier" not in head and b"com.apple.quicktime" not in head and b"Apple iOS" not in head:
return None
m = re.search(rb"[0-9A-Fa-f]{8}-[0-9A-Fa-f]{4}-[0-9A-Fa-f]{4}-[0-9A-Fa-f]{4}-[0-9A-Fa-f]{12}", head)
return m.group(0).decode("ascii") if m else None
_MAX_FULL_READ = 256 << 20 # files bigger than this are scanned head-only
_HEAD_READ = 4 << 20
def _sha256_stream(path: Path) -> str:
h = hashlib.sha256()
with open(path, "rb") as f:
for block in iter(lambda: f.read(1 << 20), b""):
h.update(block)
return h.hexdigest()
def collect_forensic_metadata(
path: Path,
*,
schema_version: int = FORENSIC_METADATA_SCHEMA_VERSION,
) -> dict[str, Any]:
"""Collect the versioned, metadata-only forensic record for ``path``.
This broad inspection record is not provenance-detector input. Use
:func:`remove_ai_watermarks.metadata_record.collect_metadata_record` for the
strict record accepted by ``identify_metadata_record``. Long-lived consumers
should request the schema they implement; unsupported versions raise before the
source is read.
"""
schema_version = require_schema_version(
schema_version,
contract="forensic metadata",
supported=(1,),
)
image_io._register_heif() # pyright: ignore[reportPrivateUsage]
stat = path.stat()
oversized = stat.st_size > _MAX_FULL_READ
if oversized:
data = None
with open(path, "rb") as f:
head = f.read(_HEAD_READ)
else:
data = path.read_bytes()
head = data
record: dict[str, Any] = {
"schema_version": schema_version,
"record_type": FORENSIC_METADATA_RECORD_TYPE,
"file": str(path),
"name": path.name,
"extension": path.suffix.lower(),
"size_bytes": stat.st_size,
"mtime": stat.st_mtime,
"birthtime": getattr(stat, "st_birthtime", None),
"sha256": _sha256_stream(path) if data is None else sha256_of(data),
"content_format": sniff_format(head),
}
if oversized:
# Preserve the same bounded byte windows used by downstream provenance
# algorithms while path-based readers (PIL, piexif, C2PA) run normally.
record["oversized"] = {"head_scanned_bytes": len(head)}
record["raw_metadata_windows"] = {"head_base64": _b64(head[:_RAW_SCAN_HEAD])}
if stat.st_size > _RAW_SCAN_TAIL:
with open(path, "rb") as f:
f.seek(-_RAW_SCAN_TAIL, 2)
record["raw_metadata_windows"]["tail_base64"] = _b64(f.read())
where_from = xattr_where_from(path)
if where_from:
record["download_source_urls"] = where_from
quarantine = xattr_quarantine(path)
if quarantine:
record["quarantine"] = quarantine
live_photo_id = apple_live_photo_id(head[: 2 << 20])
if live_photo_id:
record["live_photo_content_id"] = live_photo_id
record["pil"], record["iptc"], exif_blob = read_pil_info(path)
record["exif"], thumbnail = read_full_exif(path, exif_blob, data)
record["c2pa_store"] = read_c2pa_store(path)
if data is not None:
fmt = record["content_format"]
if fmt == "png":
record["png_chunks"], post_iend = read_png_chunks(data)
if post_iend:
record["png_post_iend_bytes"] = len(post_iend)
record["png_post_iend_base64"] = _b64(post_iend)
elif fmt == "jpeg":
record["jpeg"] = read_jpeg_segments(data)
record["jpeg_forensics"] = _jpeg_forensics_bytes(data)
elif fmt == "webp":
record["webp_chunks"] = read_webp_chunks(data)
elif fmt.startswith("isobmff"):
record["isobmff"] = read_isobmff_inventory(data)
elif record["content_format"] == "png":
late_chunks = read_png_late_metadata_path(path)
if late_chunks:
record["png_late_metadata_chunks"] = late_chunks
elif record["content_format"] == "webp":
late_chunks = read_webp_late_metadata_path(path)
if late_chunks:
record["webp_late_metadata_chunks"] = late_chunks
elif record["content_format"].startswith("isobmff"):
record["isobmff"] = read_isobmff_provenance_path(path)
if thumbnail:
record["has_exif_thumbnail"] = True
# the embedded thumbnail is its own JPEG; after an edit its encoder
# forensics commonly MISMATCH the main image (classic tamper tell)
thumb_forensics = _jpeg_forensics_bytes(thumbnail)
thumb_forensics["base64"] = _b64(thumbnail)
record["exif_thumbnail_forensics"] = thumb_forensics
return record
+238 -28
View File
@@ -22,6 +22,7 @@ from __future__ import annotations
import base64
import itertools
import logging
import struct
from dataclasses import dataclass, field
from typing import TYPE_CHECKING, Any, cast
@@ -30,6 +31,8 @@ from remove_ai_watermarks._internal.c2pa import (
cbor_text_after,
extract_c2pa_info,
soft_binding_vendors_in,
synthid_vendors_in,
synthid_verdict,
)
from remove_ai_watermarks._internal.constants import (
C2PA_AI_TOOLS,
@@ -37,6 +40,7 @@ from remove_ai_watermarks._internal.constants import (
C2PA_IDENTITY_AI_ORGS,
C2PA_ISSUERS,
)
from remove_ai_watermarks._internal.schema import require_schema_version
from remove_ai_watermarks.metadata import (
AI_METADATA_KEYS,
AIGC_MARKERS,
@@ -70,6 +74,10 @@ if TYPE_CHECKING:
logger = logging.getLogger(__name__)
# Stable JSON contract for callers that pass a verdict between services. Bump this
# only for a breaking shape or semantic change; adding optional fields is compatible.
PROVENANCE_REPORT_SCHEMA_VERSION = 1
# How much of a non-PNG container to binary-scan for the C2PA issuer.
_SCAN_BYTES = 1024 * 1024
@@ -169,7 +177,32 @@ def _external_metadata(value: Any) -> tuple[list[tuple[str, Any]], bytes]:
"""Index nested metadata and recover common encoded binary values in one pass."""
pairs: list[tuple[str, Any]] = []
parts: list[bytes] = []
diagnostic_keys = {"error", "kind"}
diagnostic_keys = {
"artifacts",
"birthtime",
"color",
"content_format",
"dct",
"ela",
"error",
"extension",
"fft",
"file",
"filename",
"full",
"gradient",
"kind",
"mtime",
"name",
"noise",
"path",
"pixel",
"provenance",
"sha256",
"signals",
"size_bytes",
"timing_ms",
}
def visit(item: Any) -> None:
if isinstance(item, dict):
@@ -177,7 +210,6 @@ def _external_metadata(value: Any) -> tuple[list[tuple[str, Any]], bytes]:
for key, nested in mapping.items():
key_text = str(key)
pairs.append((key_text, nested))
parts.append(key_text.encode("utf-8", "replace"))
if key_text.lower() in diagnostic_keys:
continue
if isinstance(nested, str) and (key_text == "base64" or key_text.endswith("_base64")):
@@ -241,17 +273,69 @@ def _external_exif_generator(pairs: list[tuple[str, Any]], scan: bytes) -> str |
return generator_from_metadata(candidates, scan)
def _metadata_source_kind(info: dict[str, Any], scan: bytes) -> str | None:
"""Normalize the source type wherever it is carried: C2PA or IPTC/XMP.
A composite marker contains ``TrainedAlgorithmicMedia`` as a substring, so it
is removed before looking for a standalone full-generation marker. When a file
genuinely carries both kinds, full generation wins.
"""
structured = info.get("ai_source_kind")
without_composites = scan.replace(b"compositeWithTrainedAlgorithmicMedia", b"").replace(b"compositeSynthetic", b"")
generated = structured == "generated" or any(
marker in without_composites for marker in (b"trainedAlgorithmicMedia", b"TrainedAlgorithmicMedia")
)
if generated:
return "generated"
if structured == "enhanced" or any(
marker in scan for marker in (b"compositeWithTrainedAlgorithmicMedia", b"compositeSynthetic")
):
return "enhanced"
return None
def evidence_from_metadata_record(
record: dict[str, Any], *, path: Path, c2pa_manifest_store: str | dict[str, Any] | None = None
) -> ProvenanceEvidence:
"""Normalize an externally collected metadata record into provenance evidence.
The record may contain arbitrary nested dictionaries and lists. Text, bytes,
hexadecimal values prefixed with ``hex:``, and fields named ``base64`` or
ending in ``_base64`` are included in the shared byte scan. No source file is
opened.
Unversioned external records may contain arbitrary nested dictionaries and
lists. Versioned native records accept only the source-derived fields emitted by
``collect_metadata_record``; other native record types and unknown schema
versions are rejected. No source file is opened.
"""
pairs, scan = _external_metadata(record)
from remove_ai_watermarks.metadata_record import METADATA_RECORD_SCHEMA_VERSION, METADATA_RECORD_TYPE
# Records produced by ``collect_metadata_record`` are a versioned transport
# contract. Only their source-derived fields are evidence: the filename,
# container label and schema bookkeeping describe the collector and must never
# become detector input. Shape-detect the pre-versioned form as well so records
# emitted by 0.26 remain safe and readable.
record_type = record.get("record_type")
if record_type not in (None, METADATA_RECORD_TYPE):
raise ValueError(f"Unsupported metadata record type: {record_type!r}")
if record_type == METADATA_RECORD_TYPE:
require_schema_version(
record.get("schema_version"),
contract="provenance metadata",
supported=(METADATA_RECORD_SCHEMA_VERSION,),
)
status = record.get("status")
if status == "error":
raise ValueError("Provenance metadata collection failed")
if status != "complete":
raise ValueError(f"Unsupported provenance metadata collection status: {status!r}")
is_portable_record = record_type == METADATA_RECORD_TYPE or {
"container",
"metadata_base64",
"tail_base64",
}.issubset(record)
evidence_record = (
{key: record[key] for key in ("metadata_base64", "tail_base64", "pil", "exif") if key in record}
if is_portable_record
else record
)
pairs, scan = _external_metadata(evidence_record)
store = c2pa_manifest_store
if store is None:
candidate = record.get("c2pa_store")
@@ -358,13 +442,13 @@ class ProvenanceReport:
is_ai_generated: bool | None # True / False is never asserted; None = unknown
platform: str | None
confidence: str # "high" | "medium" | "none"
# Coarse AI-origin kind from the C2PA digital-source-type, so a caller can
# branch on full generation vs an AI-touched real photo:
# Coarse AI-origin kind from a C2PA or standalone IPTC/XMP digital-source-type,
# so a caller can branch on full generation vs an AI-touched real photo:
# "generated" -- digitalSourceType trainedAlgorithmicMedia (fully AI).
# "enhanced" -- compositeWithTrainedAlgorithmicMedia (real content with an
# AI-composited region; scrub the AI region, keep the photo).
# None -- no C2PA AI source-type (verdict, if AI, came from another
# signal: IPTC, AIGC, local gen params, xAI, ...).
# None -- no AI digital-source-type (verdict, if AI, came from another
# signal: AIGC, local gen params, xAI, ...).
ai_source_kind: str | None = None
# True when the AI verdict rests on a metadata or embedded-invisible signal
# (C2PA AI issuer / SynthID proxy, IPTC, AIGC, local gen params, EXIF/xAI, or
@@ -383,6 +467,42 @@ class ProvenanceReport:
# inconsistent -- a strong tell of spoofed, transplanted, or laundered metadata.
integrity_clashes: list[str] = field(default_factory=list[str])
def to_dict(
self,
*,
schema_version: int = PROVENANCE_REPORT_SCHEMA_VERSION,
) -> dict[str, Any]:
"""Return the versioned, JSON-safe verdict contract.
``path`` is deliberately omitted. It is extraction context, not part of the
verdict, and local filesystem paths should not cross a service boundary.
Request an explicit schema for a long-lived transport consumer.
"""
schema_version = require_schema_version(
schema_version,
contract="provenance report",
supported=(1,),
)
return {
"schema_version": schema_version,
"is_ai_generated": self.is_ai_generated,
"platform": self.platform,
"confidence": self.confidence,
"ai_source_kind": self.ai_source_kind,
"ai_from_metadata": self.ai_from_metadata,
"watermarks": list(self.watermarks),
"signals": [
{
"name": signal.name,
"detail": signal.detail,
"confidence": signal.confidence,
}
for signal in self.signals
],
"caveats": list(self.caveats),
"integrity_clashes": list(self.integrity_clashes),
}
def extract_provenance_evidence(image_path: Path) -> ProvenanceEvidence:
"""Read all file-backed metadata needed by provenance verdict logic once."""
@@ -447,6 +567,72 @@ _DEVICE_C2PA_PLATFORM: tuple[tuple[bytes, str], ...] = (
)
def _metadata_region(head: bytes) -> bytes:
"""The part of the scan buffer that can hold metadata, with the coded pixels cut out.
The vendor registries are matched as raw substrings, and the shortest tokens are
four and five bytes (``Bria``, ``Adobe``, ``Canva``). Over a megabyte of compressed
pixel data a four-byte sequence appears by chance about once in three thousand
images -- measured: ``Bria`` matched inside the entropy-coded scan of 4 of 14,707
corpus JPEGs, in none of which the manifest names Bria. That is not a cosmetic
mislabel, because the Bria entry carries ``asserts_ai``: a chance match can declare
an image AI-generated.
``c2pa_marker_in`` already refuses a bare ``c2pa`` substring for the same reason.
This is the same defence for the registries: they see the container's metadata and
not its pixels.
JPEG keeps the marker segments before the entropy-coded scan, PNG every chunk but
``IDAT``, and both keep the trailer past the end marker. Anything ``scan_head``
APPENDED past the window is metadata by construction (late chunks, boxes, decoder
text), so it is always kept and never walked -- walking it is what produced 11 MB
records and a phantom AIGC signal in the record collector.
Trimming happens only when the container actually parses: a JPEG whose marker walk
reaches the coded scan, a PNG whose chunk walk reaches ``IDAT``. Anything else --
a malformed container, a synthetic blob, a format with no walker here -- is
returned whole. Cutting a buffer this function did not understand would drop real
evidence to avoid a chance match, which is the wrong way round.
"""
raw, appended = head[:_SCAN_BYTES], head[_SCAN_BYTES:]
if raw[:2] == b"\xff\xd8":
index, size = 2, len(raw)
while index + 1 < size:
if raw[index] != 0xFF:
return head # not a marker boundary: the walk is lost, keep everything
marker = raw[index + 1]
if marker in (0xDA, 0xD9): # SOS / EOI: the coded scan follows
end = raw.rfind(b"\xff\xd9")
return raw[:index] + (raw[end + 2 :] if end >= index else b"") + appended
if 0xD0 <= marker <= 0xD7 or marker == 0x01:
index += 2
continue
if index + 4 > size:
break
length = int.from_bytes(raw[index + 2 : index + 4], "big")
if length < 2 or index + 2 + length > size:
break
index += 2 + length
return head # ran out of buffer before the scan: nothing was skipped anyway
if raw[:8] == b"\x89PNG\r\n\x1a\n":
out = bytearray()
position, size, saw_idat = 8, len(raw), False
while position + 8 <= size:
(length,) = struct.unpack(">I", raw[position : position + 4])
chunk_type = raw[position + 4 : position + 8]
start = position + 8
if chunk_type == b"IDAT":
saw_idat = True
else:
out += chunk_type + raw[start : start + min(length, size - start)]
position = start + length + 4
if chunk_type == b"IEND":
out += raw[position:]
break
return bytes(out) + appended if saw_idat else head
return head
def _first_token_match(head: bytes, table: tuple[tuple[bytes, str], ...]) -> str | None:
"""First platform in ``table`` whose token appears in ``head``, else None.
@@ -883,23 +1069,22 @@ def _identify_from_evidence(
# score, the latter can be a by-product of our own SDXL removal pass, so
# neither is a trustworthy "the generator stamped its identity" claim.
ai_vendor_claims: dict[str, str] = {}
camera_label = _device_platform(head)
signer_label = _signer_platform(head)
# The vendor registries match short raw substrings, so they read the container's
# metadata rather than its pixels -- see `_metadata_region`. Every other check
# below keeps the full buffer: their markers are long and distinctive.
region = _metadata_region(head)
camera_label = _device_platform(region)
signer_label = _signer_platform(region)
# ── C2PA Content Credentials ────────────────────────────────────
has_c2pa = bool(info) or c2pa_marker_in(head)
issuers = [info["issuer"]] if info.get("issuer") else _issuers_in(head)
issuers = [info["issuer"]] if info.get("issuer") else _issuers_in(region)
# Full AI generation (trainedAlgorithmicMedia) vs an AI-enhanced real photo
# (compositeWithTrainedAlgorithmicMedia). The structured kind is parsed once in
# _internal.c2pa._populate_registry_fields (covers PNG + any container the c2pa-python
# reader handles); fall back to a raw head scan for the non-PNG raw-blob path
# where extract_c2pa_info returns {}. Full generation wins when both appear.
c2pa_source_kind = info.get("ai_source_kind")
if c2pa_source_kind is None:
if b"trainedAlgorithmicMedia" in head:
c2pa_source_kind = "generated"
elif b"compositeWithTrainedAlgorithmicMedia" in head:
c2pa_source_kind = "enhanced"
source_kind = _metadata_source_kind(info, head)
# An identity-AI issuer (a pure-generator brand like Dreamina) asserts AI even
# without a digitalSourceType -- some ByteDance/Dreamina manifests ship no
# trainedAlgorithmicMedia, so the registered generator name is the only signal.
@@ -907,7 +1092,7 @@ def _identify_from_evidence(
# does not reopen the incidental-mention problem the common-word issuers have.
issuer_blob = " ".join(issuers)
c2pa_identity_ai = has_c2pa and any(org in issuer_blob for org in C2PA_IDENTITY_AI_ORGS)
c2pa_is_ai = c2pa_source_kind is not None or c2pa_identity_ai
c2pa_is_ai = source_kind is not None or c2pa_identity_ai
# Generator string (for the signal detail): structured for PNG, CBOR-scanned
# for other containers. Best-effort -- some manifests key it as
# `claim_generator_info` (Pixel), so this can be None even when a device is
@@ -915,7 +1100,7 @@ def _identify_from_evidence(
generator = (
info.get("claim_generator")
or cbor_text_after(head, b"claim_generator")
or (", ".join(tools) if (tools := _ai_tools_in(head)) else None)
or (", ".join(tools) if (tools := _ai_tools_in(region)) else None)
)
# Platform: a distinctive device/camera token in the manifest wins (it is the
# signer/producer), then an editing-app/AI-device signer (Samsung Galaxy,
@@ -950,9 +1135,24 @@ def _identify_from_evidence(
platform = f"C2PA signer: {cloud_vendor} (cloud manifest)"
# ── SynthID metadata proxy ──────────────────────────────────────
# get_ai_metadata already sets synthid_watermark for both PNG (caBX parser)
# and non-PNG (its own synthid_source fallback), so no extra scan is needed.
# Structured first (the PNG caBX parser and the manifest store both fill
# `synthid_watermark`), then the byte scan for the containers that keep the
# manifest where no parser reaches it.
#
# The scan lives HERE, in the verdict, and not in extraction, for the same reason
# `soft_binding` below does: extraction has two implementations -- one reading a
# file, one reading a portable record -- and a rule that lives in only one of them
# is a rule the other silently lacks. It did: 74 corpus images reported SynthID
# through `identify` and not through the record, because `get_ai_metadata`'s own
# fallback has no counterpart on the record side. `get_ai_metadata` keeps its copy
# for its own callers; the verdict no longer depends on which extractor ran.
synthid = meta.get("synthid_watermark")
# The literal byte checks mirror `metadata.synthid_source` exactly rather than
# reusing the derived `has_c2pa` / `source_kind` above, which are broader:
# the file path's answer must not move.
trained_source = b"trainedAlgorithmicMedia" in head or b"TrainedAlgorithmicMedia" in head
if not synthid and trained_source and c2pa_marker_in(head) and (vendors := synthid_vendors_in(region)):
synthid = synthid_verdict(", ".join(vendors))
if synthid:
watermarks.append(f"SynthID watermark, inferred from C2PA metadata ({synthid})")
caveats.append(_SYNTHID_CAVEAT)
@@ -964,7 +1164,7 @@ def _identify_from_evidence(
# ── C2PA soft-binding: a named forensic/third-party watermark vendor ─
# (Adobe TrustMark, Digimarc, Imatag, ...). Present in the manifest even when
# the watermark itself can't be decoded; names whose watermark stamped the pixels.
soft_binding = meta.get("soft_binding") or (", ".join(v) if (v := soft_binding_vendors_in(head)) else None)
soft_binding = meta.get("soft_binding") or (", ".join(v) if (v := soft_binding_vendors_in(region)) else None)
if soft_binding:
signals.append(Signal("soft_binding", f"C2PA soft binding: {soft_binding}", "high"))
watermarks.append(f"Forensic watermark soft binding ({soft_binding})")
@@ -1136,9 +1336,9 @@ def _identify_from_evidence(
is_ai_generated=is_ai,
platform=platform,
confidence=confidence,
# Only meaningful when the AI verdict actually came from the C2PA source
# type; a non-C2PA AI signal (IPTC/AIGC/local gen) leaves it None.
ai_source_kind=c2pa_source_kind if (is_ai and has_c2pa) else None,
# Meaningful for the same digitalSourceType whether carried by C2PA or a
# standalone IPTC/XMP label. Other AI signals leave it None.
ai_source_kind=source_kind if (is_ai and (has_c2pa or iptc)) else None,
ai_from_metadata=ai_from_metadata,
watermarks=watermarks,
signals=signals,
@@ -1170,6 +1370,16 @@ def identify_from_evidence(
)
def identify_metadata_record(record: dict[str, Any], *, path: Path) -> ProvenanceReport:
"""Build a metadata-only verdict from a portable metadata record.
This is the service-integration entry point: the source file is never opened,
and callers receive the same verdict as the explicit
``evidence_from_metadata_record`` / ``identify_from_evidence`` sequence.
"""
return identify_from_evidence(evidence_from_metadata_record(record, path=path))
def identify(
image_path: Path,
*,
+129 -8
View File
@@ -18,6 +18,11 @@ if TYPE_CHECKING:
from collections.abc import Callable, Iterable
from pathlib import Path
from remove_ai_watermarks._internal.constants import (
PNG_METADATA_CHUNKS,
RIFF_METADATA_CHUNKS,
)
logger = logging.getLogger(__name__)
# Smaller scan_head window for the cheap marker checks (has_ai_metadata,
@@ -236,11 +241,6 @@ def _is_ai_value(value: str) -> bool:
return any(token in value_lower for token in AI_GENERATOR_TOKENS)
# PNG ancillary chunks that can carry provenance metadata (XMP, EXIF, text).
# Never IDAT -- that is the compressed pixel stream.
_PNG_META_CHUNKS: frozenset[bytes] = frozenset({b"tEXt", b"iTXt", b"zTXt", b"eXIf", b"iCCP"})
def _png_late_metadata(image_path: Path, window: int) -> bytes:
"""Payloads of PNG metadata chunks that start *beyond* the first ``window``
bytes, found by seeking past the (large) ``IDAT`` pixel stream.
@@ -272,7 +272,7 @@ def _png_late_metadata(image_path: Path, window: int) -> bytes:
# Clamp the attacker-controlled 32-bit length to the bytes that
# actually remain, so a malformed huge length can't allocate GBs.
safe_length = max(0, min(length, file_size - data_start))
if chunk_type in _PNG_META_CHUNKS and data_start >= window:
if chunk_type in PNG_METADATA_CHUNKS and data_start >= window:
f.seek(data_start)
out += f.read(safe_length)
# Advance by the CLAMPED length: a malformed/inflated `length` that
@@ -285,6 +285,55 @@ def _png_late_metadata(image_path: Path, window: int) -> bytes:
return bytes(out)
def _riff_late_metadata(image_path: Path, window: int, *, max_total: int = 4 * 1024 * 1024) -> bytes:
"""Payloads of RIFF metadata chunks that start *beyond* the first ``window``
bytes, found by stepping over the (large) coded-image chunk.
The WebP layout puts ``XMP ``/``EXIF`` AFTER the pixels, so a fixed read can stop
before an IPTC or C2PA AI label. This is the RIFF analogue of
:func:`_png_late_metadata`; it returns only chunks past ``window`` so bytes
already in the head are not duplicated, and empty when there are none.
``max_total`` caps what a metadata scan can pull into memory, the same ceiling
``isobmff.scan_c2pa_region`` applies. Clamping each chunk to the bytes that remain
is not enough on its own: a corrupt or crafted file can declare one ``XMP `` chunk
spanning most of itself, and this runs on the memoized verdict path for images from
arbitrary sources. A label that needs more than 4 MB of XMP does not exist.
"""
out = bytearray()
try:
with open(image_path, "rb") as f:
if f.read(4) != b"RIFF":
return b""
f.seek(0, 2)
file_size = f.tell()
f.seek(4)
declared_size = f.read(4)
if len(declared_size) < 4:
return b""
container_end = min(file_size, 8 + struct.unpack("<I", declared_size)[0])
position = 12 # 'RIFF' + size + form type
while position + 8 <= container_end and len(out) < max_total:
f.seek(position)
header = f.read(8)
if len(header) < 8:
break
chunk_type = header[:4]
(length,) = struct.unpack("<I", header[4:8])
start = position + 8
# Clamp to what remains: a malformed 32-bit length must not push the
# walk past EOF and abandon a genuine label chunk after it.
safe_length = max(0, min(length, container_end - start))
if chunk_type in RIFF_METADATA_CHUNKS and start >= window:
f.seek(start)
out += f.read(min(safe_length, max_total - len(out)))
position = start + safe_length + (safe_length & 1) # chunks are word-aligned
except OSError as exc:
logger.debug("RIFF late-metadata scan failed on %s: %s", image_path, exc)
return b""
return bytes(out)
def _stat_key(image_path: Path) -> tuple[str, int, int] | None:
"""Cache key identifying this file's exact CONTENT, or None when it cannot stat.
@@ -306,11 +355,16 @@ def scan_head(image_path: Path, size: int = 1024 * 1024) -> bytes:
past large boxes like ``mdat``) and PNG ``tEXt`` / ``iTXt`` / ``eXIf`` chunks
(seeking past ``IDAT``).
A file at least ``size`` bytes long additionally gets the metadata text its
decoder can reach but a raw read cannot (:func:`_decoder_visible_text`): a
compressed PNG ``zTXt`` packet, or a chunk past the window in a container with no
late-chunk reader here. A file that fits inside ``size`` is exactly
``f.read(size)``, since the raw read already holds every byte.
This is the shared input for every C2PA / AIGC / IPTC byte scan. The
extensions catch a manifest or XMP packet placed AFTER the media data -- a
non-faststart MP4 manifest, or a PNG XMP packet appended after the pixels --
which a fixed first-MB read would miss. For other inputs, and for files that
fit within ``size``, it is exactly ``f.read(size)`` -- behavior-neutral.
which a fixed first-MB read would miss.
The result is memoized per (path, size, mtime): one ``identify``/``get_ai_metadata``
call fans out to ~8 byte-scan detectors that each call this on the same file, so
@@ -347,9 +401,63 @@ def _scan_head_impl(image_path: Path, size: int) -> bytes:
# len(head) == size means the file is at least `size` bytes, so metadata
# chunks may lie beyond the window; otherwise the whole PNG is in `head`.
head += _png_late_metadata(image_path, size)
elif head[:4] == b"RIFF" and head[8:12] == b"WEBP" and len(head) == size:
head += _riff_late_metadata(image_path, size)
if len(head) >= size:
head += _decoder_visible_text(image_path, head)
return head
# Text values the image decoder can reach that a raw byte read cannot. Bounded: a
# packet larger than this is not a provenance label.
_DECODED_TEXT_LIMIT = 512 * 1024
# Decoder values that are binary payloads with their own readers, not metadata text.
# An ICC profile is colour data and can run to hundreds of kilobytes; appending it
# would bloat the buffer every later detector re-scans, for no signal.
_DECODER_BINARY_KEYS = frozenset({"icc_profile"})
def _decoder_visible_text(image_path: Path, head: bytes) -> bytes:
"""Metadata text PIL can decode but the raw window does not contain.
This is the last of two layers, not the first. Metadata placed BEYOND the window
is the structural readers' job (``_png_late_metadata``, ``_riff_late_metadata``,
the ISOBMFF box walk), and they work on a file no decoder can open. What is left
for this one is metadata the bytes do not spell at all:
* COMPRESSED -- a PNG ``zTXt`` chunk is zlib-deflated, so an XMP packet carrying
a TC260 AIGC label is unreadable as bytes while PIL inflates it on open.
It stays container-agnostic on purpose: it is the net under a placement no
structural reader here knows about yet.
Only text ALREADY MISSING from ``head`` is appended, so the common case adds
nothing and no detector sees a value twice. Skipped entirely when the file fits
inside the window, since then the raw read already holds every byte.
"""
try:
from PIL import Image
with Image.open(image_path) as img:
values = [value for key, value in img.info.items() if key not in _DECODER_BINARY_KEYS]
except Exception as exc: # a container PIL cannot open: the raw scan stands alone
logger.debug("decoder-visible text unavailable for %s: %s", image_path, exc)
return b""
out = bytearray()
for value in values:
if isinstance(value, str):
encoded = value.encode("utf-8", "replace")
elif isinstance(value, bytes):
encoded = value
else:
continue
if len(encoded) > _DECODED_TEXT_LIMIT or not encoded or encoded in head:
continue
out += b"\x00" + encoded
return bytes(out)
def has_ai_metadata(image_path: Path) -> bool:
"""Check if an image contains AI-generation metadata.
@@ -1485,3 +1593,16 @@ def xai_signature(image_path: Path) -> bool:
if key is None:
return _xai_signature_impl(image_path)
return _xai_signature_cached(*key)
# ── Shared with the portable metadata record ────────────────────────
# `metadata_record` must read exactly the windows and markers the file path reads: a
# record built from a different window is a record whose verdict can disagree with
# `identify` on the same image. Aliased rather than renamed because the private names
# are load-bearing in this module's own tests and in a corpus script.
QUICK_SCAN_BYTES = _QUICK_SCAN_BYTES
SAMSUNG_EDITOR_MARKER = _SAMSUNG_EDITOR_MARKER
read_file_tail = _read_file_tail
png_late_metadata = _png_late_metadata
riff_late_metadata = _riff_late_metadata
exif_text = _exif_text
+404
View File
@@ -0,0 +1,404 @@
"""Collect one image's provenance metadata into a portable, JSON-safe record.
WHY THIS EXISTS
``extract_provenance_evidence`` reads a file and hands back evidence in memory, so
collection and verdict must happen in the same process, on the machine holding the
image. This module splits them: collect here, judge anywhere, from a record that
survives JSON.
record = collect_metadata_record(path) # touches the file
evidence = evidence_from_metadata_record(record, path=path)
report = identify_from_evidence(evidence) # touches nothing
WHAT GOES IN, AND WHY NOT SIMPLY THE FILE HEAD
The verdict reads a scan buffer that ``scan_head`` fills with the first mebibyte of
the file. Shipping that verbatim would make a record larger than a phone photo's
worth of metadata by two orders of magnitude, because for a PNG almost all of that
mebibyte is compressed pixel data in ``IDAT`` -- bytes no provenance token can ever
live in. A record carries the metadata REGIONS instead, walked per container: the
JPEG marker segments before the coded scan, every PNG chunk but ``IDAT``, the RIFF
chunks that are not coded image, the ISOBMFF provenance boxes, and in every case the
container's trailer.
COMPLETENESS IS A MEASURED PROPERTY, NOT A CLAIM
A region walker is only correct if nothing the verdict reads falls outside the
regions it keeps, and no test over fixtures can establish that: the failure mode is
a container placement nobody thought of. The contract is therefore ALSO verified
against the file path over a real corpus -- same image, both paths, identical
``ProvenanceReport``.
The placements that defeated an earlier draft of this collector, and the reason each
rule below exists, are recorded in ``docs/module-internals.md`` under "Portable
metadata record".
"""
from __future__ import annotations
import base64
import logging
import struct
from typing import TYPE_CHECKING, Any
if TYPE_CHECKING:
from pathlib import Path
from remove_ai_watermarks._internal.constants import PNG_SIGNATURE, RIFF_CODED_IMAGE_CHUNKS
from remove_ai_watermarks._internal.schema import require_schema_version
from remove_ai_watermarks.metadata import (
QUICK_SCAN_BYTES,
SAMSUNG_EDITOR_MARKER,
exif_text,
read_file_tail,
)
logger = logging.getLogger(__name__)
# The structural walk covers the same window the file path reads raw, so the two
# cannot disagree about a chunk type inside it. A smaller window would be cheaper but
# opens a blind spot: past the window only ``png_late_metadata``'s ALLOWLIST is
# collected, while the file path still sees every chunk type up to its own window --
# and a C2PA ``caBX`` chunk is in neither that allowlist nor ``IDAT``. Walking here
# costs little because the payload of the pixel stream is skipped, not copied.
HEAD_WINDOW = 1024 * 1024
# The window searched for the container's end marker. Matches the quick-scan window
# the file path uses when it goes looking for a Samsung trailer, so a trailer visible
# to one path is visible to the other.
TAIL_WINDOW = QUICK_SCAN_BYTES
# Kept from the tail when no end marker is found, so an unrecognized container still
# contributes its last bytes without carrying half a photo.
UNKNOWN_TRAILER_WINDOW = 64 * 1024
# PNG text keys the file path reads for a generator tag, in ITS order. NovelAI stamps
# Software/Source/Title rather than EXIF, and the first match wins, so order matters.
_GENERATOR_TEXT_KEYS = ("Software", "Source", "Title", "Description")
# Stable transport contract for records produced by this module. The version is
# deliberately separate from the verdict version: collection and interpretation can
# evolve independently as long as old records remain readable.
METADATA_RECORD_SCHEMA_VERSION = 1
METADATA_RECORD_TYPE = "provenance_metadata"
def _jpeg_regions(data: bytes) -> bytes:
"""Every marker segment up to the entropy-coded scan, plus the trailer after EOI.
The scan itself is skipped by walking to SOS and then jumping to the trailing
EOI, so a 20 MB photo contributes only its markers.
TWIN: ``metadata._strip_jpeg_metadata_lossless`` walks the same marker chain. The
two were left separate on purpose -- that one couples the walk to "return False and
fall back to a PIL re-encode", a decision the lossless strip path owns and this one
must not inherit -- so a fix to marker handling belongs in BOTH.
"""
out = bytearray()
index, size = 2, len(data)
while index + 1 < size:
if data[index] != 0xFF:
break # malformed boundary: keep what was collected, the tail still follows
marker = data[index + 1]
if marker in (0xDA, 0xD9): # SOS / EOI: the coded scan follows
break
if 0xD0 <= marker <= 0xD7 or marker == 0x01: # standalone, no length
index += 2
continue
if index + 4 > size:
break
segment_length = int.from_bytes(data[index + 2 : index + 4], "big")
end = index + 2 + segment_length
if segment_length < 2 or end > size:
break
out += data[index:end]
index = end
return bytes(out)
def _png_regions(data: bytes) -> bytes:
"""Every chunk except the ``IDAT`` payloads, plus whatever follows IEND.
TWIN: ``metadata._png_late_metadata`` walks the same chunk chain by SEEKING over
the file rather than over a buffer, and keeps an allowlist rather than skipping
``IDAT``. Both filters are deliberate: inside the window the file path sees every
chunk type raw, past it only the allowlist survives.
"""
out = bytearray()
size = len(data)
position = len(PNG_SIGNATURE)
while position + 8 <= size:
(length,) = struct.unpack(">I", data[position : position + 4])
chunk_type = data[position + 4 : position + 8]
start = position + 8
# Clamp the length to the bytes that remain: a malformed 32-bit length must
# not push the walk past EOF and abandon a genuine label chunk after it.
safe_length = max(0, min(length, size - start))
if chunk_type != b"IDAT":
out += chunk_type + data[start : start + safe_length]
position = start + safe_length + 4 # payload + CRC
if chunk_type == b"IEND":
out += data[position:] # a trailer past IEND is metadata too
break
return bytes(out)
def _riff_regions(data: bytes) -> bytes:
"""Every RIFF chunk except the coded image payloads.
TWIN: ``metadata._riff_late_metadata`` (seek-based, past the scan window) and
``_internal.riff`` (AVI ``LIST/INFO``). Same chunk-stepping arithmetic, three
input models.
"""
out = bytearray(data[:12]) # 'RIFF' + size + 'WEBP'
declared_end = 8 + struct.unpack("<I", data[4:8])[0] if len(data) >= 12 else len(data)
size = min(len(data), declared_end)
position = 12
while position + 8 <= size:
chunk_type = data[position : position + 4]
(length,) = struct.unpack("<I", data[position + 4 : position + 8])
start = position + 8
safe_length = max(0, min(length, size - start))
if chunk_type not in RIFF_CODED_IMAGE_CHUNKS:
out += chunk_type + data[start : start + safe_length]
position = start + safe_length + (safe_length & 1) # chunks are word-aligned
return bytes(out)
def _isobmff_regions(image_path: Path, head: bytes) -> bytes:
"""Header window plus the provenance regions the bounded box walkers find.
ISOBMFF hides a manifest in a ``uuid``/``jumb`` box that can sit after a
multi-megabyte ``mdat``, and a TC260 label in ``moov.udta``. Both walkers seek
rather than read the media, so neither pulls the payload in.
"""
from remove_ai_watermarks._internal.isobmff import scan_c2pa_region, tc260_aigc_payloads
out = bytearray(head[:HEAD_WINDOW])
try:
out += scan_c2pa_region(image_path)
except Exception as exc:
logger.debug("ISOBMFF C2PA region scan failed on %s: %s", image_path, exc)
try:
for payload in tc260_aigc_payloads(image_path):
out += payload
except Exception as exc:
logger.debug("ISOBMFF TC260 scan failed on %s: %s", image_path, exc)
return bytes(out)
def _container_regions(image_path: Path, head: bytes) -> tuple[str, bytes]:
"""(container label, metadata bytes) for the container ``head`` starts with.
``head`` must be the file's raw first bytes. Handing this the ``scan_head``
buffer instead is a trap: that buffer is the head CONCATENATED with late metadata
payloads, so a structural walk runs off the end of the real head and parses the
appended bytes as chunks, inflating the record and creating false signals.
"""
from remove_ai_watermarks._internal.isobmff import is_isobmff
from remove_ai_watermarks.metadata import png_late_metadata, riff_late_metadata
if head.startswith(b"\xff\xd8"):
return "jpeg", _jpeg_regions(head)
if head.startswith(PNG_SIGNATURE):
# Chunks placed after the pixel stream (an XMP packet at 2.7 MB, say) are
# past the window; the same seek-past-IDAT reader the file path uses gets them.
return "png", _png_regions(head) + png_late_metadata(image_path, HEAD_WINDOW)
if head.startswith(b"RIFF") and head[8:12] == b"WEBP":
return "webp", _riff_regions(head) + riff_late_metadata(image_path, HEAD_WINDOW)
if is_isobmff(head):
return "isobmff", _isobmff_regions(image_path, head)
return "unknown", head
def _raw_head(image_path: Path) -> bytes:
"""The file's first bytes, unmodified -- the input every structural walk needs."""
try:
with open(image_path, "rb") as handle:
return handle.read(HEAD_WINDOW)
except OSError as exc:
logger.debug("head read failed for %s: %s", image_path, exc)
return b""
def _trailer(image_path: Path, container: str) -> bytes:
"""The bytes that follow the container's end marker, and nothing else.
A fixed-size tail read would be almost entirely pixels: the trailer of a 20 MB
photo is a few kilobytes at most. So the end marker is located in the tail window
and only what follows it is kept. When no marker is found (an unknown container,
or one whose end lies before the window) the window is kept as-is, bounded --
that is what a byte scan of the same file would have seen anyway.
"""
if container == "webp":
# RIFF declares its structural end in bytes 4..8. A fixed tail window is
# normally the last animation/frame payload, not a trailer, so preserve
# only bytes appended after the declared RIFF container.
try:
with open(image_path, "rb") as handle:
header = handle.read(12)
if len(header) < 12 or not header.startswith(b"RIFF"):
return b""
declared_end = 8 + struct.unpack("<I", header[4:8])[0]
handle.seek(0, 2)
file_size = handle.tell()
if declared_end < 12 or declared_end >= file_size:
return b""
handle.seek(declared_end)
return handle.read(min(file_size - declared_end, UNKNOWN_TRAILER_WINDOW))
except OSError as exc:
logger.debug("RIFF trailer read failed for %s: %s", image_path, exc)
return b""
if container == "isobmff":
# ISOBMFF has no out-of-container trailer convention. Its bounded box
# walkers already collect late provenance while skipping ``mdat``; keeping
# a blind tail here would carry coded media bytes.
return b""
tail = read_file_tail(image_path, TAIL_WINDOW)
if SAMSUNG_EDITOR_MARKER in tail:
# Galaxy AI splits its evidence: the marker sits in the post-EOI trailer, but
# the `genAIType` value it is gated on can sit INSIDE the entropy-coded scan.
# Keeping only the trailer therefore carries the marker without the value and the
# verdict silently drops the Samsung signal, so a marked file keeps the whole
# window. Only Samsung-marked files pay for it.
return tail
marker = {"jpeg": b"\xff\xd9", "png": b"IEND\xae\x42\x60\x82"}.get(container)
if marker is None:
return tail[-UNKNOWN_TRAILER_WINDOW:]
index = tail.rfind(marker)
return tail[index + len(marker) :] if index >= 0 else tail[-UNKNOWN_TRAILER_WINDOW:]
def _decoder_info(image_path: Path) -> dict[str, Any]:
"""PIL's ``info`` mapping, read once.
One open for both consumers below. They want different parts of the same mapping
(the text keys, and the raw EXIF blob), and opening twice repeats the container
header parse and, for a PNG carrying ``zTXt``, the zlib inflate with it.
"""
try:
from PIL import Image
with Image.open(image_path) as img:
# PIL types this mapping with a non-string key union (a DPI tuple key
# exists), so the keys are normalized here rather than assumed.
return {str(key): value for key, value in img.info.items()}
except Exception as exc: # a container PIL cannot open
logger.debug("PIL info unavailable for %s: %s", image_path, exc)
return {}
def _exif_pairs(info: dict[str, Any]) -> dict[str, str]:
"""The 0th-IFD tags the verdict reads, under their tag NAMES.
Not a convenience: two probes key on names rather than on the raw bytes already
in the regions. ``xai_signature_pair`` wants an (ImageDescription, Artist) pair,
and ``_external_exif_generator`` looks for Software / Make / Artist /
ImageDescription. Ship the bytes alone and both silently return nothing, which
is how a collector can silently lose Grok and NovelAI verdicts.
"""
exif_bytes = info.get("exif")
if not exif_bytes:
return {}
try:
import piexif
tags = piexif.load(exif_bytes).get("0th", {})
except Exception as exc: # malformed EXIF
logger.debug("EXIF parse failed: %s", exc)
return {}
return {
name: text
for name, tag in (
("Software", piexif.ImageIFD.Software),
("Make", piexif.ImageIFD.Make),
("Artist", piexif.ImageIFD.Artist),
("ImageDescription", piexif.ImageIFD.ImageDescription),
)
if (text := exif_text(tags, tag))
}
def _pil_info(info: dict[str, Any]) -> dict[str, str]:
"""PIL's ``info`` mapping as strings, the source of PNG text keys and ``hf-job-id``."""
def text_of(value: Any) -> str:
return value.decode("utf-8", "replace") if isinstance(value, bytes) else str(value)
# Emitted in the file path's own candidate order. ``generator_from_metadata``
# returns the FIRST candidate carrying a known token. A record using PIL's natural
# dict order can therefore choose a different platform string than the file path,
# even though the two paths are supposed to be indistinguishable.
out: dict[str, str] = {}
for key in _GENERATOR_TEXT_KEYS:
value = info.get(key)
if value is not None and not isinstance(value, (dict, list, tuple)):
out[f"info:{key}"] = text_of(value)
for key, value in info.items():
if key in _GENERATOR_TEXT_KEYS or isinstance(value, (dict, list, tuple)):
continue
out[f"info:{key}"] = text_of(value)
return out
def collect_metadata_record(
image_path: Path,
*,
schema_version: int = METADATA_RECORD_SCHEMA_VERSION,
) -> dict[str, Any]:
"""Collect everything the provenance verdict reads, as a JSON-safe record.
The record is the transport format for
:func:`identify.evidence_from_metadata_record`: it carries the metadata regions
(base64), the C2PA manifest store, and PIL's info mapping without carrying the
primary coded-pixel stream. Schema and collection status are explicit so a
consumer cannot mistake a failed read for an unknown provenance verdict.
Args:
image_path: Path to the image.
schema_version: Output schema implemented by the consumer.
Returns:
A versioned JSON-serializable dict. ``metadata_base64`` holds the
concatenated container regions, ``tail_base64`` the file trailer.
"""
schema_version = require_schema_version(
schema_version,
contract="provenance metadata",
supported=(1,),
)
from remove_ai_watermarks._internal.c2pa import read_manifest_store_json
try:
image_path.stat()
status = "complete"
issues: list[dict[str, str]] = []
except OSError as exc:
logger.debug("metadata source unavailable for %s: %s", image_path, exc)
status = "error"
issues = [{"stage": "source", "code": "unavailable"}]
container, regions = _container_regions(image_path, _raw_head(image_path))
info = _decoder_info(image_path)
record: dict[str, Any] = {
"schema_version": schema_version,
"record_type": METADATA_RECORD_TYPE,
"status": status,
"issues": issues,
"container": container,
"name": image_path.name,
"metadata_base64": base64.b64encode(regions).decode("ascii"),
# Always collected: Samsung's Galaxy AI marker is a post-EOI trailer, and a
# record without it loses that verdict outright.
"tail_base64": base64.b64encode(_trailer(image_path, container)).decode("ascii"),
# PIL info BEFORE exif: the file path prefers a PNG text tag over an EXIF
# one, and the normalizer walks the record in insertion order.
"pil": _pil_info(info),
"exif": _exif_pairs(info),
}
store = read_manifest_store_json(image_path)
if store is not None:
record["c2pa_store"] = store
return record
+484
View File
@@ -0,0 +1,484 @@
# pyright: reportUnknownMemberType=false, reportUnknownArgumentType=false, reportUnknownVariableType=false, reportMissingTypeStubs=false
"""The complete pixel-forensics layer for one image.
STATUS
Independent from provenance verdicts, removal, and the CLI. Consumers use the
versioned :meth:`PixelEvidence.to_dict` boundary; feature extraction failures are
reported per family without discarding successful measurements.
WHAT IS MEASURED
One decode, then six families of scale-robust statistics over it:
* ``dct`` -- AC coefficient histograms over the 8x8 block DCT, plus the deviation of
leading digits from Benford's law.
* ``fft`` -- radial band energies of the log-magnitude spectrum, plus the
color-filter-array periodicity peaks a demosaiced camera capture leaves.
* ``noise`` -- standard deviation and kurtosis of a high-pass residual.
* ``ela`` -- error level after a quality-90 JPEG re-save.
* ``gradient`` -- gradient-magnitude histogram and Laplacian variance.
* ``color`` -- 4x4x4 RGB histogram, mean saturation, mean value.
and, in ``artifacts``, the spatial layer those statistics are computed from: a
64-bit perceptual hash, a 128px JPEG thumbnail, and coarse ELA, noise-residual and
FFT-phase maps.
THE ARTIFACTS ARE NOT AGGREGATES
Everything above ``artifacts`` is a scalar or a fixed-length histogram, and an image
cannot be reconstructed from those. ``artifacts`` is different in kind: a thumbnail
is a picture, a perceptual hash identifies one, and the coarse maps carry layout.
Collecting them makes a record that identifies the source image, so a caller storing
or forwarding them is handling image content, not statistics about it. That is why
they are a separate field and not merged into the families.
REQUIREMENTS
Needs the ``pixels`` extra (numpy). Guard a call with :func:`is_available` when the
caller must not hard-depend on it.
"""
from __future__ import annotations
import base64
import io
import logging
import time
from dataclasses import dataclass, field
from typing import TYPE_CHECKING, Any
from remove_ai_watermarks._internal.schema import require_schema_version
if TYPE_CHECKING:
from pathlib import Path
logger = logging.getLogger(__name__)
# Analysis resolution. Every statistic here is scale-robust, and a 2048px cap keeps
# the FFT and the sliding-window residual bounded on a 100 MP input.
MAX_SIDE = 2048
# The eight lowest-frequency AC positions of the 8x8 block DCT, zig-zag order.
AC_POSITIONS = ((0, 1), (1, 0), (1, 1), (0, 2), (2, 0), (2, 1), (1, 2), (0, 3))
FFT_BANDS = 8
# A Bayer CFA shows as symmetric peaks at half the Nyquist on the diagonals.
BAYER_OFFSETS = ((1, 1), (1, -1))
INSTALL_HINT = "install the pixel extra: uv add 'remove-ai-watermarks[pixels]'"
PIXEL_EVIDENCE_SCHEMA_VERSION = 1
@dataclass(frozen=True)
class PixelEvidence:
"""Pixel statistics for one image, and the spatial artifacts behind them.
``decode`` carries the source dimensions, or ``{"error": ...}`` when the image
could not be decoded -- in which case every other field is empty. A family is also
empty when the image is too small for it (the block DCT needs 8x8, the FFT 32x32,
the residual 3x3), so a caller must treat every field as optional rather than
assume a fixed feature width.
"""
path: Path
decode: dict[str, Any]
dct: dict[str, Any] = field(default_factory=dict[str, Any])
fft: dict[str, Any] = field(default_factory=dict[str, Any])
noise: dict[str, Any] = field(default_factory=dict[str, Any])
ela: dict[str, Any] = field(default_factory=dict[str, Any])
gradient: dict[str, Any] = field(default_factory=dict[str, Any])
color: dict[str, Any] = field(default_factory=dict[str, Any])
# Identifies the source image; see the module note. Empty unless asked for.
artifacts: dict[str, Any] = field(default_factory=dict[str, Any])
# Opt-in timings for callers measuring pipeline latency. Empty by default so
# repeated evidence collection remains value-deterministic.
timing_ms: dict[str, float] = field(default_factory=dict[str, float])
@property
def decoded(self) -> bool:
"""False when the source could not be decoded at all."""
return "error" not in self.decode
@property
def status(self) -> str:
"""``complete``, ``partial`` for a failed family, or ``error`` on decode."""
if not self.decoded:
return "error"
sections = (self.dct, self.fft, self.noise, self.ela, self.gradient, self.color, self.artifacts)
return "partial" if any("error" in section for section in sections) else "complete"
def to_dict(
self,
*,
schema_version: int = PIXEL_EVIDENCE_SCHEMA_VERSION,
) -> dict[str, Any]:
"""Return the selected JSON-safe transport schema without a local path."""
schema_version = require_schema_version(
schema_version,
contract="pixel evidence",
supported=(1,),
)
return {
"schema_version": schema_version,
"status": self.status,
"decode": dict(self.decode),
"dct": dict(self.dct),
"fft": dict(self.fft),
"noise": dict(self.noise),
"ela": dict(self.ela),
"gradient": dict(self.gradient),
"color": dict(self.color),
"artifacts": dict(self.artifacts),
"timing_ms": dict(self.timing_ms),
}
def is_available() -> bool:
"""True when the optional pixel dependencies are installed."""
from remove_ai_watermarks.optional_deps import module_available
return module_available("numpy")
def _numpy() -> Any:
from remove_ai_watermarks.optional_deps import module_available
if not module_available("numpy"):
raise RuntimeError(f"Pixel evidence needs numpy -- {INSTALL_HINT}")
import numpy as np
return np
def _dct_matrix(np: Any, n: int = 8) -> Any:
"""Orthonormal n x n DCT-II basis: M[i, j] = cos(pi (2j + 1) i / 2n)."""
i = np.arange(n)[:, None]
j = np.arange(n)[None, :]
m = np.cos(np.pi * (2 * j + 1) * i / (2 * n))
m[0, :] *= 1 / np.sqrt(2)
return m * np.sqrt(2 / n)
def read_gray(image_path: Path) -> tuple[Any, Any, dict[str, Any]]:
"""Decode to float32 grayscale (and RGB for color stats), downscaled.
Pillow, not cv2, and the source dimensions are recorded BEFORE the downscale.
"""
np = _numpy()
from PIL import Image
from remove_ai_watermarks import image_io
try:
image_io._register_heif() # pyright: ignore[reportPrivateUsage]
with Image.open(image_path) as img:
info: dict[str, Any] = {"width": img.width, "height": img.height}
if max(img.size) > MAX_SIDE:
img.thumbnail((MAX_SIDE, MAX_SIDE), Image.Resampling.LANCZOS)
rgb = np.asarray(img.convert("RGB"), dtype=np.float32)
gray = np.asarray(img.convert("L"), dtype=np.float32)
except Exception as exc:
logger.debug("pixel decode failed for %s: %s", image_path, exc)
# Exception text from Pillow commonly embeds the absolute source path.
# Keep that detail in the log, not in the pathless transport contract.
return None, None, {"error": type(exc).__name__}
return gray, rgb, info
def dct_features(gray: Any) -> dict[str, Any]:
"""AC coefficient histograms over the 8x8 block DCT + Benford deviation."""
np = _numpy()
height, width = gray.shape
h8, w8 = height // 8 * 8, width // 8 * 8
if h8 < 8 or w8 < 8:
return {}
basis = _dct_matrix(np)
bins = np.linspace(-20.5, 20.5, 22)
blocks = gray[:h8, :w8].reshape(h8 // 8, 8, w8 // 8, 8).swapaxes(1, 2)
rows = basis[[row for row, _ in AC_POSITIONS]]
columns = basis[[column for _, column in AC_POSITIONS]]
coeff = np.einsum("ki,abij,kj->abk", rows, blocks, columns)
hists = []
lead_vals: list[Any] = []
for index in range(len(AC_POSITIONS)):
values = coeff[:, :, index].ravel()
hists.append(np.histogram(values, bins=bins)[0].tolist())
lead_vals.append(np.abs(values))
out: dict[str, Any] = {"dct_ac_hist": hists}
flat = np.abs(np.concatenate(lead_vals))
flat = flat[flat >= 1]
if flat.size > 100:
leading = (flat / 10 ** np.floor(np.log10(flat))).astype(int)
leading = leading[(leading >= 1) & (leading <= 9)]
if leading.size > 100:
observed = np.bincount(leading, minlength=10)[1:10] / leading.size
benford = np.log10(1 + 1 / np.arange(1, 10))
out["benford_mad"] = float(np.abs(observed - benford).mean())
return out
def noise_residual_map(gray: Any) -> Any:
"""High-pass residual, the map the noise statistics are computed from."""
np = _numpy()
from numpy.lib.stride_tricks import sliding_window_view
if gray.shape[0] < 3 or gray.shape[1] < 3:
return None
kernel = np.array([[-1.0, -1.0, -1.0], [-1.0, 8.0, -1.0], [-1.0, -1.0, -1.0]])
height, width = gray.shape
# kernel is float64, so the residual is float64 like the unchunked form
out = np.empty((height - 2, width - 2), dtype=np.float64)
# Row-chunked: the (window * kernel) temporary is ~150 MB at 2048px if
# materialized whole. Per-element 9-tap sums are computed in the same order,
# so the result is bit-identical to the unchunked form.
for y0 in range(0, height - 2, 256):
y1 = min(y0 + 256, height - 2)
window = sliding_window_view(gray[y0 : y1 + 2], (3, 3))
out[y0:y1] = (window * kernel).sum(axis=(-1, -2))
return out
def noise_features(residual: Any) -> dict[str, Any]:
"""High-pass residual std and kurtosis."""
flat = residual.ravel()
std = float(flat.std())
if std < 1e-9:
return {"noise_std": 0.0, "noise_kurtosis": 0.0}
z = (flat - flat.mean()) / std
return {"noise_std": std, "noise_kurtosis": float((z**4).mean() - 3.0)}
def fft_decompose(gray: Any) -> tuple[Any, Any] | None:
"""Log-magnitude (fftshifted) and phase of the image spectrum."""
np = _numpy()
if min(gray.shape) < 32:
return None
spectrum = np.fft.fftshift(np.fft.fft2(gray - gray.mean()))
return np.log1p(np.abs(spectrum)), np.angle(spectrum)
def fft_features(mag: Any) -> dict[str, Any]:
"""Radial magnitude band energies (no phase) + CFA periodicity peaks."""
np = _numpy()
height, width = mag.shape
cy, cx = height // 2, width // 2
# 1D broadcast instead of an mgrid: saves ~160 MB of int64 temporaries at
# 2048px. The squares are exact in float64 (values < 2^53), so band means
# are identical to the mgrid form.
r2y = (np.arange(height, dtype=np.float64) - cy) ** 2
r2x = (np.arange(width, dtype=np.float64) - cx) ** 2
radius = np.sqrt(r2y[:, None] + r2x[None, :])
r_max = radius.max()
bands = []
for index in range(FFT_BANDS):
mask = (radius >= r_max * index / FFT_BANDS) & (radius < r_max * (index + 1) / FFT_BANDS)
bands.append(float(mag[mask].mean()) if mask.any() else 0.0)
peaks = []
for dy, dx in BAYER_OFFSETS:
y, x = cy + dy * (height // 4), cx + dx * (width // 4)
neighborhood = mag[y - 2 : y + 3, x - 2 : x + 3]
peaks.append(float(neighborhood.max() - mag.mean()))
return {"fft_band_energy": bands, "cfa_peaks": peaks, "cfa_peak": max(peaks)}
def ela_map(rgb: Any) -> Any:
"""Absolute per-pixel error after a quality-90 JPEG re-save."""
np = _numpy()
from PIL import Image
try:
buffer = io.BytesIO()
Image.fromarray(rgb.astype(np.uint8)).save(buffer, "JPEG", quality=90)
buffer.seek(0)
resaved = np.asarray(Image.open(buffer).convert("RGB"), dtype=np.float32)
except Exception as exc:
logger.debug("ELA re-save failed: %s", exc)
return None
if resaved.shape != rgb.shape:
return None
return np.abs(rgb - resaved).mean(axis=-1)
def ela_features(err: Any) -> dict[str, Any]:
"""Error-level stats after a quality-90 JPEG re-save."""
np = _numpy()
return {"ela_mean": float(err.mean()), "ela_p95": float(np.percentile(err, 95))}
def gradient_features(gray: Any) -> dict[str, Any]:
np = _numpy()
gy, gx = np.gradient(gray)
mag = np.sqrt(gx**2 + gy**2)
hist = np.histogram(mag, bins=10, range=(0, 255))[0].tolist()
laplacian = np.gradient(gy, axis=0) + np.gradient(gx, axis=1)
return {"gradient_hist": hist, "laplacian_var": float(laplacian.var())}
def color_features(rgb: Any) -> dict[str, Any]:
np = _numpy()
small = rgb[::4, ::4] # decimate; the histogram is position-blind anyway
bins = (small / 256 * 4).astype(int).clip(0, 3)
index = bins[..., 0] * 16 + bins[..., 1] * 4 + bins[..., 2]
hist = np.bincount(index.ravel(), minlength=64).tolist()
mx = small.max(axis=-1)
mn = small.min(axis=-1)
saturation = np.where(mx > 0, (mx - mn) / np.maximum(mx, 1e-6), 0)
return {
"color_hist_4x4x4": hist,
"saturation_mean": float(saturation.mean()),
"value_mean": float(mx.mean() / 255),
}
def perceptual_hash(gray: Any) -> str:
"""64-bit DCT perceptual hash. Identifies an image; see the module note."""
np = _numpy()
from PIL import Image
small = np.asarray(Image.fromarray(gray.astype(np.float32), mode="F").resize((32, 32), Image.Resampling.LANCZOS))
basis = _dct_matrix(np, 32)
low_basis = basis[:8]
low = (low_basis @ small @ low_basis.T).ravel()[1:] # drop DC
bits = low > np.median(low)
return f"{int(''.join('1' if bit else '0' for bit in bits), 2):016x}"
def _coarse(np: Any, arr: Any, side: int = 64) -> Any:
"""Downscale a 2D map to at most ``side`` on the long edge."""
from PIL import Image
height, width = arr.shape
if max(height, width) <= side:
return arr
img = Image.fromarray(arr.astype(np.float32), mode="F")
img.thumbnail((side, side), Image.Resampling.BILINEAR)
return np.asarray(img)
def _array_payload(arr: Any) -> dict[str, Any]:
return {
"shape": list(arr.shape),
"dtype": str(arr.dtype),
"base64": base64.b64encode(arr.tobytes()).decode("ascii"),
}
def spatial_artifacts(gray: Any, rgb: Any, *, ela: Any, residual: Any, phase: Any) -> dict[str, Any]:
"""Perceptual hash, thumbnail, and coarse ELA / residual / phase maps.
These identify the source image rather than describe it -- see the module note.
The maps are the ones the statistics were computed from, passed in rather than
recomputed.
"""
np = _numpy()
from PIL import Image
out: dict[str, Any] = {"phash": perceptual_hash(gray)}
thumbnail = Image.fromarray(rgb.astype(np.uint8))
thumbnail.thumbnail((128, 128), Image.Resampling.LANCZOS)
buffer = io.BytesIO()
thumbnail.save(buffer, "JPEG", quality=70)
out["thumbnail_jpeg_b64"] = base64.b64encode(buffer.getvalue()).decode("ascii")
if ela is not None:
out["ela_map"] = _array_payload(_coarse(np, ela))
if residual is not None:
clipped = np.clip(residual / 4.0, -1, 1)
out["noise_residual"] = _array_payload(_coarse(np, (clipped * 127).astype(np.int8)))
if phase is not None:
out["fft_phase"] = _array_payload(_coarse(np, phase.astype(np.float32), 32))
return out
def extract_pixel_evidence(image_path: Path, *, artifacts: bool = False, timings: bool = False) -> PixelEvidence:
"""Measure every pixel-statistic family for one image in a single decode.
The image is decoded ONCE and the intermediate maps (high-pass residual, ELA
error, FFT magnitude and phase) are computed once and shared, because the
residual's sliding window and the ELA re-save are the two expensive steps and
each family would otherwise redo them.
A family that fails or does not apply is left empty rather than raising: an
undecodable file, or one too small for the block DCT, still returns a
:class:`PixelEvidence` whose ``decoded`` / empty fields say so. Missing numpy is
the one hard error, since then nothing can be measured at all.
Args:
image_path: Path to the image. Any container Pillow can open.
artifacts: Also return the spatial layer -- perceptual hash, thumbnail and
coarse maps. Off by default: those identify the source image, so asking
for them is a decision the caller makes explicitly.
timings: Measure each stage and include rounded milliseconds in
:attr:`PixelEvidence.timing_ms`.
Returns:
A :class:`PixelEvidence`.
"""
started = time.perf_counter()
stage_started = started
measured: dict[str, float] = {}
gray, rgb, info = read_gray(image_path)
measured["decode"] = time.perf_counter() - stage_started
if gray is None or rgb is None:
measured["total"] = time.perf_counter() - started
timing_ms = {name: round(seconds * 1000, 1) for name, seconds in measured.items()} if timings else {}
return PixelEvidence(path=image_path, decode=info, timing_ms=timing_ms)
families: dict[str, dict[str, Any]] = {}
residual = None
stage_started = time.perf_counter()
try:
residual = noise_residual_map(gray)
families["noise"] = noise_features(residual) if residual is not None else {}
except Exception as exc:
logger.debug("pixel family noise failed for %s: %s", image_path, exc)
families["noise"] = {"error": type(exc).__name__}
measured["noise"] = time.perf_counter() - stage_started
spectrum = None
stage_started = time.perf_counter()
try:
spectrum = fft_decompose(gray)
families["fft"] = fft_features(spectrum[0]) if spectrum is not None else {}
except Exception as exc:
logger.debug("pixel family fft failed for %s: %s", image_path, exc)
families["fft"] = {"error": type(exc).__name__}
measured["fft"] = time.perf_counter() - stage_started
error = None
stage_started = time.perf_counter()
try:
error = ela_map(rgb)
families["ela"] = ela_features(error) if error is not None else {}
except Exception as exc:
logger.debug("pixel family ela failed for %s: %s", image_path, exc)
families["ela"] = {"error": type(exc).__name__}
measured["ela"] = time.perf_counter() - stage_started
for name, compute in (
("dct", lambda: dct_features(gray)),
("gradient", lambda: gradient_features(gray)),
("color", lambda: color_features(rgb)),
):
stage_started = time.perf_counter()
try:
families[name] = compute()
except Exception as exc: # one bad family must not lose the other five
logger.debug("pixel family %s failed for %s: %s", name, image_path, exc)
families[name] = {"error": type(exc).__name__}
measured[name] = time.perf_counter() - stage_started
if artifacts:
stage_started = time.perf_counter()
try:
families["artifacts"] = spatial_artifacts(
gray, rgb, ela=error, residual=residual, phase=spectrum[1] if spectrum is not None else None
)
except Exception as exc:
logger.debug("pixel artifacts failed for %s: %s", image_path, exc)
families["artifacts"] = {"error": type(exc).__name__}
measured["full_artifacts"] = time.perf_counter() - stage_started
measured["total"] = time.perf_counter() - started
timing_ms = {name: round(seconds * 1000, 1) for name, seconds in measured.items()} if timings else {}
return PixelEvidence(path=image_path, decode=info, timing_ms=timing_ms, **families)