mirror of
https://github.com/wiltodelta/remove-ai-watermarks.git
synced 2026-08-10 16:10:33 +02:00
Merge remote-tracking branch 'origin/main' into research/video-synthid-quality-groundwork
This commit is contained in:
@@ -32,7 +32,7 @@ _os.environ.setdefault("TRANSFORMERS_VERBOSITY", "error")
|
||||
_warnings.filterwarnings("ignore", message=r".*ImageProcessorFast.*")
|
||||
|
||||
|
||||
__version__ = "0.25.0"
|
||||
__version__ = "0.26.0"
|
||||
|
||||
__all__ = [
|
||||
"BatchSummary",
|
||||
|
||||
@@ -29,7 +29,9 @@ if TYPE_CHECKING:
|
||||
from typing import BinaryIO
|
||||
|
||||
_C2paReader: Any = None
|
||||
_C2paError: Any = None
|
||||
with contextlib.suppress(Exception):
|
||||
from c2pa import C2paError as _C2paError # pyright: ignore[reportMissingTypeStubs]
|
||||
from c2pa import Reader as _C2paReader # pyright: ignore[reportMissingTypeStubs]
|
||||
|
||||
_C2PA_READER_AVAILABLE = _C2paReader is not None
|
||||
@@ -48,10 +50,22 @@ def reader_available() -> bool:
|
||||
|
||||
|
||||
def _manifest_json_uncached(path: str) -> str | None:
|
||||
"""The manifest store as JSON, or None when this file has no readable manifest.
|
||||
|
||||
Two outcomes are routine and stay at debug: a file with no manifest (``try_create``
|
||||
returns None) and a container the reader does not support. ANY other failure is
|
||||
logged at warning, because the caller cannot tell the difference from the return
|
||||
value and the consequence is severe: the verdict silently falls back to the raw
|
||||
byte scan and can lose a high-confidence signal. The log line preserves the
|
||||
diagnostic context needed to investigate an intermittent reader failure.
|
||||
"""
|
||||
try:
|
||||
reader = _C2paReader.try_create(path)
|
||||
except _C2paError.NotSupported as error:
|
||||
logger.debug("C2PA reader does not support %s: %s", path, error)
|
||||
return None
|
||||
except Exception as error:
|
||||
logger.debug("C2PA reader rejected %s: %s", path, error)
|
||||
logger.warning("C2PA reader failed to open %s: %s: %s", path, type(error).__name__, error)
|
||||
return None
|
||||
if reader is None:
|
||||
return None
|
||||
@@ -59,7 +73,9 @@ def _manifest_json_uncached(path: str) -> str | None:
|
||||
with reader:
|
||||
return cast("str", reader.json())
|
||||
except Exception as error:
|
||||
logger.debug("C2PA reader could not serialize %s: %s", path, error)
|
||||
# The reader opened the file, so a manifest is there; failing to serialize it
|
||||
# is never routine.
|
||||
logger.warning("C2PA reader could not serialize %s: %s: %s", path, type(error).__name__, error)
|
||||
return None
|
||||
|
||||
|
||||
|
||||
@@ -20,6 +20,9 @@ AI_KEYWORDS = _tokens(
|
||||
|
||||
PNG_SIGNATURE = b"\x89PNG\r\n\x1a\n"
|
||||
C2PA_CHUNK_TYPE = b"caBX"
|
||||
PNG_METADATA_CHUNKS = frozenset({b"tEXt", b"iTXt", b"zTXt", b"eXIf", b"iCCP"})
|
||||
RIFF_METADATA_CHUNKS = frozenset({b"EXIF", b"XMP ", b"ICCP", b"C2PA"})
|
||||
RIFF_CODED_IMAGE_CHUNKS = frozenset({b"VP8 ", b"VP8L", b"ALPH", b"ANMF"})
|
||||
C2PA_SIGNATURES = tuple(
|
||||
token.encode() for token in _tokens("c2pa|C2PA|jumb|jumd|JUMBF|jumbf|cbor|contentcreds|digid|assertions|manifest")
|
||||
)
|
||||
|
||||
@@ -64,7 +64,7 @@ _AI_LABEL_MARKERS: tuple[bytes, ...] = AIGC_MARKERS + IPTC_AI_MARKERS + IPTC_AI_
|
||||
# blanked in place (see ``blank_ai_xmp_packets``).
|
||||
_XMP_PACKET_RE = re.compile(rb"<\?xpacket begin=.*?<\?xpacket end=[^>]*?\?>", re.DOTALL)
|
||||
_STREAM_COPY_BYTES = 1024 * 1024
|
||||
_STREAM_SCAN_BYTES = 4 * 1024 * 1024
|
||||
STREAM_SCAN_BYTES = 4 * 1024 * 1024
|
||||
|
||||
|
||||
# TC260-PG-20257A stores an MP4/MOV label as an ``AIGC`` key in
|
||||
@@ -133,7 +133,7 @@ def _read_box_header(
|
||||
return end, box_type, payload_off
|
||||
|
||||
|
||||
def _iter_file_boxes(
|
||||
def iter_file_boxes(
|
||||
stream: BinaryIO,
|
||||
start: int,
|
||||
end: int,
|
||||
@@ -193,17 +193,17 @@ def _tc260_aigc_regions(
|
||||
Each tuple is ``(key_start, key_end, value_start, value_end, value)``.
|
||||
"""
|
||||
regions: list[tuple[int, int, int, int, bytes]] = []
|
||||
for _moov_start, moov_end, moov_type, moov_payload in _iter_file_boxes(stream, 0, file_size):
|
||||
for _moov_start, moov_end, moov_type, moov_payload in iter_file_boxes(stream, 0, file_size):
|
||||
if moov_type != b"moov":
|
||||
continue
|
||||
for _udta_start, udta_end, udta_type, udta_payload in _iter_file_boxes(
|
||||
for _udta_start, udta_end, udta_type, udta_payload in iter_file_boxes(
|
||||
stream,
|
||||
moov_payload,
|
||||
moov_end,
|
||||
):
|
||||
if udta_type != b"udta":
|
||||
continue
|
||||
for _meta_start, meta_end, meta_type, meta_payload in _iter_file_boxes(
|
||||
for _meta_start, meta_end, meta_type, meta_payload in iter_file_boxes(
|
||||
stream,
|
||||
udta_payload,
|
||||
udta_end,
|
||||
@@ -212,7 +212,7 @@ def _tc260_aigc_regions(
|
||||
continue
|
||||
keys: dict[int, tuple[int, int]] = {}
|
||||
ilst_boxes: list[tuple[int, int]] = []
|
||||
for _child_start, child_end, child_type, child_payload in _iter_file_boxes(
|
||||
for _child_start, child_end, child_type, child_payload in iter_file_boxes(
|
||||
stream,
|
||||
meta_payload + 4,
|
||||
meta_end,
|
||||
@@ -224,7 +224,7 @@ def _tc260_aigc_regions(
|
||||
if not keys:
|
||||
continue
|
||||
for ilst_payload, ilst_end in ilst_boxes:
|
||||
for _item_start, item_end, item_type, item_payload in _iter_file_boxes(
|
||||
for _item_start, item_end, item_type, item_payload in iter_file_boxes(
|
||||
stream,
|
||||
ilst_payload,
|
||||
ilst_end,
|
||||
@@ -233,7 +233,7 @@ def _tc260_aigc_regions(
|
||||
key_span = keys.get(index)
|
||||
if key_span is None:
|
||||
continue
|
||||
for _data_start, data_end, data_type, data_payload in _iter_file_boxes(
|
||||
for _data_start, data_end, data_type, data_payload in iter_file_boxes(
|
||||
stream,
|
||||
item_payload,
|
||||
item_end,
|
||||
@@ -425,7 +425,7 @@ def strip_isobmff_media_file(
|
||||
source: str | Path,
|
||||
output: str | Path,
|
||||
*,
|
||||
max_box_scan: int = _STREAM_SCAN_BYTES,
|
||||
max_box_scan: int = STREAM_SCAN_BYTES,
|
||||
) -> tuple[int, int]:
|
||||
"""Stream-copy an MP4/MOV/M4A while removing supported AI metadata.
|
||||
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
"""Shared validation for versioned JSON transport contracts."""
|
||||
|
||||
from collections.abc import Collection
|
||||
|
||||
|
||||
def require_schema_version(
|
||||
value: object,
|
||||
*,
|
||||
contract: str,
|
||||
supported: Collection[int],
|
||||
) -> int:
|
||||
"""Return an explicitly supported integer schema version or raise."""
|
||||
if type(value) is not int or value not in supported:
|
||||
versions = ", ".join(str(version) for version in sorted(supported))
|
||||
raise ValueError(f"Unsupported {contract} schema: {value!r}; supported versions: {versions}")
|
||||
return value
|
||||
@@ -0,0 +1,861 @@
|
||||
"""Collect JSON-safe metadata and container forensics for one media file.
|
||||
|
||||
The collector is deliberately evidence-only: it preserves raw EXIF, IPTC, C2PA,
|
||||
container metadata, encoder structure, hashes, timestamps, and bounded binary
|
||||
payloads without deciding whether the content is AI-generated. Provenance verdicts
|
||||
and pixel statistics are separate library stages.
|
||||
"""
|
||||
|
||||
import base64
|
||||
import contextlib
|
||||
import hashlib
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import plistlib
|
||||
import re
|
||||
import struct
|
||||
import zlib
|
||||
from pathlib import Path
|
||||
from typing import Any, cast
|
||||
|
||||
import piexif
|
||||
from PIL import Image
|
||||
from PIL.IptcImagePlugin import getiptcinfo
|
||||
|
||||
from remove_ai_watermarks import image_io
|
||||
from remove_ai_watermarks._internal.constants import (
|
||||
PNG_METADATA_CHUNKS,
|
||||
RIFF_CODED_IMAGE_CHUNKS,
|
||||
RIFF_METADATA_CHUNKS,
|
||||
)
|
||||
from remove_ai_watermarks._internal.isobmff import (
|
||||
C2PA_BOX_TYPES,
|
||||
STREAM_SCAN_BYTES,
|
||||
iter_file_boxes,
|
||||
)
|
||||
from remove_ai_watermarks._internal.schema import require_schema_version
|
||||
from remove_ai_watermarks.metadata import QUICK_SCAN_BYTES
|
||||
from remove_ai_watermarks.metadata_record import HEAD_WINDOW
|
||||
|
||||
__all__ = [
|
||||
"FORENSIC_METADATA_RECORD_TYPE",
|
||||
"FORENSIC_METADATA_SCHEMA_VERSION",
|
||||
"SUPPORTED_EXTENSIONS",
|
||||
"collect_forensic_metadata",
|
||||
]
|
||||
|
||||
SUPPORTED_EXTENSIONS = {
|
||||
".png",
|
||||
".jpg",
|
||||
".jpeg",
|
||||
".webp",
|
||||
".heic",
|
||||
".heif",
|
||||
".avif",
|
||||
".tif",
|
||||
".tiff",
|
||||
".bmp",
|
||||
".gif",
|
||||
# video/px containers: no pixel decode, but C2PA reads them (Sora/Veo
|
||||
# carry C2PA manifests) and the byte scans still apply
|
||||
".mp4",
|
||||
".mov",
|
||||
".m4v",
|
||||
".jxl",
|
||||
}
|
||||
|
||||
FORENSIC_METADATA_SCHEMA_VERSION = 1
|
||||
FORENSIC_METADATA_RECORD_TYPE = "forensic_metadata"
|
||||
|
||||
_B64_CAP = 1 << 20 # 1 MB safety ceiling per embedded blob
|
||||
_TEXT_CAP = 1 << 20 # decoded PNG text ceiling per chunk
|
||||
# Preserve enough top-level ISOBMFF uuid/jumb payload data for downstream
|
||||
# provenance algorithms without requiring them to reopen the source file.
|
||||
_PROVENANCE_B64_CAP = STREAM_SCAN_BYTES
|
||||
_RAW_SCAN_HEAD = HEAD_WINDOW
|
||||
_RAW_SCAN_TAIL = QUICK_SCAN_BYTES
|
||||
|
||||
|
||||
def _safe_str(v: Any) -> str:
|
||||
try:
|
||||
return str(v)
|
||||
except Exception:
|
||||
return repr(v)
|
||||
|
||||
|
||||
def _b64(b: bytes, *, cap: int = _B64_CAP) -> str:
|
||||
"""Legacy base64 value, with an explicit marker when the payload is capped."""
|
||||
encoded = base64.b64encode(b[:cap]).decode("ascii")
|
||||
return encoded + f"...TRUNCATED({len(b)} bytes total)" if len(b) > cap else encoded
|
||||
|
||||
|
||||
def _decode_exif_value(v: Any) -> Any:
|
||||
"""Make a piexif value JSON-safe; bytes are kept in full as hex."""
|
||||
if isinstance(v, bytes):
|
||||
if len(v) <= 64:
|
||||
try:
|
||||
return v.decode("utf-8", "strict")
|
||||
except (UnicodeDecodeError, ValueError):
|
||||
return f"hex:{v.hex()}"
|
||||
return f"hex:{v.hex()}"
|
||||
if isinstance(v, tuple | list):
|
||||
sequence = cast("list[Any] | tuple[Any, ...]", v)
|
||||
return [_decode_exif_value(item) for item in sequence]
|
||||
return v
|
||||
|
||||
|
||||
def read_full_exif(
|
||||
path: Path, exif_blob: bytes | None = None, data: bytes | None = None
|
||||
) -> tuple[dict[str, Any], bytes | None]:
|
||||
"""All EXIF IFDs with decoded tag names (piexif, no re-encode), plus the
|
||||
raw embedded-thumbnail bytes for the caller's own thumbnail forensics.
|
||||
|
||||
``exif_blob`` is the PIL-exposed EXIF blob (PNG/WebP/HEIC path) so the
|
||||
caller's single Image.open is not repeated here. ``data`` is the
|
||||
already-read file bytes so piexif does not re-read the file."""
|
||||
try:
|
||||
exif: dict[str, Any] = piexif.load(data) if data is not None else piexif.load(str(path))
|
||||
except Exception:
|
||||
if not exif_blob:
|
||||
return {}, None
|
||||
try:
|
||||
exif = piexif.load(exif_blob)
|
||||
except Exception as exc:
|
||||
return {"error": _safe_str(exc)}, None
|
||||
out: dict[str, Any] = {}
|
||||
thumbnail: bytes | None = None
|
||||
for ifd, tags in exif.items():
|
||||
if ifd == "thumbnail":
|
||||
thumbnail = tags if isinstance(tags, bytes) else None
|
||||
out["thumbnail"] = f"{len(tags)} bytes" if isinstance(tags, bytes) else None
|
||||
continue
|
||||
if not isinstance(tags, dict):
|
||||
continue
|
||||
all_tag_names = cast("dict[str, dict[int, dict[str, Any]]]", getattr(piexif, "TAGS", {}))
|
||||
tag_names = all_tag_names.get(ifd, {})
|
||||
decoded: dict[str, Any] = {}
|
||||
for tag, value in cast("dict[int, Any]", tags).items():
|
||||
name = str(tag_names.get(tag, {}).get("name", f"tag_{tag}"))
|
||||
if name == "MakerNote" and isinstance(value, bytes):
|
||||
# full hex, no cap: measured on real uploads, Apple is ~2 KB
|
||||
# but Canon reaches 28 KB and Sony 38 KB (AF data, serials,
|
||||
# embedded previews) -- a cap would silently drop exactly the
|
||||
# camera-original evidence this scan exists to preserve
|
||||
decoded[name] = f"hex:{value.hex()}"
|
||||
else:
|
||||
decoded[name] = _decode_exif_value(value)
|
||||
out[ifd] = decoded
|
||||
return out, thumbnail
|
||||
|
||||
|
||||
def _png_text_decode(ctype: str, body: bytes) -> str:
|
||||
"""Decode a tEXt/zTXt/iTXt chunk, inflating zlib where used.
|
||||
|
||||
The compressed forms are where ComfyUI / Automatic1111 hide the
|
||||
generation workflow and prompt, so skipping the inflate would drop
|
||||
the strongest AI-provenance text a PNG can carry."""
|
||||
if ctype == "tEXt":
|
||||
suffix = b"...TRUNCATED" if len(body) > _TEXT_CAP else b""
|
||||
return (body[:_TEXT_CAP] + suffix).decode("utf-8", "replace")
|
||||
if ctype == "zTXt":
|
||||
nul = body.find(b"\x00")
|
||||
if nul == -1:
|
||||
return body[:_TEXT_CAP].decode("utf-8", "replace")
|
||||
keyword = body[:nul].decode("latin-1", "replace")
|
||||
# body[nul+1] = compression method (0 = zlib)
|
||||
try:
|
||||
inflater = zlib.decompressobj()
|
||||
decoded = inflater.decompress(body[nul + 2 :], _TEXT_CAP + 1)
|
||||
suffix = "...TRUNCATED" if len(decoded) > _TEXT_CAP else ""
|
||||
text = decoded[:_TEXT_CAP].decode("utf-8", "replace") + suffix
|
||||
except zlib.error:
|
||||
text = body[:_TEXT_CAP].decode("utf-8", "replace")
|
||||
return f"{keyword}\x00{text}"
|
||||
# iTXt: keyword\0 compflag(1) compmethod(1) lang\0 translated\0 text
|
||||
parts = body.split(b"\x00", 1)
|
||||
if len(parts) < 2:
|
||||
return body[:_TEXT_CAP].decode("utf-8", "replace")
|
||||
keyword = parts[0].decode("latin-1", "replace")
|
||||
rest = parts[1]
|
||||
if len(rest) < 2:
|
||||
return body[:_TEXT_CAP].decode("utf-8", "replace")
|
||||
compflag = rest[0]
|
||||
tail = rest[2:]
|
||||
for _ in range(2): # skip language tag and translated keyword
|
||||
nul = tail.find(b"\x00")
|
||||
if nul == -1:
|
||||
return body[:_TEXT_CAP].decode("utf-8", "replace")
|
||||
tail = tail[nul + 1 :]
|
||||
if compflag:
|
||||
with contextlib.suppress(zlib.error):
|
||||
inflater = zlib.decompressobj()
|
||||
tail = inflater.decompress(tail, _TEXT_CAP + 1)
|
||||
if len(tail) > _TEXT_CAP:
|
||||
tail = tail[:_TEXT_CAP] + b"...TRUNCATED"
|
||||
return f"{keyword}\x00{tail.decode('utf-8', 'replace')}"
|
||||
|
||||
|
||||
def read_png_chunks(data: bytes) -> tuple[list[dict[str, Any]], bytes]:
|
||||
"""Every PNG chunk in order (type, length; text chunks decoded and
|
||||
inflated, binary chunks as base64) plus the post-IEND trailer bytes."""
|
||||
chunks: list[dict[str, Any]] = []
|
||||
post_iend = b""
|
||||
try:
|
||||
pos = 8
|
||||
while pos + 12 <= len(data):
|
||||
length = struct.unpack(">I", data[pos : pos + 4])[0]
|
||||
ctype = data[pos + 4 : pos + 8].decode("latin-1")
|
||||
body = data[pos + 8 : pos + 8 + length]
|
||||
entry: dict[str, Any] = {"type": ctype, "length": length}
|
||||
if ctype in ("tEXt", "zTXt", "iTXt"):
|
||||
entry["text"] = _png_text_decode(ctype, body)
|
||||
if entry["text"].startswith("XML:com.adobe.xmp"):
|
||||
entry["kind"] = "xmp"
|
||||
elif ctype == "tIME" and length == 7:
|
||||
y, mo, d, h, mi, s = struct.unpack(">HBBBBB", body)
|
||||
entry["time"] = f"{y:04d}-{mo:02d}-{d:02d}T{h:02d}:{mi:02d}:{s:02d}Z"
|
||||
elif ctype == "gAMA" and length == 4:
|
||||
entry["gamma"] = struct.unpack(">I", body)[0] / 100000
|
||||
elif ctype == "sRGB" and length == 1:
|
||||
entry["rendering_intent"] = body[0]
|
||||
elif ctype == "iCCP":
|
||||
nul = body.find(b"\x00")
|
||||
if nul > 0:
|
||||
entry["profile_name"] = body[:nul].decode("latin-1", "replace")
|
||||
entry["base64"] = _b64(body)
|
||||
elif ctype == "iDOT":
|
||||
# present in iOS/macOS screenshots
|
||||
entry["apple_screenshot_marker"] = True
|
||||
elif ctype in ("IHDR", "IDAT"):
|
||||
pass # pixel-data / header chunks: length is signal enough
|
||||
elif length:
|
||||
entry["base64"] = _b64(body)
|
||||
chunks.append(entry)
|
||||
pos += 12 + length
|
||||
if ctype == "IEND":
|
||||
post_iend = data[pos:]
|
||||
break
|
||||
except Exception as exc:
|
||||
chunks.append({"error": _safe_str(exc)})
|
||||
return chunks, post_iend
|
||||
|
||||
|
||||
def _set_jpeg_trailer(result: dict[str, Any], data: bytes, eoi: int) -> None:
|
||||
"""Preserve bytes after JPEG EOI for Samsung Galaxy AI detection."""
|
||||
trailer = data[eoi + 2 :]
|
||||
result["post_eoi_bytes"] = len(trailer)
|
||||
if trailer:
|
||||
result["post_eoi_base64"] = _b64(trailer)
|
||||
|
||||
|
||||
def read_jpeg_segments(data: bytes) -> dict[str, Any]:
|
||||
"""Every JPEG APP segment in order, plus post-EOI trailer size.
|
||||
|
||||
XMP APP1 segments are kept as full text; every other segment body is
|
||||
kept as full base64 (1 MB ceiling per segment).
|
||||
"""
|
||||
result: dict[str, Any] = {"segments": [], "post_eoi_bytes": 0}
|
||||
try:
|
||||
pos = 2
|
||||
while pos + 4 <= len(data):
|
||||
if data[pos] != 0xFF:
|
||||
break
|
||||
marker = data[pos + 1]
|
||||
if marker == 0xD9: # EOI
|
||||
_set_jpeg_trailer(result, data, pos)
|
||||
break
|
||||
if marker == 0xDA: # SOS: entropy-coded data follows
|
||||
eoi = data.rfind(b"\xff\xd9")
|
||||
if eoi != -1:
|
||||
_set_jpeg_trailer(result, data, eoi)
|
||||
break
|
||||
if not (0xE0 <= marker <= 0xEF):
|
||||
length = struct.unpack(">H", data[pos + 2 : pos + 4])[0]
|
||||
pos += 2 + length
|
||||
continue
|
||||
length = struct.unpack(">H", data[pos + 2 : pos + 4])[0]
|
||||
body = data[pos + 4 : pos + 2 + length]
|
||||
name = f"APP{marker - 0xE0}"
|
||||
entry: dict[str, Any] = {"marker": name, "length": length}
|
||||
# Adobe JPEG XMP APP1 magic (namespace URI in the packet, not a request).
|
||||
if body.startswith(b"http://ns.adobe.com/xap/1.0/\x00"): # NOSONAR
|
||||
entry["kind"] = "xmp"
|
||||
entry["text"] = body[29:].decode("utf-8", "replace")
|
||||
elif name == "APP2" and body.startswith(b"MPF\x00"):
|
||||
# Multi-Picture Format: Ultra HDR gain map, Samsung dual shot
|
||||
entry["kind"] = "mpf"
|
||||
entry["base64"] = _b64(body)
|
||||
elif name == "APP2" and body.startswith(b"ICC_PROFILE"):
|
||||
entry["kind"] = "icc"
|
||||
entry["base64"] = _b64(body)
|
||||
elif name == "APP2" and body.startswith(b"FPXR"):
|
||||
entry["kind"] = "flashpix"
|
||||
entry["base64"] = _b64(body)
|
||||
elif name == "APP11":
|
||||
entry["kind"] = "c2pa_or_jumbf"
|
||||
# the parsed manifest is in c2pa_store, but the raw JUMBF
|
||||
# also carries assertion thumbnails the JSON may omit
|
||||
entry["base64"] = _b64(body)
|
||||
elif body.startswith(b"Exif\x00\x00"):
|
||||
entry["kind"] = "exif"
|
||||
entry["base64"] = _b64(body)
|
||||
elif body.startswith(b"Photoshop 3.0\x00"):
|
||||
entry["kind"] = "iptc_iim"
|
||||
entry["base64"] = _b64(body)
|
||||
else:
|
||||
entry["base64"] = _b64(body)
|
||||
result["segments"].append(entry)
|
||||
pos += 2 + length
|
||||
except Exception as exc:
|
||||
result["error"] = _safe_str(exc)
|
||||
return result
|
||||
|
||||
|
||||
def read_pil_info(path: Path) -> tuple[dict[str, Any], dict[str, Any], bytes | None]:
|
||||
"""One Image.open serving all PIL-derived data: container basics,
|
||||
img.info passthrough (XMP, comments), the IPTC-IIM dataset, and the
|
||||
raw EXIF blob (for the caller's piexif parse on PNG/WebP/HEIC)."""
|
||||
out: dict[str, Any] = {}
|
||||
iptc: dict[str, Any] = {}
|
||||
exif_blob: bytes | None = None
|
||||
try:
|
||||
with Image.open(path) as img:
|
||||
out["format"] = img.format
|
||||
out["mode"] = img.mode
|
||||
out["width"], out["height"] = img.size
|
||||
out["n_frames"] = getattr(img, "n_frames", 1)
|
||||
dpi = img.info.get("dpi")
|
||||
if dpi:
|
||||
out["dpi"] = [round(float(d), 2) for d in dpi]
|
||||
icc = img.info.get("icc_profile")
|
||||
if icc:
|
||||
out["icc_profile"] = {
|
||||
"length": len(icc),
|
||||
# header: profile class, color space, PCS (bytes 12-24)
|
||||
"header_hex": icc[12:24].hex() if len(icc) >= 24 else "",
|
||||
"base64": _b64(icc),
|
||||
}
|
||||
blob = img.info.get("exif")
|
||||
if isinstance(blob, bytes):
|
||||
exif_blob = blob
|
||||
try:
|
||||
info = getiptcinfo(img)
|
||||
except Exception:
|
||||
info = None
|
||||
if info:
|
||||
iptc = {f"{k[0]}:{k[1]}": _decode_exif_value(v) for k, v in info.items()}
|
||||
for key, value in img.info.items():
|
||||
if key in ("icc_profile", "exif", "dpi"):
|
||||
continue
|
||||
if isinstance(value, bytes):
|
||||
try:
|
||||
out[f"info:{key}"] = value.decode("utf-8", "strict")
|
||||
except (UnicodeDecodeError, ValueError):
|
||||
out[f"info:{key}"] = f"base64:{_b64(value)}"
|
||||
else:
|
||||
out[f"info:{key}"] = _safe_str(value)
|
||||
except Exception as exc:
|
||||
out["error"] = _safe_str(exc)
|
||||
return out, iptc, exif_blob
|
||||
|
||||
|
||||
def read_c2pa_store(path: Path) -> dict[str, Any]:
|
||||
"""Full C2PA manifest store through the package's cached reader."""
|
||||
from remove_ai_watermarks._internal.c2pa import read_manifest_store_json
|
||||
|
||||
raw = read_manifest_store_json(path)
|
||||
if raw is None:
|
||||
return {}
|
||||
try:
|
||||
value: Any = json.loads(raw)
|
||||
return (
|
||||
cast("dict[str, Any]", value)
|
||||
if isinstance(value, dict)
|
||||
else {"error": "C2PA manifest store is not an object"}
|
||||
)
|
||||
except (TypeError, ValueError) as exc:
|
||||
return {"error": _safe_str(exc)}
|
||||
|
||||
|
||||
def sniff_format(head: bytes) -> str:
|
||||
if head.startswith(b"\x89PNG"):
|
||||
return "png"
|
||||
if head.startswith(b"\xff\xd8"):
|
||||
return "jpeg"
|
||||
if head.startswith(b"RIFF") and head[8:12] == b"WEBP":
|
||||
return "webp"
|
||||
if head[:6] in (b"GIF87a", b"GIF89a"):
|
||||
return "gif"
|
||||
if head.startswith(b"BM"):
|
||||
return "bmp"
|
||||
if head.startswith((b"II*\x00", b"MM\x00*")):
|
||||
return "tiff"
|
||||
if head[4:8] == b"ftyp":
|
||||
return f"isobmff:{head[8:12].decode('latin-1', 'replace')}"
|
||||
return f"unknown:{head[:16].hex()}"
|
||||
|
||||
|
||||
# --- JPEG encoder structure (metadata layer) ---
|
||||
|
||||
|
||||
def _jpeg_forensics_bytes(data: bytes) -> dict[str, Any]:
|
||||
"""Structure-level JPEG forensics: DQT tables (encoder fingerprint),
|
||||
SOF type (baseline/progressive) + chroma subsampling, DHT Huffman
|
||||
tables (custom = optimizing encoder), per-scan spectral selection
|
||||
(progressive scan script), JFIF/Adobe app markers, COM, DRI."""
|
||||
out: dict[str, Any] = {}
|
||||
try:
|
||||
if not data.startswith(b"\xff\xd8"):
|
||||
return out
|
||||
pos = 2
|
||||
scans: list[dict[str, int]] = []
|
||||
dqt: dict[str, list[int]] = {}
|
||||
dht: list[str] = []
|
||||
comments: list[str] = []
|
||||
while pos + 4 <= len(data):
|
||||
if data[pos] != 0xFF:
|
||||
break
|
||||
marker = data[pos + 1]
|
||||
if marker in (0xD8, 0x01) or 0xD0 <= marker <= 0xD7:
|
||||
pos += 2
|
||||
continue
|
||||
if marker == 0xD9:
|
||||
break
|
||||
length = struct.unpack(">H", data[pos + 2 : pos + 4])[0]
|
||||
body = data[pos + 4 : pos + 2 + length]
|
||||
if marker == 0xDB: # DQT
|
||||
off = 0
|
||||
while off < len(body):
|
||||
tid = body[off] & 0x0F
|
||||
prec = body[off] >> 4
|
||||
n = 128 if prec else 64
|
||||
vals = list(body[off + 1 : off + 1 + n])
|
||||
if prec: # 16-bit entries
|
||||
vals = [struct.unpack(">H", bytes(vals[i : i + 2]))[0] for i in range(0, len(vals) - 1, 2)]
|
||||
dqt[str(tid)] = vals[:64]
|
||||
off += 1 + n
|
||||
elif marker == 0xC4: # DHT: custom tables mean an optimizing encoder
|
||||
dht.append(body.hex())
|
||||
elif marker == 0xDD and len(body) >= 2: # DRI
|
||||
out["restart_interval"] = struct.unpack(">H", body[:2])[0]
|
||||
elif marker == 0xE0 and body.startswith(b"JFIF\x00") and len(body) >= 12:
|
||||
out["jfif"] = {
|
||||
"version": f"{body[5]}.{body[6]}",
|
||||
"density_units": body[7],
|
||||
"x_density": struct.unpack(">H", body[8:10])[0],
|
||||
"y_density": struct.unpack(">H", body[10:12])[0],
|
||||
}
|
||||
elif marker == 0xEE and body.startswith(b"Adobe") and len(body) >= 12:
|
||||
out["adobe_transform"] = body[11]
|
||||
elif marker in (0xC0, 0xC1, 0xC2) and len(body) >= 6:
|
||||
out["progressive"] = marker == 0xC2
|
||||
out["precision_bits"] = body[0]
|
||||
out["sof_height"] = struct.unpack(">H", body[1:3])[0]
|
||||
out["sof_width"] = struct.unpack(">H", body[3:5])[0]
|
||||
comps: list[dict[str, int]] = []
|
||||
for i in range(body[5]):
|
||||
c = body[6 + i * 3 : 9 + i * 3]
|
||||
if len(c) == 3:
|
||||
comps.append({"h": c[1] >> 4, "v": c[1] & 0x0F, "tq": c[2]})
|
||||
if len(comps) >= 3:
|
||||
lum = comps[0]
|
||||
subs = {1: "4:4:4", 2: "4:2:2"}.get(lum["h"] * lum["v"])
|
||||
out["subsampling"] = subs or f"{lum['h']}x{lum['v']}"
|
||||
elif marker == 0xFE: # COM
|
||||
comments.append(body.decode("utf-8", "replace")[:2000])
|
||||
elif marker == 0xDA:
|
||||
# SOS spectral selection: the progressive scan script
|
||||
# differs across libjpeg / mozjpeg / Photoshop
|
||||
if len(body) >= 3:
|
||||
ns = body[0]
|
||||
tail = body[1 + ns * 2 :]
|
||||
if len(tail) >= 3:
|
||||
scans.append({"ss": tail[0], "se": tail[1], "ah": tail[2] >> 4, "al": tail[2] & 0x0F})
|
||||
# skip entropy-coded data to the next marker
|
||||
end = data.find(b"\xff\xd9", pos)
|
||||
nxt = data.find(b"\xff", pos + 2)
|
||||
while nxt != -1 and nxt + 1 < len(data) and data[nxt + 1] == 0x00:
|
||||
nxt = data.find(b"\xff", nxt + 2)
|
||||
if nxt == -1 or (end != -1 and nxt >= end):
|
||||
break
|
||||
pos = nxt
|
||||
continue
|
||||
pos += 2 + length
|
||||
if dqt:
|
||||
out["quant_tables"] = dqt
|
||||
if dht:
|
||||
out["huffman_tables_hex"] = dht
|
||||
if comments:
|
||||
out["comments"] = comments
|
||||
if scans:
|
||||
out["scan_count"] = len(scans)
|
||||
out["scan_script"] = scans
|
||||
except Exception as exc:
|
||||
out["error"] = _safe_str(exc)
|
||||
return out
|
||||
|
||||
|
||||
def read_webp_chunks(data: bytes) -> list[dict[str, Any]]:
|
||||
"""WebP RIFF chunk inventory (VP8X/VP8/VP8L/EXIF/XMP/ICCP/ANIM...)."""
|
||||
chunks: list[dict[str, Any]] = []
|
||||
try:
|
||||
pos = 12
|
||||
declared_end = 8 + struct.unpack("<I", data[4:8])[0] if len(data) >= 12 else len(data)
|
||||
container_end = min(len(data), declared_end)
|
||||
while pos + 8 <= container_end:
|
||||
chunk_type = data[pos : pos + 4]
|
||||
ctype = chunk_type.decode("latin-1")
|
||||
length = struct.unpack("<I", data[pos + 4 : pos + 8])[0]
|
||||
body = data[pos + 8 : min(pos + 8 + length, container_end)]
|
||||
entry: dict[str, Any] = {"type": ctype, "length": length}
|
||||
if ctype == "XMP ":
|
||||
entry["kind"] = "xmp"
|
||||
entry["text"] = body.decode("utf-8", "replace")
|
||||
elif chunk_type in RIFF_CODED_IMAGE_CHUNKS:
|
||||
pass # pixel-data chunks: length is signal enough
|
||||
elif length:
|
||||
entry["base64"] = _b64(body)
|
||||
chunks.append(entry)
|
||||
pos += 8 + length + (length & 1) # chunks are 2-byte aligned
|
||||
except Exception as exc:
|
||||
chunks.append({"error": _safe_str(exc)})
|
||||
return chunks
|
||||
|
||||
|
||||
def read_webp_late_metadata_path(path: Path, window: int = _RAW_SCAN_HEAD) -> list[dict[str, Any]]:
|
||||
"""Stream metadata chunks after ``window`` while seeking over coded frames."""
|
||||
chunks: list[dict[str, Any]] = []
|
||||
try:
|
||||
file_size = path.stat().st_size
|
||||
with open(path, "rb") as handle:
|
||||
header = handle.read(12)
|
||||
if len(header) < 12 or not header.startswith(b"RIFF") or header[8:12] != b"WEBP":
|
||||
return chunks
|
||||
container_end = min(file_size, 8 + struct.unpack("<I", header[4:8])[0])
|
||||
position = 12
|
||||
while position + 8 <= container_end:
|
||||
handle.seek(position)
|
||||
chunk_header = handle.read(8)
|
||||
if len(chunk_header) < 8:
|
||||
break
|
||||
chunk_type = chunk_header[:4]
|
||||
(length,) = struct.unpack("<I", chunk_header[4:8])
|
||||
start = position + 8
|
||||
safe_length = max(0, min(length, container_end - start))
|
||||
if chunk_type in RIFF_METADATA_CHUNKS and start >= window:
|
||||
handle.seek(start)
|
||||
body = handle.read(min(safe_length, _B64_CAP))
|
||||
entry: dict[str, Any] = {
|
||||
"type": chunk_type.decode("latin-1"),
|
||||
"length": length,
|
||||
"base64": _b64(body),
|
||||
}
|
||||
if len(body) < safe_length:
|
||||
entry["truncated"] = True
|
||||
chunks.append(entry)
|
||||
position = start + safe_length + (safe_length & 1)
|
||||
except (OSError, struct.error) as exc:
|
||||
chunks.append({"error": _safe_str(exc)})
|
||||
return chunks
|
||||
|
||||
|
||||
def sha256_of(data: bytes) -> str:
|
||||
return hashlib.sha256(data).hexdigest()
|
||||
|
||||
|
||||
def xattr_where_from(path: Path) -> list[str]:
|
||||
"""macOS download-source URLs (kMDItemWhereFroms), empty elsewhere."""
|
||||
try:
|
||||
getter = cast("Any", os.getxattr) # pyright: ignore[reportAttributeAccessIssue, reportUnknownMemberType]
|
||||
raw = cast("bytes", getter(path, "com.apple.metadata:kMDItemWhereFroms"))
|
||||
value = plistlib.loads(raw)
|
||||
values = cast("list[Any]", value) if isinstance(value, list) else [value]
|
||||
return [str(item) for item in values]
|
||||
except (AttributeError, OSError, ValueError):
|
||||
return []
|
||||
|
||||
|
||||
def xattr_quarantine(path: Path) -> str | None:
|
||||
"""macOS quarantine string: flags; timestamp; downloading agent (Safari,
|
||||
Telegram, Chrome...). Presence alone means 'came from the internet'."""
|
||||
try:
|
||||
getter = cast("Any", os.getxattr) # pyright: ignore[reportAttributeAccessIssue, reportUnknownMemberType]
|
||||
raw = cast("bytes", getter(path, "com.apple.quarantine"))
|
||||
return raw.decode("utf-8", "replace")[:500]
|
||||
except (AttributeError, OSError):
|
||||
return None
|
||||
|
||||
|
||||
def read_isobmff_inventory(data: bytes) -> dict[str, Any]:
|
||||
"""HEIC/AVIF/MOV box inventory: top-level boxes plus the meta item
|
||||
types (Exif, mime=XMP, auxl depth/gain-map, aae Apple-edits plist,
|
||||
irot derived images). Strong phone-provenance signal."""
|
||||
out: dict[str, Any] = {}
|
||||
try:
|
||||
stream = io.BytesIO(data)
|
||||
|
||||
def boxes(start: int, end: int) -> list[tuple[str, int, int]]:
|
||||
return [
|
||||
(box_type.decode("latin-1"), payload_offset, box_end)
|
||||
for _, box_end, box_type, payload_offset in iter_file_boxes(stream, start, end)
|
||||
]
|
||||
|
||||
top = boxes(0, len(data))
|
||||
out["boxes"] = [t for t, _, _ in top]
|
||||
provenance_boxes: list[dict[str, Any]] = []
|
||||
for t, s, e in top:
|
||||
if t.encode("latin-1") in C2PA_BOX_TYPES:
|
||||
provenance_boxes.append(
|
||||
{"type": t, "length": e - s, "base64": _b64(data[s:e], cap=_PROVENANCE_B64_CAP)}
|
||||
)
|
||||
if t == "moov":
|
||||
for ct, cs, ce in boxes(s, e):
|
||||
if ct == "mvhd" and ce - cs >= 24:
|
||||
# full box + creation/modification times (1904 epoch)
|
||||
version = data[cs]
|
||||
base = cs + 4
|
||||
creation = struct.unpack(">I", data[base : base + 4])[0] if version == 0 else None
|
||||
if creation:
|
||||
out["mvhd_creation_time"] = creation - 2082844800
|
||||
elif t == "meta":
|
||||
# full box: 4 bytes version/flags, then child boxes
|
||||
for ct, cs, ce in boxes(s + 4, e):
|
||||
if ct == "iinf":
|
||||
# full box + entry count, then infe entries
|
||||
count = struct.unpack(">H", data[cs + 4 : cs + 6])[0]
|
||||
out["meta_item_count"] = count
|
||||
item_types: list[str] = []
|
||||
for it, is_, ie in boxes(cs + 6, ce):
|
||||
if it == "infe" and ie - is_ >= 8:
|
||||
# infe full box: version(1)+flags(3), then
|
||||
# v2: item_ID(2)+protection(2)+item_type(4)
|
||||
# v3: item_ID(4)+protection(2)+item_type(4)
|
||||
version = data[is_]
|
||||
off = is_ + 4 + (4 if version == 3 else 2) + 2
|
||||
if off + 4 <= ie:
|
||||
item_types.append(data[off : off + 4].decode("latin-1", "replace"))
|
||||
if item_types:
|
||||
out["meta_item_types"] = sorted(set(item_types))
|
||||
elif ct == "iprp":
|
||||
out["has_iprp"] = True
|
||||
for pt, ps, pe in boxes(cs, ce):
|
||||
if pt == "ipco":
|
||||
props = [t for t, _, _ in boxes(ps, pe)]
|
||||
out["ipco_properties"] = props
|
||||
# auxC holds the auxiliary image type URN
|
||||
for box_type, qs, qe in boxes(ps, pe):
|
||||
if box_type == "auxC":
|
||||
out["auxc_types"] = (
|
||||
data[qs + 4 : qe].split(b"\x00")[0].decode("latin-1", "replace")
|
||||
)
|
||||
elif ct == "iref":
|
||||
out["has_iref"] = True
|
||||
if provenance_boxes:
|
||||
out["provenance_boxes"] = provenance_boxes
|
||||
# QuickTime metadata keys (©mak/©mod/©swr) for the MOV side of
|
||||
# Live Photos: tolerant printable-string grab after each atom
|
||||
qt: dict[str, str] = {}
|
||||
for atom, key in ((b"\xa9mak", "make"), (b"\xa9mod", "model"), (b"\xa9swr", "software")):
|
||||
idx = data.find(atom)
|
||||
if idx != -1:
|
||||
m = re.search(rb"[ -~]{4,80}", data[idx + 4 : idx + 200])
|
||||
if m:
|
||||
qt[key] = m.group(0).decode("ascii", "replace")
|
||||
if qt:
|
||||
out["quicktime"] = qt
|
||||
except Exception as exc:
|
||||
out["error"] = _safe_str(exc)
|
||||
return out
|
||||
|
||||
|
||||
def read_isobmff_provenance_path(path: Path) -> dict[str, Any]:
|
||||
"""Stream top-level ISOBMFF boxes and preserve provenance payloads.
|
||||
|
||||
This is the large-file counterpart to :func:`read_isobmff_inventory`.
|
||||
It seeks over media payloads instead of loading them into memory.
|
||||
"""
|
||||
out: dict[str, Any] = {"boxes": []}
|
||||
provenance_boxes: list[dict[str, Any]] = []
|
||||
collected = 0
|
||||
try:
|
||||
file_size = path.stat().st_size
|
||||
with open(path, "rb") as f:
|
||||
for _, box_end, box_type_raw, payload_offset in iter_file_boxes(f, 0, file_size):
|
||||
box_type = box_type_raw.decode("latin-1")
|
||||
out["boxes"].append(box_type)
|
||||
payload_length = box_end - payload_offset
|
||||
if box_type_raw in C2PA_BOX_TYPES and collected < _PROVENANCE_B64_CAP:
|
||||
to_read = min(payload_length, _PROVENANCE_B64_CAP - collected)
|
||||
f.seek(payload_offset)
|
||||
payload = f.read(to_read)
|
||||
entry: dict[str, Any] = {
|
||||
"type": box_type,
|
||||
"length": payload_length,
|
||||
"base64": _b64(payload, cap=_PROVENANCE_B64_CAP),
|
||||
}
|
||||
if to_read < payload_length:
|
||||
entry["truncated"] = True
|
||||
provenance_boxes.append(entry)
|
||||
collected += len(payload)
|
||||
except (OSError, struct.error) as exc:
|
||||
out["error"] = _safe_str(exc)
|
||||
if provenance_boxes:
|
||||
out["provenance_boxes"] = provenance_boxes
|
||||
return out
|
||||
|
||||
|
||||
def read_png_late_metadata_path(path: Path, window: int = _RAW_SCAN_HEAD) -> list[dict[str, Any]]:
|
||||
"""Stream PNG metadata chunks whose payload starts after ``window``."""
|
||||
chunks: list[dict[str, Any]] = []
|
||||
try:
|
||||
file_size = path.stat().st_size
|
||||
with open(path, "rb") as f:
|
||||
if f.read(8) != b"\x89PNG\r\n\x1a\n":
|
||||
return chunks
|
||||
pos = 8
|
||||
while pos + 12 <= file_size:
|
||||
f.seek(pos)
|
||||
header = f.read(8)
|
||||
if len(header) < 8:
|
||||
break
|
||||
length, chunk_type = struct.unpack(">I4s", header)
|
||||
data_start = pos + 8
|
||||
safe_length = max(0, min(length, file_size - data_start))
|
||||
if chunk_type in PNG_METADATA_CHUNKS and data_start >= window:
|
||||
body = f.read(min(safe_length, _B64_CAP))
|
||||
entry: dict[str, Any] = {
|
||||
"type": chunk_type.decode("latin-1"),
|
||||
"length": length,
|
||||
"base64": _b64(body),
|
||||
}
|
||||
if len(body) < safe_length:
|
||||
entry["truncated"] = True
|
||||
chunks.append(entry)
|
||||
pos = data_start + safe_length + 4
|
||||
if chunk_type == b"IEND":
|
||||
break
|
||||
except (OSError, struct.error) as exc:
|
||||
chunks.append({"error": _safe_str(exc)})
|
||||
return chunks
|
||||
|
||||
|
||||
def apple_live_photo_id(head: bytes) -> str | None:
|
||||
"""Apple Live Photo content identifier (links the still to its MOV).
|
||||
|
||||
The UUID sits in the Apple MakerNote (tag 17) of the still and in the
|
||||
MOV metadata; a raw head scan finds it in either container."""
|
||||
# the UUID string sits next to "content.identifier" in the MOV, but in
|
||||
# the STILL it is a bare UUID inside the Apple MakerNote (whose header
|
||||
# is "Apple iOS"), so gate on either marker
|
||||
if b"content.identifier" not in head and b"com.apple.quicktime" not in head and b"Apple iOS" not in head:
|
||||
return None
|
||||
m = re.search(rb"[0-9A-Fa-f]{8}-[0-9A-Fa-f]{4}-[0-9A-Fa-f]{4}-[0-9A-Fa-f]{4}-[0-9A-Fa-f]{12}", head)
|
||||
return m.group(0).decode("ascii") if m else None
|
||||
|
||||
|
||||
_MAX_FULL_READ = 256 << 20 # files bigger than this are scanned head-only
|
||||
_HEAD_READ = 4 << 20
|
||||
|
||||
|
||||
def _sha256_stream(path: Path) -> str:
|
||||
h = hashlib.sha256()
|
||||
with open(path, "rb") as f:
|
||||
for block in iter(lambda: f.read(1 << 20), b""):
|
||||
h.update(block)
|
||||
return h.hexdigest()
|
||||
|
||||
|
||||
def collect_forensic_metadata(
|
||||
path: Path,
|
||||
*,
|
||||
schema_version: int = FORENSIC_METADATA_SCHEMA_VERSION,
|
||||
) -> dict[str, Any]:
|
||||
"""Collect the versioned, metadata-only forensic record for ``path``.
|
||||
|
||||
This broad inspection record is not provenance-detector input. Use
|
||||
:func:`remove_ai_watermarks.metadata_record.collect_metadata_record` for the
|
||||
strict record accepted by ``identify_metadata_record``. Long-lived consumers
|
||||
should request the schema they implement; unsupported versions raise before the
|
||||
source is read.
|
||||
"""
|
||||
schema_version = require_schema_version(
|
||||
schema_version,
|
||||
contract="forensic metadata",
|
||||
supported=(1,),
|
||||
)
|
||||
image_io._register_heif() # pyright: ignore[reportPrivateUsage]
|
||||
stat = path.stat()
|
||||
oversized = stat.st_size > _MAX_FULL_READ
|
||||
if oversized:
|
||||
data = None
|
||||
with open(path, "rb") as f:
|
||||
head = f.read(_HEAD_READ)
|
||||
else:
|
||||
data = path.read_bytes()
|
||||
head = data
|
||||
record: dict[str, Any] = {
|
||||
"schema_version": schema_version,
|
||||
"record_type": FORENSIC_METADATA_RECORD_TYPE,
|
||||
"file": str(path),
|
||||
"name": path.name,
|
||||
"extension": path.suffix.lower(),
|
||||
"size_bytes": stat.st_size,
|
||||
"mtime": stat.st_mtime,
|
||||
"birthtime": getattr(stat, "st_birthtime", None),
|
||||
"sha256": _sha256_stream(path) if data is None else sha256_of(data),
|
||||
"content_format": sniff_format(head),
|
||||
}
|
||||
if oversized:
|
||||
# Preserve the same bounded byte windows used by downstream provenance
|
||||
# algorithms while path-based readers (PIL, piexif, C2PA) run normally.
|
||||
record["oversized"] = {"head_scanned_bytes": len(head)}
|
||||
record["raw_metadata_windows"] = {"head_base64": _b64(head[:_RAW_SCAN_HEAD])}
|
||||
if stat.st_size > _RAW_SCAN_TAIL:
|
||||
with open(path, "rb") as f:
|
||||
f.seek(-_RAW_SCAN_TAIL, 2)
|
||||
record["raw_metadata_windows"]["tail_base64"] = _b64(f.read())
|
||||
where_from = xattr_where_from(path)
|
||||
if where_from:
|
||||
record["download_source_urls"] = where_from
|
||||
quarantine = xattr_quarantine(path)
|
||||
if quarantine:
|
||||
record["quarantine"] = quarantine
|
||||
live_photo_id = apple_live_photo_id(head[: 2 << 20])
|
||||
if live_photo_id:
|
||||
record["live_photo_content_id"] = live_photo_id
|
||||
record["pil"], record["iptc"], exif_blob = read_pil_info(path)
|
||||
record["exif"], thumbnail = read_full_exif(path, exif_blob, data)
|
||||
record["c2pa_store"] = read_c2pa_store(path)
|
||||
if data is not None:
|
||||
fmt = record["content_format"]
|
||||
if fmt == "png":
|
||||
record["png_chunks"], post_iend = read_png_chunks(data)
|
||||
if post_iend:
|
||||
record["png_post_iend_bytes"] = len(post_iend)
|
||||
record["png_post_iend_base64"] = _b64(post_iend)
|
||||
elif fmt == "jpeg":
|
||||
record["jpeg"] = read_jpeg_segments(data)
|
||||
record["jpeg_forensics"] = _jpeg_forensics_bytes(data)
|
||||
elif fmt == "webp":
|
||||
record["webp_chunks"] = read_webp_chunks(data)
|
||||
elif fmt.startswith("isobmff"):
|
||||
record["isobmff"] = read_isobmff_inventory(data)
|
||||
elif record["content_format"] == "png":
|
||||
late_chunks = read_png_late_metadata_path(path)
|
||||
if late_chunks:
|
||||
record["png_late_metadata_chunks"] = late_chunks
|
||||
elif record["content_format"] == "webp":
|
||||
late_chunks = read_webp_late_metadata_path(path)
|
||||
if late_chunks:
|
||||
record["webp_late_metadata_chunks"] = late_chunks
|
||||
elif record["content_format"].startswith("isobmff"):
|
||||
record["isobmff"] = read_isobmff_provenance_path(path)
|
||||
if thumbnail:
|
||||
record["has_exif_thumbnail"] = True
|
||||
# the embedded thumbnail is its own JPEG; after an edit its encoder
|
||||
# forensics commonly MISMATCH the main image (classic tamper tell)
|
||||
thumb_forensics = _jpeg_forensics_bytes(thumbnail)
|
||||
thumb_forensics["base64"] = _b64(thumbnail)
|
||||
record["exif_thumbnail_forensics"] = thumb_forensics
|
||||
return record
|
||||
@@ -22,6 +22,7 @@ from __future__ import annotations
|
||||
import base64
|
||||
import itertools
|
||||
import logging
|
||||
import struct
|
||||
from dataclasses import dataclass, field
|
||||
from typing import TYPE_CHECKING, Any, cast
|
||||
|
||||
@@ -30,6 +31,8 @@ from remove_ai_watermarks._internal.c2pa import (
|
||||
cbor_text_after,
|
||||
extract_c2pa_info,
|
||||
soft_binding_vendors_in,
|
||||
synthid_vendors_in,
|
||||
synthid_verdict,
|
||||
)
|
||||
from remove_ai_watermarks._internal.constants import (
|
||||
C2PA_AI_TOOLS,
|
||||
@@ -37,6 +40,7 @@ from remove_ai_watermarks._internal.constants import (
|
||||
C2PA_IDENTITY_AI_ORGS,
|
||||
C2PA_ISSUERS,
|
||||
)
|
||||
from remove_ai_watermarks._internal.schema import require_schema_version
|
||||
from remove_ai_watermarks.metadata import (
|
||||
AI_METADATA_KEYS,
|
||||
AIGC_MARKERS,
|
||||
@@ -70,6 +74,10 @@ if TYPE_CHECKING:
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Stable JSON contract for callers that pass a verdict between services. Bump this
|
||||
# only for a breaking shape or semantic change; adding optional fields is compatible.
|
||||
PROVENANCE_REPORT_SCHEMA_VERSION = 1
|
||||
|
||||
# How much of a non-PNG container to binary-scan for the C2PA issuer.
|
||||
_SCAN_BYTES = 1024 * 1024
|
||||
|
||||
@@ -169,7 +177,32 @@ def _external_metadata(value: Any) -> tuple[list[tuple[str, Any]], bytes]:
|
||||
"""Index nested metadata and recover common encoded binary values in one pass."""
|
||||
pairs: list[tuple[str, Any]] = []
|
||||
parts: list[bytes] = []
|
||||
diagnostic_keys = {"error", "kind"}
|
||||
diagnostic_keys = {
|
||||
"artifacts",
|
||||
"birthtime",
|
||||
"color",
|
||||
"content_format",
|
||||
"dct",
|
||||
"ela",
|
||||
"error",
|
||||
"extension",
|
||||
"fft",
|
||||
"file",
|
||||
"filename",
|
||||
"full",
|
||||
"gradient",
|
||||
"kind",
|
||||
"mtime",
|
||||
"name",
|
||||
"noise",
|
||||
"path",
|
||||
"pixel",
|
||||
"provenance",
|
||||
"sha256",
|
||||
"signals",
|
||||
"size_bytes",
|
||||
"timing_ms",
|
||||
}
|
||||
|
||||
def visit(item: Any) -> None:
|
||||
if isinstance(item, dict):
|
||||
@@ -177,7 +210,6 @@ def _external_metadata(value: Any) -> tuple[list[tuple[str, Any]], bytes]:
|
||||
for key, nested in mapping.items():
|
||||
key_text = str(key)
|
||||
pairs.append((key_text, nested))
|
||||
parts.append(key_text.encode("utf-8", "replace"))
|
||||
if key_text.lower() in diagnostic_keys:
|
||||
continue
|
||||
if isinstance(nested, str) and (key_text == "base64" or key_text.endswith("_base64")):
|
||||
@@ -241,17 +273,69 @@ def _external_exif_generator(pairs: list[tuple[str, Any]], scan: bytes) -> str |
|
||||
return generator_from_metadata(candidates, scan)
|
||||
|
||||
|
||||
def _metadata_source_kind(info: dict[str, Any], scan: bytes) -> str | None:
|
||||
"""Normalize the source type wherever it is carried: C2PA or IPTC/XMP.
|
||||
|
||||
A composite marker contains ``TrainedAlgorithmicMedia`` as a substring, so it
|
||||
is removed before looking for a standalone full-generation marker. When a file
|
||||
genuinely carries both kinds, full generation wins.
|
||||
"""
|
||||
structured = info.get("ai_source_kind")
|
||||
without_composites = scan.replace(b"compositeWithTrainedAlgorithmicMedia", b"").replace(b"compositeSynthetic", b"")
|
||||
generated = structured == "generated" or any(
|
||||
marker in without_composites for marker in (b"trainedAlgorithmicMedia", b"TrainedAlgorithmicMedia")
|
||||
)
|
||||
if generated:
|
||||
return "generated"
|
||||
if structured == "enhanced" or any(
|
||||
marker in scan for marker in (b"compositeWithTrainedAlgorithmicMedia", b"compositeSynthetic")
|
||||
):
|
||||
return "enhanced"
|
||||
return None
|
||||
|
||||
|
||||
def evidence_from_metadata_record(
|
||||
record: dict[str, Any], *, path: Path, c2pa_manifest_store: str | dict[str, Any] | None = None
|
||||
) -> ProvenanceEvidence:
|
||||
"""Normalize an externally collected metadata record into provenance evidence.
|
||||
|
||||
The record may contain arbitrary nested dictionaries and lists. Text, bytes,
|
||||
hexadecimal values prefixed with ``hex:``, and fields named ``base64`` or
|
||||
ending in ``_base64`` are included in the shared byte scan. No source file is
|
||||
opened.
|
||||
Unversioned external records may contain arbitrary nested dictionaries and
|
||||
lists. Versioned native records accept only the source-derived fields emitted by
|
||||
``collect_metadata_record``; other native record types and unknown schema
|
||||
versions are rejected. No source file is opened.
|
||||
"""
|
||||
pairs, scan = _external_metadata(record)
|
||||
from remove_ai_watermarks.metadata_record import METADATA_RECORD_SCHEMA_VERSION, METADATA_RECORD_TYPE
|
||||
|
||||
# Records produced by ``collect_metadata_record`` are a versioned transport
|
||||
# contract. Only their source-derived fields are evidence: the filename,
|
||||
# container label and schema bookkeeping describe the collector and must never
|
||||
# become detector input. Shape-detect the pre-versioned form as well so records
|
||||
# emitted by 0.26 remain safe and readable.
|
||||
record_type = record.get("record_type")
|
||||
if record_type not in (None, METADATA_RECORD_TYPE):
|
||||
raise ValueError(f"Unsupported metadata record type: {record_type!r}")
|
||||
if record_type == METADATA_RECORD_TYPE:
|
||||
require_schema_version(
|
||||
record.get("schema_version"),
|
||||
contract="provenance metadata",
|
||||
supported=(METADATA_RECORD_SCHEMA_VERSION,),
|
||||
)
|
||||
status = record.get("status")
|
||||
if status == "error":
|
||||
raise ValueError("Provenance metadata collection failed")
|
||||
if status != "complete":
|
||||
raise ValueError(f"Unsupported provenance metadata collection status: {status!r}")
|
||||
is_portable_record = record_type == METADATA_RECORD_TYPE or {
|
||||
"container",
|
||||
"metadata_base64",
|
||||
"tail_base64",
|
||||
}.issubset(record)
|
||||
evidence_record = (
|
||||
{key: record[key] for key in ("metadata_base64", "tail_base64", "pil", "exif") if key in record}
|
||||
if is_portable_record
|
||||
else record
|
||||
)
|
||||
pairs, scan = _external_metadata(evidence_record)
|
||||
store = c2pa_manifest_store
|
||||
if store is None:
|
||||
candidate = record.get("c2pa_store")
|
||||
@@ -358,13 +442,13 @@ class ProvenanceReport:
|
||||
is_ai_generated: bool | None # True / False is never asserted; None = unknown
|
||||
platform: str | None
|
||||
confidence: str # "high" | "medium" | "none"
|
||||
# Coarse AI-origin kind from the C2PA digital-source-type, so a caller can
|
||||
# branch on full generation vs an AI-touched real photo:
|
||||
# Coarse AI-origin kind from a C2PA or standalone IPTC/XMP digital-source-type,
|
||||
# so a caller can branch on full generation vs an AI-touched real photo:
|
||||
# "generated" -- digitalSourceType trainedAlgorithmicMedia (fully AI).
|
||||
# "enhanced" -- compositeWithTrainedAlgorithmicMedia (real content with an
|
||||
# AI-composited region; scrub the AI region, keep the photo).
|
||||
# None -- no C2PA AI source-type (verdict, if AI, came from another
|
||||
# signal: IPTC, AIGC, local gen params, xAI, ...).
|
||||
# None -- no AI digital-source-type (verdict, if AI, came from another
|
||||
# signal: AIGC, local gen params, xAI, ...).
|
||||
ai_source_kind: str | None = None
|
||||
# True when the AI verdict rests on a metadata or embedded-invisible signal
|
||||
# (C2PA AI issuer / SynthID proxy, IPTC, AIGC, local gen params, EXIF/xAI, or
|
||||
@@ -383,6 +467,42 @@ class ProvenanceReport:
|
||||
# inconsistent -- a strong tell of spoofed, transplanted, or laundered metadata.
|
||||
integrity_clashes: list[str] = field(default_factory=list[str])
|
||||
|
||||
def to_dict(
|
||||
self,
|
||||
*,
|
||||
schema_version: int = PROVENANCE_REPORT_SCHEMA_VERSION,
|
||||
) -> dict[str, Any]:
|
||||
"""Return the versioned, JSON-safe verdict contract.
|
||||
|
||||
``path`` is deliberately omitted. It is extraction context, not part of the
|
||||
verdict, and local filesystem paths should not cross a service boundary.
|
||||
Request an explicit schema for a long-lived transport consumer.
|
||||
"""
|
||||
schema_version = require_schema_version(
|
||||
schema_version,
|
||||
contract="provenance report",
|
||||
supported=(1,),
|
||||
)
|
||||
return {
|
||||
"schema_version": schema_version,
|
||||
"is_ai_generated": self.is_ai_generated,
|
||||
"platform": self.platform,
|
||||
"confidence": self.confidence,
|
||||
"ai_source_kind": self.ai_source_kind,
|
||||
"ai_from_metadata": self.ai_from_metadata,
|
||||
"watermarks": list(self.watermarks),
|
||||
"signals": [
|
||||
{
|
||||
"name": signal.name,
|
||||
"detail": signal.detail,
|
||||
"confidence": signal.confidence,
|
||||
}
|
||||
for signal in self.signals
|
||||
],
|
||||
"caveats": list(self.caveats),
|
||||
"integrity_clashes": list(self.integrity_clashes),
|
||||
}
|
||||
|
||||
|
||||
def extract_provenance_evidence(image_path: Path) -> ProvenanceEvidence:
|
||||
"""Read all file-backed metadata needed by provenance verdict logic once."""
|
||||
@@ -447,6 +567,72 @@ _DEVICE_C2PA_PLATFORM: tuple[tuple[bytes, str], ...] = (
|
||||
)
|
||||
|
||||
|
||||
def _metadata_region(head: bytes) -> bytes:
|
||||
"""The part of the scan buffer that can hold metadata, with the coded pixels cut out.
|
||||
|
||||
The vendor registries are matched as raw substrings, and the shortest tokens are
|
||||
four and five bytes (``Bria``, ``Adobe``, ``Canva``). Over a megabyte of compressed
|
||||
pixel data a four-byte sequence appears by chance about once in three thousand
|
||||
images -- measured: ``Bria`` matched inside the entropy-coded scan of 4 of 14,707
|
||||
corpus JPEGs, in none of which the manifest names Bria. That is not a cosmetic
|
||||
mislabel, because the Bria entry carries ``asserts_ai``: a chance match can declare
|
||||
an image AI-generated.
|
||||
|
||||
``c2pa_marker_in`` already refuses a bare ``c2pa`` substring for the same reason.
|
||||
This is the same defence for the registries: they see the container's metadata and
|
||||
not its pixels.
|
||||
|
||||
JPEG keeps the marker segments before the entropy-coded scan, PNG every chunk but
|
||||
``IDAT``, and both keep the trailer past the end marker. Anything ``scan_head``
|
||||
APPENDED past the window is metadata by construction (late chunks, boxes, decoder
|
||||
text), so it is always kept and never walked -- walking it is what produced 11 MB
|
||||
records and a phantom AIGC signal in the record collector.
|
||||
|
||||
Trimming happens only when the container actually parses: a JPEG whose marker walk
|
||||
reaches the coded scan, a PNG whose chunk walk reaches ``IDAT``. Anything else --
|
||||
a malformed container, a synthetic blob, a format with no walker here -- is
|
||||
returned whole. Cutting a buffer this function did not understand would drop real
|
||||
evidence to avoid a chance match, which is the wrong way round.
|
||||
"""
|
||||
raw, appended = head[:_SCAN_BYTES], head[_SCAN_BYTES:]
|
||||
if raw[:2] == b"\xff\xd8":
|
||||
index, size = 2, len(raw)
|
||||
while index + 1 < size:
|
||||
if raw[index] != 0xFF:
|
||||
return head # not a marker boundary: the walk is lost, keep everything
|
||||
marker = raw[index + 1]
|
||||
if marker in (0xDA, 0xD9): # SOS / EOI: the coded scan follows
|
||||
end = raw.rfind(b"\xff\xd9")
|
||||
return raw[:index] + (raw[end + 2 :] if end >= index else b"") + appended
|
||||
if 0xD0 <= marker <= 0xD7 or marker == 0x01:
|
||||
index += 2
|
||||
continue
|
||||
if index + 4 > size:
|
||||
break
|
||||
length = int.from_bytes(raw[index + 2 : index + 4], "big")
|
||||
if length < 2 or index + 2 + length > size:
|
||||
break
|
||||
index += 2 + length
|
||||
return head # ran out of buffer before the scan: nothing was skipped anyway
|
||||
if raw[:8] == b"\x89PNG\r\n\x1a\n":
|
||||
out = bytearray()
|
||||
position, size, saw_idat = 8, len(raw), False
|
||||
while position + 8 <= size:
|
||||
(length,) = struct.unpack(">I", raw[position : position + 4])
|
||||
chunk_type = raw[position + 4 : position + 8]
|
||||
start = position + 8
|
||||
if chunk_type == b"IDAT":
|
||||
saw_idat = True
|
||||
else:
|
||||
out += chunk_type + raw[start : start + min(length, size - start)]
|
||||
position = start + length + 4
|
||||
if chunk_type == b"IEND":
|
||||
out += raw[position:]
|
||||
break
|
||||
return bytes(out) + appended if saw_idat else head
|
||||
return head
|
||||
|
||||
|
||||
def _first_token_match(head: bytes, table: tuple[tuple[bytes, str], ...]) -> str | None:
|
||||
"""First platform in ``table`` whose token appears in ``head``, else None.
|
||||
|
||||
@@ -883,23 +1069,22 @@ def _identify_from_evidence(
|
||||
# score, the latter can be a by-product of our own SDXL removal pass, so
|
||||
# neither is a trustworthy "the generator stamped its identity" claim.
|
||||
ai_vendor_claims: dict[str, str] = {}
|
||||
camera_label = _device_platform(head)
|
||||
signer_label = _signer_platform(head)
|
||||
# The vendor registries match short raw substrings, so they read the container's
|
||||
# metadata rather than its pixels -- see `_metadata_region`. Every other check
|
||||
# below keeps the full buffer: their markers are long and distinctive.
|
||||
region = _metadata_region(head)
|
||||
camera_label = _device_platform(region)
|
||||
signer_label = _signer_platform(region)
|
||||
|
||||
# ── C2PA Content Credentials ────────────────────────────────────
|
||||
has_c2pa = bool(info) or c2pa_marker_in(head)
|
||||
issuers = [info["issuer"]] if info.get("issuer") else _issuers_in(head)
|
||||
issuers = [info["issuer"]] if info.get("issuer") else _issuers_in(region)
|
||||
# Full AI generation (trainedAlgorithmicMedia) vs an AI-enhanced real photo
|
||||
# (compositeWithTrainedAlgorithmicMedia). The structured kind is parsed once in
|
||||
# _internal.c2pa._populate_registry_fields (covers PNG + any container the c2pa-python
|
||||
# reader handles); fall back to a raw head scan for the non-PNG raw-blob path
|
||||
# where extract_c2pa_info returns {}. Full generation wins when both appear.
|
||||
c2pa_source_kind = info.get("ai_source_kind")
|
||||
if c2pa_source_kind is None:
|
||||
if b"trainedAlgorithmicMedia" in head:
|
||||
c2pa_source_kind = "generated"
|
||||
elif b"compositeWithTrainedAlgorithmicMedia" in head:
|
||||
c2pa_source_kind = "enhanced"
|
||||
source_kind = _metadata_source_kind(info, head)
|
||||
# An identity-AI issuer (a pure-generator brand like Dreamina) asserts AI even
|
||||
# without a digitalSourceType -- some ByteDance/Dreamina manifests ship no
|
||||
# trainedAlgorithmicMedia, so the registered generator name is the only signal.
|
||||
@@ -907,7 +1092,7 @@ def _identify_from_evidence(
|
||||
# does not reopen the incidental-mention problem the common-word issuers have.
|
||||
issuer_blob = " ".join(issuers)
|
||||
c2pa_identity_ai = has_c2pa and any(org in issuer_blob for org in C2PA_IDENTITY_AI_ORGS)
|
||||
c2pa_is_ai = c2pa_source_kind is not None or c2pa_identity_ai
|
||||
c2pa_is_ai = source_kind is not None or c2pa_identity_ai
|
||||
# Generator string (for the signal detail): structured for PNG, CBOR-scanned
|
||||
# for other containers. Best-effort -- some manifests key it as
|
||||
# `claim_generator_info` (Pixel), so this can be None even when a device is
|
||||
@@ -915,7 +1100,7 @@ def _identify_from_evidence(
|
||||
generator = (
|
||||
info.get("claim_generator")
|
||||
or cbor_text_after(head, b"claim_generator")
|
||||
or (", ".join(tools) if (tools := _ai_tools_in(head)) else None)
|
||||
or (", ".join(tools) if (tools := _ai_tools_in(region)) else None)
|
||||
)
|
||||
# Platform: a distinctive device/camera token in the manifest wins (it is the
|
||||
# signer/producer), then an editing-app/AI-device signer (Samsung Galaxy,
|
||||
@@ -950,9 +1135,24 @@ def _identify_from_evidence(
|
||||
platform = f"C2PA signer: {cloud_vendor} (cloud manifest)"
|
||||
|
||||
# ── SynthID metadata proxy ──────────────────────────────────────
|
||||
# get_ai_metadata already sets synthid_watermark for both PNG (caBX parser)
|
||||
# and non-PNG (its own synthid_source fallback), so no extra scan is needed.
|
||||
# Structured first (the PNG caBX parser and the manifest store both fill
|
||||
# `synthid_watermark`), then the byte scan for the containers that keep the
|
||||
# manifest where no parser reaches it.
|
||||
#
|
||||
# The scan lives HERE, in the verdict, and not in extraction, for the same reason
|
||||
# `soft_binding` below does: extraction has two implementations -- one reading a
|
||||
# file, one reading a portable record -- and a rule that lives in only one of them
|
||||
# is a rule the other silently lacks. It did: 74 corpus images reported SynthID
|
||||
# through `identify` and not through the record, because `get_ai_metadata`'s own
|
||||
# fallback has no counterpart on the record side. `get_ai_metadata` keeps its copy
|
||||
# for its own callers; the verdict no longer depends on which extractor ran.
|
||||
synthid = meta.get("synthid_watermark")
|
||||
# The literal byte checks mirror `metadata.synthid_source` exactly rather than
|
||||
# reusing the derived `has_c2pa` / `source_kind` above, which are broader:
|
||||
# the file path's answer must not move.
|
||||
trained_source = b"trainedAlgorithmicMedia" in head or b"TrainedAlgorithmicMedia" in head
|
||||
if not synthid and trained_source and c2pa_marker_in(head) and (vendors := synthid_vendors_in(region)):
|
||||
synthid = synthid_verdict(", ".join(vendors))
|
||||
if synthid:
|
||||
watermarks.append(f"SynthID watermark, inferred from C2PA metadata ({synthid})")
|
||||
caveats.append(_SYNTHID_CAVEAT)
|
||||
@@ -964,7 +1164,7 @@ def _identify_from_evidence(
|
||||
# ── C2PA soft-binding: a named forensic/third-party watermark vendor ─
|
||||
# (Adobe TrustMark, Digimarc, Imatag, ...). Present in the manifest even when
|
||||
# the watermark itself can't be decoded; names whose watermark stamped the pixels.
|
||||
soft_binding = meta.get("soft_binding") or (", ".join(v) if (v := soft_binding_vendors_in(head)) else None)
|
||||
soft_binding = meta.get("soft_binding") or (", ".join(v) if (v := soft_binding_vendors_in(region)) else None)
|
||||
if soft_binding:
|
||||
signals.append(Signal("soft_binding", f"C2PA soft binding: {soft_binding}", "high"))
|
||||
watermarks.append(f"Forensic watermark soft binding ({soft_binding})")
|
||||
@@ -1136,9 +1336,9 @@ def _identify_from_evidence(
|
||||
is_ai_generated=is_ai,
|
||||
platform=platform,
|
||||
confidence=confidence,
|
||||
# Only meaningful when the AI verdict actually came from the C2PA source
|
||||
# type; a non-C2PA AI signal (IPTC/AIGC/local gen) leaves it None.
|
||||
ai_source_kind=c2pa_source_kind if (is_ai and has_c2pa) else None,
|
||||
# Meaningful for the same digitalSourceType whether carried by C2PA or a
|
||||
# standalone IPTC/XMP label. Other AI signals leave it None.
|
||||
ai_source_kind=source_kind if (is_ai and (has_c2pa or iptc)) else None,
|
||||
ai_from_metadata=ai_from_metadata,
|
||||
watermarks=watermarks,
|
||||
signals=signals,
|
||||
@@ -1170,6 +1370,16 @@ def identify_from_evidence(
|
||||
)
|
||||
|
||||
|
||||
def identify_metadata_record(record: dict[str, Any], *, path: Path) -> ProvenanceReport:
|
||||
"""Build a metadata-only verdict from a portable metadata record.
|
||||
|
||||
This is the service-integration entry point: the source file is never opened,
|
||||
and callers receive the same verdict as the explicit
|
||||
``evidence_from_metadata_record`` / ``identify_from_evidence`` sequence.
|
||||
"""
|
||||
return identify_from_evidence(evidence_from_metadata_record(record, path=path))
|
||||
|
||||
|
||||
def identify(
|
||||
image_path: Path,
|
||||
*,
|
||||
|
||||
@@ -18,6 +18,11 @@ if TYPE_CHECKING:
|
||||
from collections.abc import Callable, Iterable
|
||||
from pathlib import Path
|
||||
|
||||
from remove_ai_watermarks._internal.constants import (
|
||||
PNG_METADATA_CHUNKS,
|
||||
RIFF_METADATA_CHUNKS,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Smaller scan_head window for the cheap marker checks (has_ai_metadata,
|
||||
@@ -236,11 +241,6 @@ def _is_ai_value(value: str) -> bool:
|
||||
return any(token in value_lower for token in AI_GENERATOR_TOKENS)
|
||||
|
||||
|
||||
# PNG ancillary chunks that can carry provenance metadata (XMP, EXIF, text).
|
||||
# Never IDAT -- that is the compressed pixel stream.
|
||||
_PNG_META_CHUNKS: frozenset[bytes] = frozenset({b"tEXt", b"iTXt", b"zTXt", b"eXIf", b"iCCP"})
|
||||
|
||||
|
||||
def _png_late_metadata(image_path: Path, window: int) -> bytes:
|
||||
"""Payloads of PNG metadata chunks that start *beyond* the first ``window``
|
||||
bytes, found by seeking past the (large) ``IDAT`` pixel stream.
|
||||
@@ -272,7 +272,7 @@ def _png_late_metadata(image_path: Path, window: int) -> bytes:
|
||||
# Clamp the attacker-controlled 32-bit length to the bytes that
|
||||
# actually remain, so a malformed huge length can't allocate GBs.
|
||||
safe_length = max(0, min(length, file_size - data_start))
|
||||
if chunk_type in _PNG_META_CHUNKS and data_start >= window:
|
||||
if chunk_type in PNG_METADATA_CHUNKS and data_start >= window:
|
||||
f.seek(data_start)
|
||||
out += f.read(safe_length)
|
||||
# Advance by the CLAMPED length: a malformed/inflated `length` that
|
||||
@@ -285,6 +285,55 @@ def _png_late_metadata(image_path: Path, window: int) -> bytes:
|
||||
return bytes(out)
|
||||
|
||||
|
||||
def _riff_late_metadata(image_path: Path, window: int, *, max_total: int = 4 * 1024 * 1024) -> bytes:
|
||||
"""Payloads of RIFF metadata chunks that start *beyond* the first ``window``
|
||||
bytes, found by stepping over the (large) coded-image chunk.
|
||||
|
||||
The WebP layout puts ``XMP ``/``EXIF`` AFTER the pixels, so a fixed read can stop
|
||||
before an IPTC or C2PA AI label. This is the RIFF analogue of
|
||||
:func:`_png_late_metadata`; it returns only chunks past ``window`` so bytes
|
||||
already in the head are not duplicated, and empty when there are none.
|
||||
|
||||
``max_total`` caps what a metadata scan can pull into memory, the same ceiling
|
||||
``isobmff.scan_c2pa_region`` applies. Clamping each chunk to the bytes that remain
|
||||
is not enough on its own: a corrupt or crafted file can declare one ``XMP `` chunk
|
||||
spanning most of itself, and this runs on the memoized verdict path for images from
|
||||
arbitrary sources. A label that needs more than 4 MB of XMP does not exist.
|
||||
"""
|
||||
out = bytearray()
|
||||
try:
|
||||
with open(image_path, "rb") as f:
|
||||
if f.read(4) != b"RIFF":
|
||||
return b""
|
||||
f.seek(0, 2)
|
||||
file_size = f.tell()
|
||||
f.seek(4)
|
||||
declared_size = f.read(4)
|
||||
if len(declared_size) < 4:
|
||||
return b""
|
||||
container_end = min(file_size, 8 + struct.unpack("<I", declared_size)[0])
|
||||
position = 12 # 'RIFF' + size + form type
|
||||
while position + 8 <= container_end and len(out) < max_total:
|
||||
f.seek(position)
|
||||
header = f.read(8)
|
||||
if len(header) < 8:
|
||||
break
|
||||
chunk_type = header[:4]
|
||||
(length,) = struct.unpack("<I", header[4:8])
|
||||
start = position + 8
|
||||
# Clamp to what remains: a malformed 32-bit length must not push the
|
||||
# walk past EOF and abandon a genuine label chunk after it.
|
||||
safe_length = max(0, min(length, container_end - start))
|
||||
if chunk_type in RIFF_METADATA_CHUNKS and start >= window:
|
||||
f.seek(start)
|
||||
out += f.read(min(safe_length, max_total - len(out)))
|
||||
position = start + safe_length + (safe_length & 1) # chunks are word-aligned
|
||||
except OSError as exc:
|
||||
logger.debug("RIFF late-metadata scan failed on %s: %s", image_path, exc)
|
||||
return b""
|
||||
return bytes(out)
|
||||
|
||||
|
||||
def _stat_key(image_path: Path) -> tuple[str, int, int] | None:
|
||||
"""Cache key identifying this file's exact CONTENT, or None when it cannot stat.
|
||||
|
||||
@@ -306,11 +355,16 @@ def scan_head(image_path: Path, size: int = 1024 * 1024) -> bytes:
|
||||
past large boxes like ``mdat``) and PNG ``tEXt`` / ``iTXt`` / ``eXIf`` chunks
|
||||
(seeking past ``IDAT``).
|
||||
|
||||
A file at least ``size`` bytes long additionally gets the metadata text its
|
||||
decoder can reach but a raw read cannot (:func:`_decoder_visible_text`): a
|
||||
compressed PNG ``zTXt`` packet, or a chunk past the window in a container with no
|
||||
late-chunk reader here. A file that fits inside ``size`` is exactly
|
||||
``f.read(size)``, since the raw read already holds every byte.
|
||||
|
||||
This is the shared input for every C2PA / AIGC / IPTC byte scan. The
|
||||
extensions catch a manifest or XMP packet placed AFTER the media data -- a
|
||||
non-faststart MP4 manifest, or a PNG XMP packet appended after the pixels --
|
||||
which a fixed first-MB read would miss. For other inputs, and for files that
|
||||
fit within ``size``, it is exactly ``f.read(size)`` -- behavior-neutral.
|
||||
which a fixed first-MB read would miss.
|
||||
|
||||
The result is memoized per (path, size, mtime): one ``identify``/``get_ai_metadata``
|
||||
call fans out to ~8 byte-scan detectors that each call this on the same file, so
|
||||
@@ -347,9 +401,63 @@ def _scan_head_impl(image_path: Path, size: int) -> bytes:
|
||||
# len(head) == size means the file is at least `size` bytes, so metadata
|
||||
# chunks may lie beyond the window; otherwise the whole PNG is in `head`.
|
||||
head += _png_late_metadata(image_path, size)
|
||||
elif head[:4] == b"RIFF" and head[8:12] == b"WEBP" and len(head) == size:
|
||||
head += _riff_late_metadata(image_path, size)
|
||||
if len(head) >= size:
|
||||
head += _decoder_visible_text(image_path, head)
|
||||
return head
|
||||
|
||||
|
||||
# Text values the image decoder can reach that a raw byte read cannot. Bounded: a
|
||||
# packet larger than this is not a provenance label.
|
||||
_DECODED_TEXT_LIMIT = 512 * 1024
|
||||
# Decoder values that are binary payloads with their own readers, not metadata text.
|
||||
# An ICC profile is colour data and can run to hundreds of kilobytes; appending it
|
||||
# would bloat the buffer every later detector re-scans, for no signal.
|
||||
_DECODER_BINARY_KEYS = frozenset({"icc_profile"})
|
||||
|
||||
|
||||
def _decoder_visible_text(image_path: Path, head: bytes) -> bytes:
|
||||
"""Metadata text PIL can decode but the raw window does not contain.
|
||||
|
||||
This is the last of two layers, not the first. Metadata placed BEYOND the window
|
||||
is the structural readers' job (``_png_late_metadata``, ``_riff_late_metadata``,
|
||||
the ISOBMFF box walk), and they work on a file no decoder can open. What is left
|
||||
for this one is metadata the bytes do not spell at all:
|
||||
|
||||
* COMPRESSED -- a PNG ``zTXt`` chunk is zlib-deflated, so an XMP packet carrying
|
||||
a TC260 AIGC label is unreadable as bytes while PIL inflates it on open.
|
||||
|
||||
It stays container-agnostic on purpose: it is the net under a placement no
|
||||
structural reader here knows about yet.
|
||||
|
||||
Only text ALREADY MISSING from ``head`` is appended, so the common case adds
|
||||
nothing and no detector sees a value twice. Skipped entirely when the file fits
|
||||
inside the window, since then the raw read already holds every byte.
|
||||
"""
|
||||
try:
|
||||
from PIL import Image
|
||||
|
||||
with Image.open(image_path) as img:
|
||||
values = [value for key, value in img.info.items() if key not in _DECODER_BINARY_KEYS]
|
||||
except Exception as exc: # a container PIL cannot open: the raw scan stands alone
|
||||
logger.debug("decoder-visible text unavailable for %s: %s", image_path, exc)
|
||||
return b""
|
||||
|
||||
out = bytearray()
|
||||
for value in values:
|
||||
if isinstance(value, str):
|
||||
encoded = value.encode("utf-8", "replace")
|
||||
elif isinstance(value, bytes):
|
||||
encoded = value
|
||||
else:
|
||||
continue
|
||||
if len(encoded) > _DECODED_TEXT_LIMIT or not encoded or encoded in head:
|
||||
continue
|
||||
out += b"\x00" + encoded
|
||||
return bytes(out)
|
||||
|
||||
|
||||
def has_ai_metadata(image_path: Path) -> bool:
|
||||
"""Check if an image contains AI-generation metadata.
|
||||
|
||||
@@ -1485,3 +1593,16 @@ def xai_signature(image_path: Path) -> bool:
|
||||
if key is None:
|
||||
return _xai_signature_impl(image_path)
|
||||
return _xai_signature_cached(*key)
|
||||
|
||||
|
||||
# ── Shared with the portable metadata record ────────────────────────
|
||||
# `metadata_record` must read exactly the windows and markers the file path reads: a
|
||||
# record built from a different window is a record whose verdict can disagree with
|
||||
# `identify` on the same image. Aliased rather than renamed because the private names
|
||||
# are load-bearing in this module's own tests and in a corpus script.
|
||||
QUICK_SCAN_BYTES = _QUICK_SCAN_BYTES
|
||||
SAMSUNG_EDITOR_MARKER = _SAMSUNG_EDITOR_MARKER
|
||||
read_file_tail = _read_file_tail
|
||||
png_late_metadata = _png_late_metadata
|
||||
riff_late_metadata = _riff_late_metadata
|
||||
exif_text = _exif_text
|
||||
|
||||
@@ -0,0 +1,404 @@
|
||||
"""Collect one image's provenance metadata into a portable, JSON-safe record.
|
||||
|
||||
WHY THIS EXISTS
|
||||
|
||||
``extract_provenance_evidence`` reads a file and hands back evidence in memory, so
|
||||
collection and verdict must happen in the same process, on the machine holding the
|
||||
image. This module splits them: collect here, judge anywhere, from a record that
|
||||
survives JSON.
|
||||
|
||||
record = collect_metadata_record(path) # touches the file
|
||||
evidence = evidence_from_metadata_record(record, path=path)
|
||||
report = identify_from_evidence(evidence) # touches nothing
|
||||
|
||||
WHAT GOES IN, AND WHY NOT SIMPLY THE FILE HEAD
|
||||
|
||||
The verdict reads a scan buffer that ``scan_head`` fills with the first mebibyte of
|
||||
the file. Shipping that verbatim would make a record larger than a phone photo's
|
||||
worth of metadata by two orders of magnitude, because for a PNG almost all of that
|
||||
mebibyte is compressed pixel data in ``IDAT`` -- bytes no provenance token can ever
|
||||
live in. A record carries the metadata REGIONS instead, walked per container: the
|
||||
JPEG marker segments before the coded scan, every PNG chunk but ``IDAT``, the RIFF
|
||||
chunks that are not coded image, the ISOBMFF provenance boxes, and in every case the
|
||||
container's trailer.
|
||||
|
||||
COMPLETENESS IS A MEASURED PROPERTY, NOT A CLAIM
|
||||
|
||||
A region walker is only correct if nothing the verdict reads falls outside the
|
||||
regions it keeps, and no test over fixtures can establish that: the failure mode is
|
||||
a container placement nobody thought of. The contract is therefore ALSO verified
|
||||
against the file path over a real corpus -- same image, both paths, identical
|
||||
``ProvenanceReport``.
|
||||
|
||||
The placements that defeated an earlier draft of this collector, and the reason each
|
||||
rule below exists, are recorded in ``docs/module-internals.md`` under "Portable
|
||||
metadata record".
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import logging
|
||||
import struct
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
|
||||
from remove_ai_watermarks._internal.constants import PNG_SIGNATURE, RIFF_CODED_IMAGE_CHUNKS
|
||||
from remove_ai_watermarks._internal.schema import require_schema_version
|
||||
from remove_ai_watermarks.metadata import (
|
||||
QUICK_SCAN_BYTES,
|
||||
SAMSUNG_EDITOR_MARKER,
|
||||
exif_text,
|
||||
read_file_tail,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# The structural walk covers the same window the file path reads raw, so the two
|
||||
# cannot disagree about a chunk type inside it. A smaller window would be cheaper but
|
||||
# opens a blind spot: past the window only ``png_late_metadata``'s ALLOWLIST is
|
||||
# collected, while the file path still sees every chunk type up to its own window --
|
||||
# and a C2PA ``caBX`` chunk is in neither that allowlist nor ``IDAT``. Walking here
|
||||
# costs little because the payload of the pixel stream is skipped, not copied.
|
||||
HEAD_WINDOW = 1024 * 1024
|
||||
# The window searched for the container's end marker. Matches the quick-scan window
|
||||
# the file path uses when it goes looking for a Samsung trailer, so a trailer visible
|
||||
# to one path is visible to the other.
|
||||
TAIL_WINDOW = QUICK_SCAN_BYTES
|
||||
# Kept from the tail when no end marker is found, so an unrecognized container still
|
||||
# contributes its last bytes without carrying half a photo.
|
||||
UNKNOWN_TRAILER_WINDOW = 64 * 1024
|
||||
|
||||
# PNG text keys the file path reads for a generator tag, in ITS order. NovelAI stamps
|
||||
# Software/Source/Title rather than EXIF, and the first match wins, so order matters.
|
||||
_GENERATOR_TEXT_KEYS = ("Software", "Source", "Title", "Description")
|
||||
# Stable transport contract for records produced by this module. The version is
|
||||
# deliberately separate from the verdict version: collection and interpretation can
|
||||
# evolve independently as long as old records remain readable.
|
||||
METADATA_RECORD_SCHEMA_VERSION = 1
|
||||
METADATA_RECORD_TYPE = "provenance_metadata"
|
||||
|
||||
|
||||
def _jpeg_regions(data: bytes) -> bytes:
|
||||
"""Every marker segment up to the entropy-coded scan, plus the trailer after EOI.
|
||||
|
||||
The scan itself is skipped by walking to SOS and then jumping to the trailing
|
||||
EOI, so a 20 MB photo contributes only its markers.
|
||||
|
||||
TWIN: ``metadata._strip_jpeg_metadata_lossless`` walks the same marker chain. The
|
||||
two were left separate on purpose -- that one couples the walk to "return False and
|
||||
fall back to a PIL re-encode", a decision the lossless strip path owns and this one
|
||||
must not inherit -- so a fix to marker handling belongs in BOTH.
|
||||
"""
|
||||
out = bytearray()
|
||||
index, size = 2, len(data)
|
||||
while index + 1 < size:
|
||||
if data[index] != 0xFF:
|
||||
break # malformed boundary: keep what was collected, the tail still follows
|
||||
marker = data[index + 1]
|
||||
if marker in (0xDA, 0xD9): # SOS / EOI: the coded scan follows
|
||||
break
|
||||
if 0xD0 <= marker <= 0xD7 or marker == 0x01: # standalone, no length
|
||||
index += 2
|
||||
continue
|
||||
if index + 4 > size:
|
||||
break
|
||||
segment_length = int.from_bytes(data[index + 2 : index + 4], "big")
|
||||
end = index + 2 + segment_length
|
||||
if segment_length < 2 or end > size:
|
||||
break
|
||||
out += data[index:end]
|
||||
index = end
|
||||
return bytes(out)
|
||||
|
||||
|
||||
def _png_regions(data: bytes) -> bytes:
|
||||
"""Every chunk except the ``IDAT`` payloads, plus whatever follows IEND.
|
||||
|
||||
TWIN: ``metadata._png_late_metadata`` walks the same chunk chain by SEEKING over
|
||||
the file rather than over a buffer, and keeps an allowlist rather than skipping
|
||||
``IDAT``. Both filters are deliberate: inside the window the file path sees every
|
||||
chunk type raw, past it only the allowlist survives.
|
||||
"""
|
||||
out = bytearray()
|
||||
size = len(data)
|
||||
position = len(PNG_SIGNATURE)
|
||||
while position + 8 <= size:
|
||||
(length,) = struct.unpack(">I", data[position : position + 4])
|
||||
chunk_type = data[position + 4 : position + 8]
|
||||
start = position + 8
|
||||
# Clamp the length to the bytes that remain: a malformed 32-bit length must
|
||||
# not push the walk past EOF and abandon a genuine label chunk after it.
|
||||
safe_length = max(0, min(length, size - start))
|
||||
if chunk_type != b"IDAT":
|
||||
out += chunk_type + data[start : start + safe_length]
|
||||
position = start + safe_length + 4 # payload + CRC
|
||||
if chunk_type == b"IEND":
|
||||
out += data[position:] # a trailer past IEND is metadata too
|
||||
break
|
||||
return bytes(out)
|
||||
|
||||
|
||||
def _riff_regions(data: bytes) -> bytes:
|
||||
"""Every RIFF chunk except the coded image payloads.
|
||||
|
||||
TWIN: ``metadata._riff_late_metadata`` (seek-based, past the scan window) and
|
||||
``_internal.riff`` (AVI ``LIST/INFO``). Same chunk-stepping arithmetic, three
|
||||
input models.
|
||||
"""
|
||||
out = bytearray(data[:12]) # 'RIFF' + size + 'WEBP'
|
||||
declared_end = 8 + struct.unpack("<I", data[4:8])[0] if len(data) >= 12 else len(data)
|
||||
size = min(len(data), declared_end)
|
||||
position = 12
|
||||
while position + 8 <= size:
|
||||
chunk_type = data[position : position + 4]
|
||||
(length,) = struct.unpack("<I", data[position + 4 : position + 8])
|
||||
start = position + 8
|
||||
safe_length = max(0, min(length, size - start))
|
||||
if chunk_type not in RIFF_CODED_IMAGE_CHUNKS:
|
||||
out += chunk_type + data[start : start + safe_length]
|
||||
position = start + safe_length + (safe_length & 1) # chunks are word-aligned
|
||||
return bytes(out)
|
||||
|
||||
|
||||
def _isobmff_regions(image_path: Path, head: bytes) -> bytes:
|
||||
"""Header window plus the provenance regions the bounded box walkers find.
|
||||
|
||||
ISOBMFF hides a manifest in a ``uuid``/``jumb`` box that can sit after a
|
||||
multi-megabyte ``mdat``, and a TC260 label in ``moov.udta``. Both walkers seek
|
||||
rather than read the media, so neither pulls the payload in.
|
||||
"""
|
||||
from remove_ai_watermarks._internal.isobmff import scan_c2pa_region, tc260_aigc_payloads
|
||||
|
||||
out = bytearray(head[:HEAD_WINDOW])
|
||||
try:
|
||||
out += scan_c2pa_region(image_path)
|
||||
except Exception as exc:
|
||||
logger.debug("ISOBMFF C2PA region scan failed on %s: %s", image_path, exc)
|
||||
try:
|
||||
for payload in tc260_aigc_payloads(image_path):
|
||||
out += payload
|
||||
except Exception as exc:
|
||||
logger.debug("ISOBMFF TC260 scan failed on %s: %s", image_path, exc)
|
||||
return bytes(out)
|
||||
|
||||
|
||||
def _container_regions(image_path: Path, head: bytes) -> tuple[str, bytes]:
|
||||
"""(container label, metadata bytes) for the container ``head`` starts with.
|
||||
|
||||
``head`` must be the file's raw first bytes. Handing this the ``scan_head``
|
||||
buffer instead is a trap: that buffer is the head CONCATENATED with late metadata
|
||||
payloads, so a structural walk runs off the end of the real head and parses the
|
||||
appended bytes as chunks, inflating the record and creating false signals.
|
||||
"""
|
||||
from remove_ai_watermarks._internal.isobmff import is_isobmff
|
||||
from remove_ai_watermarks.metadata import png_late_metadata, riff_late_metadata
|
||||
|
||||
if head.startswith(b"\xff\xd8"):
|
||||
return "jpeg", _jpeg_regions(head)
|
||||
if head.startswith(PNG_SIGNATURE):
|
||||
# Chunks placed after the pixel stream (an XMP packet at 2.7 MB, say) are
|
||||
# past the window; the same seek-past-IDAT reader the file path uses gets them.
|
||||
return "png", _png_regions(head) + png_late_metadata(image_path, HEAD_WINDOW)
|
||||
if head.startswith(b"RIFF") and head[8:12] == b"WEBP":
|
||||
return "webp", _riff_regions(head) + riff_late_metadata(image_path, HEAD_WINDOW)
|
||||
if is_isobmff(head):
|
||||
return "isobmff", _isobmff_regions(image_path, head)
|
||||
return "unknown", head
|
||||
|
||||
|
||||
def _raw_head(image_path: Path) -> bytes:
|
||||
"""The file's first bytes, unmodified -- the input every structural walk needs."""
|
||||
try:
|
||||
with open(image_path, "rb") as handle:
|
||||
return handle.read(HEAD_WINDOW)
|
||||
except OSError as exc:
|
||||
logger.debug("head read failed for %s: %s", image_path, exc)
|
||||
return b""
|
||||
|
||||
|
||||
def _trailer(image_path: Path, container: str) -> bytes:
|
||||
"""The bytes that follow the container's end marker, and nothing else.
|
||||
|
||||
A fixed-size tail read would be almost entirely pixels: the trailer of a 20 MB
|
||||
photo is a few kilobytes at most. So the end marker is located in the tail window
|
||||
and only what follows it is kept. When no marker is found (an unknown container,
|
||||
or one whose end lies before the window) the window is kept as-is, bounded --
|
||||
that is what a byte scan of the same file would have seen anyway.
|
||||
"""
|
||||
if container == "webp":
|
||||
# RIFF declares its structural end in bytes 4..8. A fixed tail window is
|
||||
# normally the last animation/frame payload, not a trailer, so preserve
|
||||
# only bytes appended after the declared RIFF container.
|
||||
try:
|
||||
with open(image_path, "rb") as handle:
|
||||
header = handle.read(12)
|
||||
if len(header) < 12 or not header.startswith(b"RIFF"):
|
||||
return b""
|
||||
declared_end = 8 + struct.unpack("<I", header[4:8])[0]
|
||||
handle.seek(0, 2)
|
||||
file_size = handle.tell()
|
||||
if declared_end < 12 or declared_end >= file_size:
|
||||
return b""
|
||||
handle.seek(declared_end)
|
||||
return handle.read(min(file_size - declared_end, UNKNOWN_TRAILER_WINDOW))
|
||||
except OSError as exc:
|
||||
logger.debug("RIFF trailer read failed for %s: %s", image_path, exc)
|
||||
return b""
|
||||
if container == "isobmff":
|
||||
# ISOBMFF has no out-of-container trailer convention. Its bounded box
|
||||
# walkers already collect late provenance while skipping ``mdat``; keeping
|
||||
# a blind tail here would carry coded media bytes.
|
||||
return b""
|
||||
|
||||
tail = read_file_tail(image_path, TAIL_WINDOW)
|
||||
if SAMSUNG_EDITOR_MARKER in tail:
|
||||
# Galaxy AI splits its evidence: the marker sits in the post-EOI trailer, but
|
||||
# the `genAIType` value it is gated on can sit INSIDE the entropy-coded scan.
|
||||
# Keeping only the trailer therefore carries the marker without the value and the
|
||||
# verdict silently drops the Samsung signal, so a marked file keeps the whole
|
||||
# window. Only Samsung-marked files pay for it.
|
||||
return tail
|
||||
marker = {"jpeg": b"\xff\xd9", "png": b"IEND\xae\x42\x60\x82"}.get(container)
|
||||
if marker is None:
|
||||
return tail[-UNKNOWN_TRAILER_WINDOW:]
|
||||
index = tail.rfind(marker)
|
||||
return tail[index + len(marker) :] if index >= 0 else tail[-UNKNOWN_TRAILER_WINDOW:]
|
||||
|
||||
|
||||
def _decoder_info(image_path: Path) -> dict[str, Any]:
|
||||
"""PIL's ``info`` mapping, read once.
|
||||
|
||||
One open for both consumers below. They want different parts of the same mapping
|
||||
(the text keys, and the raw EXIF blob), and opening twice repeats the container
|
||||
header parse and, for a PNG carrying ``zTXt``, the zlib inflate with it.
|
||||
"""
|
||||
try:
|
||||
from PIL import Image
|
||||
|
||||
with Image.open(image_path) as img:
|
||||
# PIL types this mapping with a non-string key union (a DPI tuple key
|
||||
# exists), so the keys are normalized here rather than assumed.
|
||||
return {str(key): value for key, value in img.info.items()}
|
||||
except Exception as exc: # a container PIL cannot open
|
||||
logger.debug("PIL info unavailable for %s: %s", image_path, exc)
|
||||
return {}
|
||||
|
||||
|
||||
def _exif_pairs(info: dict[str, Any]) -> dict[str, str]:
|
||||
"""The 0th-IFD tags the verdict reads, under their tag NAMES.
|
||||
|
||||
Not a convenience: two probes key on names rather than on the raw bytes already
|
||||
in the regions. ``xai_signature_pair`` wants an (ImageDescription, Artist) pair,
|
||||
and ``_external_exif_generator`` looks for Software / Make / Artist /
|
||||
ImageDescription. Ship the bytes alone and both silently return nothing, which
|
||||
is how a collector can silently lose Grok and NovelAI verdicts.
|
||||
"""
|
||||
exif_bytes = info.get("exif")
|
||||
if not exif_bytes:
|
||||
return {}
|
||||
try:
|
||||
import piexif
|
||||
|
||||
tags = piexif.load(exif_bytes).get("0th", {})
|
||||
except Exception as exc: # malformed EXIF
|
||||
logger.debug("EXIF parse failed: %s", exc)
|
||||
return {}
|
||||
|
||||
return {
|
||||
name: text
|
||||
for name, tag in (
|
||||
("Software", piexif.ImageIFD.Software),
|
||||
("Make", piexif.ImageIFD.Make),
|
||||
("Artist", piexif.ImageIFD.Artist),
|
||||
("ImageDescription", piexif.ImageIFD.ImageDescription),
|
||||
)
|
||||
if (text := exif_text(tags, tag))
|
||||
}
|
||||
|
||||
|
||||
def _pil_info(info: dict[str, Any]) -> dict[str, str]:
|
||||
"""PIL's ``info`` mapping as strings, the source of PNG text keys and ``hf-job-id``."""
|
||||
|
||||
def text_of(value: Any) -> str:
|
||||
return value.decode("utf-8", "replace") if isinstance(value, bytes) else str(value)
|
||||
|
||||
# Emitted in the file path's own candidate order. ``generator_from_metadata``
|
||||
# returns the FIRST candidate carrying a known token. A record using PIL's natural
|
||||
# dict order can therefore choose a different platform string than the file path,
|
||||
# even though the two paths are supposed to be indistinguishable.
|
||||
out: dict[str, str] = {}
|
||||
for key in _GENERATOR_TEXT_KEYS:
|
||||
value = info.get(key)
|
||||
if value is not None and not isinstance(value, (dict, list, tuple)):
|
||||
out[f"info:{key}"] = text_of(value)
|
||||
for key, value in info.items():
|
||||
if key in _GENERATOR_TEXT_KEYS or isinstance(value, (dict, list, tuple)):
|
||||
continue
|
||||
out[f"info:{key}"] = text_of(value)
|
||||
return out
|
||||
|
||||
|
||||
def collect_metadata_record(
|
||||
image_path: Path,
|
||||
*,
|
||||
schema_version: int = METADATA_RECORD_SCHEMA_VERSION,
|
||||
) -> dict[str, Any]:
|
||||
"""Collect everything the provenance verdict reads, as a JSON-safe record.
|
||||
|
||||
The record is the transport format for
|
||||
:func:`identify.evidence_from_metadata_record`: it carries the metadata regions
|
||||
(base64), the C2PA manifest store, and PIL's info mapping without carrying the
|
||||
primary coded-pixel stream. Schema and collection status are explicit so a
|
||||
consumer cannot mistake a failed read for an unknown provenance verdict.
|
||||
|
||||
Args:
|
||||
image_path: Path to the image.
|
||||
schema_version: Output schema implemented by the consumer.
|
||||
|
||||
Returns:
|
||||
A versioned JSON-serializable dict. ``metadata_base64`` holds the
|
||||
concatenated container regions, ``tail_base64`` the file trailer.
|
||||
"""
|
||||
schema_version = require_schema_version(
|
||||
schema_version,
|
||||
contract="provenance metadata",
|
||||
supported=(1,),
|
||||
)
|
||||
|
||||
from remove_ai_watermarks._internal.c2pa import read_manifest_store_json
|
||||
|
||||
try:
|
||||
image_path.stat()
|
||||
status = "complete"
|
||||
issues: list[dict[str, str]] = []
|
||||
except OSError as exc:
|
||||
logger.debug("metadata source unavailable for %s: %s", image_path, exc)
|
||||
status = "error"
|
||||
issues = [{"stage": "source", "code": "unavailable"}]
|
||||
|
||||
container, regions = _container_regions(image_path, _raw_head(image_path))
|
||||
|
||||
info = _decoder_info(image_path)
|
||||
record: dict[str, Any] = {
|
||||
"schema_version": schema_version,
|
||||
"record_type": METADATA_RECORD_TYPE,
|
||||
"status": status,
|
||||
"issues": issues,
|
||||
"container": container,
|
||||
"name": image_path.name,
|
||||
"metadata_base64": base64.b64encode(regions).decode("ascii"),
|
||||
# Always collected: Samsung's Galaxy AI marker is a post-EOI trailer, and a
|
||||
# record without it loses that verdict outright.
|
||||
"tail_base64": base64.b64encode(_trailer(image_path, container)).decode("ascii"),
|
||||
# PIL info BEFORE exif: the file path prefers a PNG text tag over an EXIF
|
||||
# one, and the normalizer walks the record in insertion order.
|
||||
"pil": _pil_info(info),
|
||||
"exif": _exif_pairs(info),
|
||||
}
|
||||
store = read_manifest_store_json(image_path)
|
||||
if store is not None:
|
||||
record["c2pa_store"] = store
|
||||
return record
|
||||
@@ -0,0 +1,484 @@
|
||||
# pyright: reportUnknownMemberType=false, reportUnknownArgumentType=false, reportUnknownVariableType=false, reportMissingTypeStubs=false
|
||||
"""The complete pixel-forensics layer for one image.
|
||||
|
||||
STATUS
|
||||
|
||||
Independent from provenance verdicts, removal, and the CLI. Consumers use the
|
||||
versioned :meth:`PixelEvidence.to_dict` boundary; feature extraction failures are
|
||||
reported per family without discarding successful measurements.
|
||||
|
||||
WHAT IS MEASURED
|
||||
|
||||
One decode, then six families of scale-robust statistics over it:
|
||||
|
||||
* ``dct`` -- AC coefficient histograms over the 8x8 block DCT, plus the deviation of
|
||||
leading digits from Benford's law.
|
||||
* ``fft`` -- radial band energies of the log-magnitude spectrum, plus the
|
||||
color-filter-array periodicity peaks a demosaiced camera capture leaves.
|
||||
* ``noise`` -- standard deviation and kurtosis of a high-pass residual.
|
||||
* ``ela`` -- error level after a quality-90 JPEG re-save.
|
||||
* ``gradient`` -- gradient-magnitude histogram and Laplacian variance.
|
||||
* ``color`` -- 4x4x4 RGB histogram, mean saturation, mean value.
|
||||
|
||||
and, in ``artifacts``, the spatial layer those statistics are computed from: a
|
||||
64-bit perceptual hash, a 128px JPEG thumbnail, and coarse ELA, noise-residual and
|
||||
FFT-phase maps.
|
||||
|
||||
THE ARTIFACTS ARE NOT AGGREGATES
|
||||
|
||||
Everything above ``artifacts`` is a scalar or a fixed-length histogram, and an image
|
||||
cannot be reconstructed from those. ``artifacts`` is different in kind: a thumbnail
|
||||
is a picture, a perceptual hash identifies one, and the coarse maps carry layout.
|
||||
Collecting them makes a record that identifies the source image, so a caller storing
|
||||
or forwarding them is handling image content, not statistics about it. That is why
|
||||
they are a separate field and not merged into the families.
|
||||
|
||||
REQUIREMENTS
|
||||
|
||||
Needs the ``pixels`` extra (numpy). Guard a call with :func:`is_available` when the
|
||||
caller must not hard-depend on it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import io
|
||||
import logging
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from remove_ai_watermarks._internal.schema import require_schema_version
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Analysis resolution. Every statistic here is scale-robust, and a 2048px cap keeps
|
||||
# the FFT and the sliding-window residual bounded on a 100 MP input.
|
||||
MAX_SIDE = 2048
|
||||
# The eight lowest-frequency AC positions of the 8x8 block DCT, zig-zag order.
|
||||
AC_POSITIONS = ((0, 1), (1, 0), (1, 1), (0, 2), (2, 0), (2, 1), (1, 2), (0, 3))
|
||||
FFT_BANDS = 8
|
||||
# A Bayer CFA shows as symmetric peaks at half the Nyquist on the diagonals.
|
||||
BAYER_OFFSETS = ((1, 1), (1, -1))
|
||||
INSTALL_HINT = "install the pixel extra: uv add 'remove-ai-watermarks[pixels]'"
|
||||
PIXEL_EVIDENCE_SCHEMA_VERSION = 1
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PixelEvidence:
|
||||
"""Pixel statistics for one image, and the spatial artifacts behind them.
|
||||
|
||||
``decode`` carries the source dimensions, or ``{"error": ...}`` when the image
|
||||
could not be decoded -- in which case every other field is empty. A family is also
|
||||
empty when the image is too small for it (the block DCT needs 8x8, the FFT 32x32,
|
||||
the residual 3x3), so a caller must treat every field as optional rather than
|
||||
assume a fixed feature width.
|
||||
"""
|
||||
|
||||
path: Path
|
||||
decode: dict[str, Any]
|
||||
dct: dict[str, Any] = field(default_factory=dict[str, Any])
|
||||
fft: dict[str, Any] = field(default_factory=dict[str, Any])
|
||||
noise: dict[str, Any] = field(default_factory=dict[str, Any])
|
||||
ela: dict[str, Any] = field(default_factory=dict[str, Any])
|
||||
gradient: dict[str, Any] = field(default_factory=dict[str, Any])
|
||||
color: dict[str, Any] = field(default_factory=dict[str, Any])
|
||||
# Identifies the source image; see the module note. Empty unless asked for.
|
||||
artifacts: dict[str, Any] = field(default_factory=dict[str, Any])
|
||||
# Opt-in timings for callers measuring pipeline latency. Empty by default so
|
||||
# repeated evidence collection remains value-deterministic.
|
||||
timing_ms: dict[str, float] = field(default_factory=dict[str, float])
|
||||
|
||||
@property
|
||||
def decoded(self) -> bool:
|
||||
"""False when the source could not be decoded at all."""
|
||||
return "error" not in self.decode
|
||||
|
||||
@property
|
||||
def status(self) -> str:
|
||||
"""``complete``, ``partial`` for a failed family, or ``error`` on decode."""
|
||||
if not self.decoded:
|
||||
return "error"
|
||||
sections = (self.dct, self.fft, self.noise, self.ela, self.gradient, self.color, self.artifacts)
|
||||
return "partial" if any("error" in section for section in sections) else "complete"
|
||||
|
||||
def to_dict(
|
||||
self,
|
||||
*,
|
||||
schema_version: int = PIXEL_EVIDENCE_SCHEMA_VERSION,
|
||||
) -> dict[str, Any]:
|
||||
"""Return the selected JSON-safe transport schema without a local path."""
|
||||
schema_version = require_schema_version(
|
||||
schema_version,
|
||||
contract="pixel evidence",
|
||||
supported=(1,),
|
||||
)
|
||||
return {
|
||||
"schema_version": schema_version,
|
||||
"status": self.status,
|
||||
"decode": dict(self.decode),
|
||||
"dct": dict(self.dct),
|
||||
"fft": dict(self.fft),
|
||||
"noise": dict(self.noise),
|
||||
"ela": dict(self.ela),
|
||||
"gradient": dict(self.gradient),
|
||||
"color": dict(self.color),
|
||||
"artifacts": dict(self.artifacts),
|
||||
"timing_ms": dict(self.timing_ms),
|
||||
}
|
||||
|
||||
|
||||
def is_available() -> bool:
|
||||
"""True when the optional pixel dependencies are installed."""
|
||||
from remove_ai_watermarks.optional_deps import module_available
|
||||
|
||||
return module_available("numpy")
|
||||
|
||||
|
||||
def _numpy() -> Any:
|
||||
from remove_ai_watermarks.optional_deps import module_available
|
||||
|
||||
if not module_available("numpy"):
|
||||
raise RuntimeError(f"Pixel evidence needs numpy -- {INSTALL_HINT}")
|
||||
import numpy as np
|
||||
|
||||
return np
|
||||
|
||||
|
||||
def _dct_matrix(np: Any, n: int = 8) -> Any:
|
||||
"""Orthonormal n x n DCT-II basis: M[i, j] = cos(pi (2j + 1) i / 2n)."""
|
||||
i = np.arange(n)[:, None]
|
||||
j = np.arange(n)[None, :]
|
||||
m = np.cos(np.pi * (2 * j + 1) * i / (2 * n))
|
||||
m[0, :] *= 1 / np.sqrt(2)
|
||||
return m * np.sqrt(2 / n)
|
||||
|
||||
|
||||
def read_gray(image_path: Path) -> tuple[Any, Any, dict[str, Any]]:
|
||||
"""Decode to float32 grayscale (and RGB for color stats), downscaled.
|
||||
|
||||
Pillow, not cv2, and the source dimensions are recorded BEFORE the downscale.
|
||||
"""
|
||||
np = _numpy()
|
||||
from PIL import Image
|
||||
|
||||
from remove_ai_watermarks import image_io
|
||||
|
||||
try:
|
||||
image_io._register_heif() # pyright: ignore[reportPrivateUsage]
|
||||
with Image.open(image_path) as img:
|
||||
info: dict[str, Any] = {"width": img.width, "height": img.height}
|
||||
if max(img.size) > MAX_SIDE:
|
||||
img.thumbnail((MAX_SIDE, MAX_SIDE), Image.Resampling.LANCZOS)
|
||||
rgb = np.asarray(img.convert("RGB"), dtype=np.float32)
|
||||
gray = np.asarray(img.convert("L"), dtype=np.float32)
|
||||
except Exception as exc:
|
||||
logger.debug("pixel decode failed for %s: %s", image_path, exc)
|
||||
# Exception text from Pillow commonly embeds the absolute source path.
|
||||
# Keep that detail in the log, not in the pathless transport contract.
|
||||
return None, None, {"error": type(exc).__name__}
|
||||
return gray, rgb, info
|
||||
|
||||
|
||||
def dct_features(gray: Any) -> dict[str, Any]:
|
||||
"""AC coefficient histograms over the 8x8 block DCT + Benford deviation."""
|
||||
np = _numpy()
|
||||
height, width = gray.shape
|
||||
h8, w8 = height // 8 * 8, width // 8 * 8
|
||||
if h8 < 8 or w8 < 8:
|
||||
return {}
|
||||
basis = _dct_matrix(np)
|
||||
bins = np.linspace(-20.5, 20.5, 22)
|
||||
blocks = gray[:h8, :w8].reshape(h8 // 8, 8, w8 // 8, 8).swapaxes(1, 2)
|
||||
rows = basis[[row for row, _ in AC_POSITIONS]]
|
||||
columns = basis[[column for _, column in AC_POSITIONS]]
|
||||
coeff = np.einsum("ki,abij,kj->abk", rows, blocks, columns)
|
||||
hists = []
|
||||
lead_vals: list[Any] = []
|
||||
for index in range(len(AC_POSITIONS)):
|
||||
values = coeff[:, :, index].ravel()
|
||||
hists.append(np.histogram(values, bins=bins)[0].tolist())
|
||||
lead_vals.append(np.abs(values))
|
||||
out: dict[str, Any] = {"dct_ac_hist": hists}
|
||||
flat = np.abs(np.concatenate(lead_vals))
|
||||
flat = flat[flat >= 1]
|
||||
if flat.size > 100:
|
||||
leading = (flat / 10 ** np.floor(np.log10(flat))).astype(int)
|
||||
leading = leading[(leading >= 1) & (leading <= 9)]
|
||||
if leading.size > 100:
|
||||
observed = np.bincount(leading, minlength=10)[1:10] / leading.size
|
||||
benford = np.log10(1 + 1 / np.arange(1, 10))
|
||||
out["benford_mad"] = float(np.abs(observed - benford).mean())
|
||||
return out
|
||||
|
||||
|
||||
def noise_residual_map(gray: Any) -> Any:
|
||||
"""High-pass residual, the map the noise statistics are computed from."""
|
||||
np = _numpy()
|
||||
from numpy.lib.stride_tricks import sliding_window_view
|
||||
|
||||
if gray.shape[0] < 3 or gray.shape[1] < 3:
|
||||
return None
|
||||
kernel = np.array([[-1.0, -1.0, -1.0], [-1.0, 8.0, -1.0], [-1.0, -1.0, -1.0]])
|
||||
height, width = gray.shape
|
||||
# kernel is float64, so the residual is float64 like the unchunked form
|
||||
out = np.empty((height - 2, width - 2), dtype=np.float64)
|
||||
# Row-chunked: the (window * kernel) temporary is ~150 MB at 2048px if
|
||||
# materialized whole. Per-element 9-tap sums are computed in the same order,
|
||||
# so the result is bit-identical to the unchunked form.
|
||||
for y0 in range(0, height - 2, 256):
|
||||
y1 = min(y0 + 256, height - 2)
|
||||
window = sliding_window_view(gray[y0 : y1 + 2], (3, 3))
|
||||
out[y0:y1] = (window * kernel).sum(axis=(-1, -2))
|
||||
return out
|
||||
|
||||
|
||||
def noise_features(residual: Any) -> dict[str, Any]:
|
||||
"""High-pass residual std and kurtosis."""
|
||||
flat = residual.ravel()
|
||||
std = float(flat.std())
|
||||
if std < 1e-9:
|
||||
return {"noise_std": 0.0, "noise_kurtosis": 0.0}
|
||||
z = (flat - flat.mean()) / std
|
||||
return {"noise_std": std, "noise_kurtosis": float((z**4).mean() - 3.0)}
|
||||
|
||||
|
||||
def fft_decompose(gray: Any) -> tuple[Any, Any] | None:
|
||||
"""Log-magnitude (fftshifted) and phase of the image spectrum."""
|
||||
np = _numpy()
|
||||
if min(gray.shape) < 32:
|
||||
return None
|
||||
spectrum = np.fft.fftshift(np.fft.fft2(gray - gray.mean()))
|
||||
return np.log1p(np.abs(spectrum)), np.angle(spectrum)
|
||||
|
||||
|
||||
def fft_features(mag: Any) -> dict[str, Any]:
|
||||
"""Radial magnitude band energies (no phase) + CFA periodicity peaks."""
|
||||
np = _numpy()
|
||||
height, width = mag.shape
|
||||
cy, cx = height // 2, width // 2
|
||||
# 1D broadcast instead of an mgrid: saves ~160 MB of int64 temporaries at
|
||||
# 2048px. The squares are exact in float64 (values < 2^53), so band means
|
||||
# are identical to the mgrid form.
|
||||
r2y = (np.arange(height, dtype=np.float64) - cy) ** 2
|
||||
r2x = (np.arange(width, dtype=np.float64) - cx) ** 2
|
||||
radius = np.sqrt(r2y[:, None] + r2x[None, :])
|
||||
r_max = radius.max()
|
||||
bands = []
|
||||
for index in range(FFT_BANDS):
|
||||
mask = (radius >= r_max * index / FFT_BANDS) & (radius < r_max * (index + 1) / FFT_BANDS)
|
||||
bands.append(float(mag[mask].mean()) if mask.any() else 0.0)
|
||||
peaks = []
|
||||
for dy, dx in BAYER_OFFSETS:
|
||||
y, x = cy + dy * (height // 4), cx + dx * (width // 4)
|
||||
neighborhood = mag[y - 2 : y + 3, x - 2 : x + 3]
|
||||
peaks.append(float(neighborhood.max() - mag.mean()))
|
||||
return {"fft_band_energy": bands, "cfa_peaks": peaks, "cfa_peak": max(peaks)}
|
||||
|
||||
|
||||
def ela_map(rgb: Any) -> Any:
|
||||
"""Absolute per-pixel error after a quality-90 JPEG re-save."""
|
||||
np = _numpy()
|
||||
from PIL import Image
|
||||
|
||||
try:
|
||||
buffer = io.BytesIO()
|
||||
Image.fromarray(rgb.astype(np.uint8)).save(buffer, "JPEG", quality=90)
|
||||
buffer.seek(0)
|
||||
resaved = np.asarray(Image.open(buffer).convert("RGB"), dtype=np.float32)
|
||||
except Exception as exc:
|
||||
logger.debug("ELA re-save failed: %s", exc)
|
||||
return None
|
||||
if resaved.shape != rgb.shape:
|
||||
return None
|
||||
return np.abs(rgb - resaved).mean(axis=-1)
|
||||
|
||||
|
||||
def ela_features(err: Any) -> dict[str, Any]:
|
||||
"""Error-level stats after a quality-90 JPEG re-save."""
|
||||
np = _numpy()
|
||||
return {"ela_mean": float(err.mean()), "ela_p95": float(np.percentile(err, 95))}
|
||||
|
||||
|
||||
def gradient_features(gray: Any) -> dict[str, Any]:
|
||||
np = _numpy()
|
||||
gy, gx = np.gradient(gray)
|
||||
mag = np.sqrt(gx**2 + gy**2)
|
||||
hist = np.histogram(mag, bins=10, range=(0, 255))[0].tolist()
|
||||
laplacian = np.gradient(gy, axis=0) + np.gradient(gx, axis=1)
|
||||
return {"gradient_hist": hist, "laplacian_var": float(laplacian.var())}
|
||||
|
||||
|
||||
def color_features(rgb: Any) -> dict[str, Any]:
|
||||
np = _numpy()
|
||||
small = rgb[::4, ::4] # decimate; the histogram is position-blind anyway
|
||||
bins = (small / 256 * 4).astype(int).clip(0, 3)
|
||||
index = bins[..., 0] * 16 + bins[..., 1] * 4 + bins[..., 2]
|
||||
hist = np.bincount(index.ravel(), minlength=64).tolist()
|
||||
mx = small.max(axis=-1)
|
||||
mn = small.min(axis=-1)
|
||||
saturation = np.where(mx > 0, (mx - mn) / np.maximum(mx, 1e-6), 0)
|
||||
return {
|
||||
"color_hist_4x4x4": hist,
|
||||
"saturation_mean": float(saturation.mean()),
|
||||
"value_mean": float(mx.mean() / 255),
|
||||
}
|
||||
|
||||
|
||||
def perceptual_hash(gray: Any) -> str:
|
||||
"""64-bit DCT perceptual hash. Identifies an image; see the module note."""
|
||||
np = _numpy()
|
||||
from PIL import Image
|
||||
|
||||
small = np.asarray(Image.fromarray(gray.astype(np.float32), mode="F").resize((32, 32), Image.Resampling.LANCZOS))
|
||||
basis = _dct_matrix(np, 32)
|
||||
low_basis = basis[:8]
|
||||
low = (low_basis @ small @ low_basis.T).ravel()[1:] # drop DC
|
||||
bits = low > np.median(low)
|
||||
return f"{int(''.join('1' if bit else '0' for bit in bits), 2):016x}"
|
||||
|
||||
|
||||
def _coarse(np: Any, arr: Any, side: int = 64) -> Any:
|
||||
"""Downscale a 2D map to at most ``side`` on the long edge."""
|
||||
from PIL import Image
|
||||
|
||||
height, width = arr.shape
|
||||
if max(height, width) <= side:
|
||||
return arr
|
||||
img = Image.fromarray(arr.astype(np.float32), mode="F")
|
||||
img.thumbnail((side, side), Image.Resampling.BILINEAR)
|
||||
return np.asarray(img)
|
||||
|
||||
|
||||
def _array_payload(arr: Any) -> dict[str, Any]:
|
||||
return {
|
||||
"shape": list(arr.shape),
|
||||
"dtype": str(arr.dtype),
|
||||
"base64": base64.b64encode(arr.tobytes()).decode("ascii"),
|
||||
}
|
||||
|
||||
|
||||
def spatial_artifacts(gray: Any, rgb: Any, *, ela: Any, residual: Any, phase: Any) -> dict[str, Any]:
|
||||
"""Perceptual hash, thumbnail, and coarse ELA / residual / phase maps.
|
||||
|
||||
These identify the source image rather than describe it -- see the module note.
|
||||
The maps are the ones the statistics were computed from, passed in rather than
|
||||
recomputed.
|
||||
"""
|
||||
np = _numpy()
|
||||
from PIL import Image
|
||||
|
||||
out: dict[str, Any] = {"phash": perceptual_hash(gray)}
|
||||
thumbnail = Image.fromarray(rgb.astype(np.uint8))
|
||||
thumbnail.thumbnail((128, 128), Image.Resampling.LANCZOS)
|
||||
buffer = io.BytesIO()
|
||||
thumbnail.save(buffer, "JPEG", quality=70)
|
||||
out["thumbnail_jpeg_b64"] = base64.b64encode(buffer.getvalue()).decode("ascii")
|
||||
|
||||
if ela is not None:
|
||||
out["ela_map"] = _array_payload(_coarse(np, ela))
|
||||
if residual is not None:
|
||||
clipped = np.clip(residual / 4.0, -1, 1)
|
||||
out["noise_residual"] = _array_payload(_coarse(np, (clipped * 127).astype(np.int8)))
|
||||
if phase is not None:
|
||||
out["fft_phase"] = _array_payload(_coarse(np, phase.astype(np.float32), 32))
|
||||
return out
|
||||
|
||||
|
||||
def extract_pixel_evidence(image_path: Path, *, artifacts: bool = False, timings: bool = False) -> PixelEvidence:
|
||||
"""Measure every pixel-statistic family for one image in a single decode.
|
||||
|
||||
The image is decoded ONCE and the intermediate maps (high-pass residual, ELA
|
||||
error, FFT magnitude and phase) are computed once and shared, because the
|
||||
residual's sliding window and the ELA re-save are the two expensive steps and
|
||||
each family would otherwise redo them.
|
||||
|
||||
A family that fails or does not apply is left empty rather than raising: an
|
||||
undecodable file, or one too small for the block DCT, still returns a
|
||||
:class:`PixelEvidence` whose ``decoded`` / empty fields say so. Missing numpy is
|
||||
the one hard error, since then nothing can be measured at all.
|
||||
|
||||
Args:
|
||||
image_path: Path to the image. Any container Pillow can open.
|
||||
artifacts: Also return the spatial layer -- perceptual hash, thumbnail and
|
||||
coarse maps. Off by default: those identify the source image, so asking
|
||||
for them is a decision the caller makes explicitly.
|
||||
timings: Measure each stage and include rounded milliseconds in
|
||||
:attr:`PixelEvidence.timing_ms`.
|
||||
|
||||
Returns:
|
||||
A :class:`PixelEvidence`.
|
||||
"""
|
||||
started = time.perf_counter()
|
||||
stage_started = started
|
||||
measured: dict[str, float] = {}
|
||||
|
||||
gray, rgb, info = read_gray(image_path)
|
||||
measured["decode"] = time.perf_counter() - stage_started
|
||||
if gray is None or rgb is None:
|
||||
measured["total"] = time.perf_counter() - started
|
||||
timing_ms = {name: round(seconds * 1000, 1) for name, seconds in measured.items()} if timings else {}
|
||||
return PixelEvidence(path=image_path, decode=info, timing_ms=timing_ms)
|
||||
|
||||
families: dict[str, dict[str, Any]] = {}
|
||||
|
||||
residual = None
|
||||
stage_started = time.perf_counter()
|
||||
try:
|
||||
residual = noise_residual_map(gray)
|
||||
families["noise"] = noise_features(residual) if residual is not None else {}
|
||||
except Exception as exc:
|
||||
logger.debug("pixel family noise failed for %s: %s", image_path, exc)
|
||||
families["noise"] = {"error": type(exc).__name__}
|
||||
measured["noise"] = time.perf_counter() - stage_started
|
||||
|
||||
spectrum = None
|
||||
stage_started = time.perf_counter()
|
||||
try:
|
||||
spectrum = fft_decompose(gray)
|
||||
families["fft"] = fft_features(spectrum[0]) if spectrum is not None else {}
|
||||
except Exception as exc:
|
||||
logger.debug("pixel family fft failed for %s: %s", image_path, exc)
|
||||
families["fft"] = {"error": type(exc).__name__}
|
||||
measured["fft"] = time.perf_counter() - stage_started
|
||||
|
||||
error = None
|
||||
stage_started = time.perf_counter()
|
||||
try:
|
||||
error = ela_map(rgb)
|
||||
families["ela"] = ela_features(error) if error is not None else {}
|
||||
except Exception as exc:
|
||||
logger.debug("pixel family ela failed for %s: %s", image_path, exc)
|
||||
families["ela"] = {"error": type(exc).__name__}
|
||||
measured["ela"] = time.perf_counter() - stage_started
|
||||
|
||||
for name, compute in (
|
||||
("dct", lambda: dct_features(gray)),
|
||||
("gradient", lambda: gradient_features(gray)),
|
||||
("color", lambda: color_features(rgb)),
|
||||
):
|
||||
stage_started = time.perf_counter()
|
||||
try:
|
||||
families[name] = compute()
|
||||
except Exception as exc: # one bad family must not lose the other five
|
||||
logger.debug("pixel family %s failed for %s: %s", name, image_path, exc)
|
||||
families[name] = {"error": type(exc).__name__}
|
||||
measured[name] = time.perf_counter() - stage_started
|
||||
|
||||
if artifacts:
|
||||
stage_started = time.perf_counter()
|
||||
try:
|
||||
families["artifacts"] = spatial_artifacts(
|
||||
gray, rgb, ela=error, residual=residual, phase=spectrum[1] if spectrum is not None else None
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.debug("pixel artifacts failed for %s: %s", image_path, exc)
|
||||
families["artifacts"] = {"error": type(exc).__name__}
|
||||
measured["full_artifacts"] = time.perf_counter() - stage_started
|
||||
|
||||
measured["total"] = time.perf_counter() - started
|
||||
timing_ms = {name: round(seconds * 1000, 1) for name, seconds in measured.items()} if timings else {}
|
||||
return PixelEvidence(path=image_path, decode=info, timing_ms=timing_ms, **families)
|
||||
Reference in New Issue
Block a user