"""Inspect/clean AI provenance metadata in non-raster containers. Formats: SVG, PDF (best-effort), DOCX, ODT, HTML, Markdown frontmatter. Stdlib-first; PDF prefers optional exiftool/c2patool when present. """ import base64 import io import posixpath import re import subprocess import urllib.parse import zipfile from dataclasses import dataclass, field from pathlib import Path from typing import Any from common import ( classify_finding_confidence, safe_arg, safe_write_bytes, safe_write_text, subprocess_preexec_fn, which, ) from image_meta import ( AI_META_HINTS, C2PA_MARKERS, inspect_isobmff, inspect_jpeg, inspect_png, inspect_webp, run_optional_tools, strip_isobmff, strip_jpeg, strip_png, strip_webp, ) from image_meta import ( detect_format as detect_image_format, ) # Frontmatter / meta keys that often carry AI provenance AI_FRONTMATTER_KEYS = frozenset( { "generator", "ai", "ai_generated", "ai-generated", "claude", "anthropic", "openai", "gemini", "synthid", "c2pa", "content_credentials", "contentcredentials", "provenance", "digital_source_type", "digitalsourcetype", "created_with", "createdwith", "model", "llm", } ) AI_META_NAME_RE = re.compile( r"generator|ai[-_ ]?generated|claude|anthropic|openai|gemini|synthid|" r"c2pa|content.?credential|provenance|digital.?source|aigc", re.I, ) SVG_DROP_TAGS = frozenset( { "{http://www.w3.org/2000/svg}metadata", "metadata", "{http://www.w3.org/1999/02/22-rdf-syntax-ns#}RDF", "{adobe:ns:meta/}xmpmeta", } ) @dataclass class ContainerInspectReport: path: str format: str has_c2pa: bool has_ai_metadata: bool findings: list[str] = field(default_factory=list) tools: dict[str, Any] = field(default_factory=dict) details: dict[str, Any] = field(default_factory=dict) notes: list[str] = field(default_factory=list) # Layer A (invisible/format Unicode) scan of the text body, populated only # for the formats clean_container() actually scrubs. Without it, inspect # reported a markdown/html file carrying invisible carriers as clean while # clean then went on to remove them. layer_a_total: int = 0 layer_a_hits: list[dict] = field(default_factory=list) def to_dict(self) -> dict: return { "path": self.path, "format": self.format, "has_c2pa": self.has_c2pa, "has_ai_metadata": self.has_ai_metadata, "findings": self.findings, "findings_confidence": [classify_finding_confidence(f) for f in self.findings], "tools": self.tools, "details": self.details, "notes": self.notes, # Same key as TextInspectReport so every caller — including the # HTTP server's `suspicious` flag — reads both report kinds alike. "suspicious_total": self.layer_a_total, "layer_a_hits": self.layer_a_hits, } def detect_container_format(path: Path, data: bytes | None = None) -> str: ext = path.suffix.lower() if ext in (".svg",): return "svg" if ext in (".pdf",): return "pdf" if ext in (".docx",): return "docx" if ext in (".xlsx",): return "xlsx" if ext in (".pptx",): return "pptx" if ext in (".odt",): return "odt" if ext in (".html", ".htm"): return "html" if ext in (".md", ".markdown", ".mdx"): return "markdown" if data is not None: if data[:4] == b"%PDF": return "pdf" if data[:100].lstrip().startswith(b"<") and b"svg" in data[:500].lower(): return "svg" if data[:2] == b"PK": # zip-based; sniff try: with zipfile.ZipFile(io.BytesIO(data)) as zf: names = set(zf.namelist()) if "word/document.xml" in names: return "docx" if "xl/workbook.xml" in names: return "xlsx" if "ppt/presentation.xml" in names: return "pptx" if "content.xml" in names and "meta.xml" in names: return "odt" except zipfile.BadZipFile: pass return "unknown" def _blob_hits(blob: bytes) -> tuple[bool, bool, list[str]]: lower = blob.lower() findings: list[str] = [] has_c2pa = False has_ai = False for n in C2PA_MARKERS: if n.lower() in lower: has_c2pa = True findings.append(f"marker:{n.decode('ascii', errors='replace')}") for n in AI_META_HINTS: if n.lower() in lower: has_ai = True label = n.decode("ascii", errors="replace") if label not in {f.split(":", 1)[-1] for f in findings}: findings.append(f"ai:{label}") return has_c2pa, has_ai or has_c2pa, findings[:30] RE_DATA_IMAGE_URI = re.compile( r"data:image\/(?P[a-zA-Z0-9\+\-\.]+)(?P;[^\s\"'\)<>]+)?,(?P[A-Za-z0-9+/=\s%]+)", re.I, ) def _inspect_embedded_data_uris(text: str) -> tuple[bool, bool, list[str]]: has_c2pa = False has_ai = False findings: list[str] = [] for m in RE_DATA_IMAGE_URI.finditer(text): mime = m.group("mime").lower() params = (m.group("params") or "").lower() payload = m.group("payload") is_b64 = "base64" in params try: if is_b64: raw_b64 = re.sub(r"\s+", "", payload) pad = len(raw_b64) % 4 if pad: raw_b64 += "=" * (4 - pad) data = base64.b64decode(raw_b64) else: data = urllib.parse.unquote_to_bytes(payload) except Exception: # noqa: S112 - malformed data URI; skip to next match continue if not data: continue fmt = detect_image_format(data) if fmt == "png": sub_c2pa, sub_ai, sub_findings = inspect_png(data) elif fmt == "jpeg": sub_c2pa, sub_ai, sub_findings = inspect_jpeg(data) elif fmt == "webp": sub_c2pa, sub_ai, sub_findings = inspect_webp(data) elif fmt in ("avif", "heic"): sub_c2pa, sub_ai, sub_findings = inspect_isobmff(data, fmt) elif "svg" in mime or data.lstrip().startswith(b"<"): sub_c2pa, sub_ai, sub_findings, _ = inspect_svg(data) else: sub_c2pa, sub_ai, sub_findings = _blob_hits(data) if sub_c2pa: has_c2pa = True if sub_ai or sub_c2pa: has_ai = True for f in sub_findings: findings.append(f"embedded data:image/{mime}: {f}") return has_c2pa, has_ai, findings def _clean_embedded_data_uris( text: str, *, strip_all_metadata: bool = True ) -> tuple[str, list[str]]: actions: list[str] = [] def _replace_uri(m: re.Match[str]) -> str: full_match = m.group(0) mime = m.group("mime") params = m.group("params") or "" payload = m.group("payload") is_b64 = "base64" in params.lower() try: if is_b64: raw_b64 = re.sub(r"\s+", "", payload) pad = len(raw_b64) % 4 if pad: raw_b64 += "=" * (4 - pad) data = base64.b64decode(raw_b64) else: data = urllib.parse.unquote_to_bytes(payload) except Exception: return full_match if not data: return full_match fmt = detect_image_format(data) sub_actions: list[str] = [] cleaned_bytes = data try: if fmt == "png": cleaned_bytes, sub_actions = strip_png(data, strip_all_text=strip_all_metadata) elif fmt == "jpeg": cleaned_bytes, sub_actions = strip_jpeg(data, strip_all_app=strip_all_metadata) elif fmt == "webp": cleaned_bytes, sub_actions = strip_webp(data, strip_all_metadata=strip_all_metadata) elif fmt in ("avif", "heic"): cleaned_bytes, sub_actions = strip_isobmff( data, fmt, strip_all_metadata=strip_all_metadata ) elif "svg" in mime.lower() or data.lstrip().startswith(b"<"): cleaned_bytes, sub_actions = clean_svg(data) except Exception: return full_match if not any("drop" in a.lower() for a in sub_actions) or cleaned_bytes == data: return full_match actions.append(f"cleaned embedded data:image/{mime} ({', '.join(sub_actions[:2])})") if is_b64: new_b64 = base64.b64encode(cleaned_bytes).decode("ascii") return f"data:image/{mime}{params},{new_b64}" else: new_payload = urllib.parse.quote_from_bytes(cleaned_bytes) return f"data:image/{mime}{params},{new_payload}" out = RE_DATA_IMAGE_URI.sub(_replace_uri, text) return out, actions # --------------------------------------------------------------------------- # Markdown frontmatter # --------------------------------------------------------------------------- _FM_RE = re.compile(r"\A---\r?\n(.*?)\r?\n---\r?\n?", re.DOTALL) def _parse_simple_yaml_keys(block: str) -> list[tuple[str, str, int]]: """Return list of (key, full_line, line_index) for top-level keys only.""" rows: list[tuple[str, str, int]] = [] for i, line in enumerate(block.splitlines()): if not line.strip() or line.strip().startswith("#"): continue if line[0] in (" ", "\t", "-"): continue # nested / list — leave alone m = re.match(r"^([A-Za-z0-9_.-]+)\s*:", line) if m: rows.append((m.group(1), line, i)) return rows def inspect_markdown(text: str) -> tuple[bool, bool, list[str], dict]: findings: list[str] = [] has_ai = False has_fm = False keys = [] m = _FM_RE.match(text) if m: has_fm = True block = m.group(1) for key, _line, _i in _parse_simple_yaml_keys(block): keys.append(key) if key.lower() in AI_FRONTMATTER_KEYS or AI_META_NAME_RE.search(key): has_ai = True findings.append(f"frontmatter key: {key}") # also check value val = _line.split(":", 1)[1] if ":" in _line else "" if AI_META_NAME_RE.search(val): has_ai = True findings.append(f"frontmatter value hit on {key}") uri_c2pa, uri_ai, uri_findings = _inspect_embedded_data_uris(text) if uri_c2pa: has_ai = True if uri_ai: has_ai = True findings.extend(uri_findings) c2pa = uri_c2pa or any("c2pa" in f.lower() or "content" in f.lower() for f in findings) return c2pa, has_ai, findings, {"has_frontmatter": has_fm, "keys": keys} def clean_markdown(text: str) -> tuple[str, list[str]]: actions: list[str] = [] m = _FM_RE.match(text) if m: block = m.group(1) body = text[m.end() :] kept: list[str] = [] dropping = False # inside the nested block of a dropped top-level key for line in block.splitlines(): stripped = line.strip() # Blank lines and comments belong to whichever block we are inside. if not stripped or stripped.startswith("#"): if not dropping: kept.append(line) continue # Continuation lines (nested mappings, list items) follow their parent. if line[0] in (" ", "\t", "-"): if not dropping: kept.append(line) continue km = re.match(r"^([A-Za-z0-9_.-]+)\s*:", line) if not km: dropping = False kept.append(line) continue key = km.group(1) val = line.split(":", 1)[1] if ":" in line else "" if key.lower() in AI_FRONTMATTER_KEYS or AI_META_NAME_RE.search(key): actions.append(f"drop frontmatter key: {key}") dropping = True continue if AI_META_NAME_RE.search(val): actions.append(f"drop frontmatter key (value hit): {key}") dropping = True continue dropping = False kept.append(line) new_block = "\n".join(kept).strip("\n") if new_block: out = f"---\n{new_block}\n---\n{body}" else: out = body.lstrip("\n") actions.append("removed empty frontmatter block") else: out = text out, uri_actions = _clean_embedded_data_uris(out) if uri_actions: actions.extend(uri_actions) if not actions: actions.append("no AI frontmatter keys or embedded data URIs removed") return out, actions # --------------------------------------------------------------------------- # HTML # --------------------------------------------------------------------------- _META_TAG_RE = re.compile( r"]*>", re.I, ) _META_ATTR_RE = re.compile( r"""(name|property|content|generator)\s*=\s*["']([^"']*)["']""", re.I, ) # Known AI vendor names for the "generator" meta tag. A plain CMS generator # (WordPress, Elementor) is CMS provenance, not AI-generator metadata. _GENERATOR_AI_RE = re.compile( r"claude|anthropic|openai|chatgpt|gemini|synthid|copilot|midjourney|dall.?e|stable.?diffusion", re.I, ) def _meta_attrs(tag: str) -> dict[str, str]: return {name.lower(): value for name, value in _META_ATTR_RE.findall(tag)} def _is_cms_generator_meta(tag: str) -> bool: """Return True for a generator meta tag that is CMS provenance, not AI.""" attrs = _meta_attrs(tag) name_or_prop = ( attrs.get("name") or attrs.get("property") or attrs.get("generator") or "" ).lower() if name_or_prop != "generator": return False return not (_GENERATOR_AI_RE.search(attrs.get("content", "")) or _GENERATOR_AI_RE.search(tag)) _JSONLD_RE = re.compile( r"]*type\s*=\s*[\"']application/ld\+json[\"'][^>]*>.*?", re.I | re.DOTALL, ) def inspect_html(text: str) -> tuple[bool, bool, list[str], dict]: findings: list[str] = [] has_ai = False has_c2pa = False for tag in _META_TAG_RE.findall(text): if re.search(r"c2pa|content.?credential", tag, re.I): has_c2pa = True if _is_cms_generator_meta(tag): findings.append(f"info: cms generator: {tag[:120]}") continue if AI_META_NAME_RE.search(tag) or any( h.decode("ascii", "ignore").lower() in tag.lower() for h in AI_META_HINTS[:12] ): has_ai = True findings.append(f"meta: {tag[:120]}") for m in _JSONLD_RE.finditer(text): blob = m.group(0) if AI_META_NAME_RE.search(blob) or re.search( r"DigitalSourceType|trainedAlgorithmicMedia|SoftwareAgent", blob, re.I ): has_ai = True findings.append("json-ld provenance-like block") if re.search(r"c2pa|contentcredential", blob, re.I): has_c2pa = True # data-ai* attributes for m in re.finditer(r"\bdata-ai[\w-]*\s*=\s*[\"'][^\"']*[\"']", text, re.I): has_ai = True findings.append(f"attr: {m.group(0)[:80]}") uri_c2pa, uri_ai, uri_findings = _inspect_embedded_data_uris(text) if uri_c2pa: has_c2pa = True if uri_ai: has_ai = True findings.extend(uri_findings) return has_c2pa, has_ai, findings, {} def clean_html(text: str) -> tuple[str, list[str]]: actions: list[str] = [] def _meta_sub(m: re.Match[str]) -> str: tag = m.group(0) if _is_cms_generator_meta(tag): return tag if AI_META_NAME_RE.search(tag) or re.search( r"generator|claude|anthropic|openai|gemini|synthid|c2pa|aigc", tag, re.I ): actions.append(f"drop meta: {tag[:80]}") return "" return tag out = _META_TAG_RE.sub(_meta_sub, text) def _jsonld_sub(m: re.Match[str]) -> str: blob = m.group(0) if AI_META_NAME_RE.search(blob) or re.search( r"DigitalSourceType|trainedAlgorithmicMedia|SoftwareAgent", blob, re.I ): actions.append("drop json-ld provenance-like script") return "" return blob out = _JSONLD_RE.sub(_jsonld_sub, out) out2, n = re.subn(r"\sdata-ai[\w-]*\s*=\s*[\"'][^\"']*[\"']", "", out, flags=re.I) if n: actions.append(f"drop data-ai* attributes x{n}") out = out2 out, uri_actions = _clean_embedded_data_uris(out) if uri_actions: actions.extend(uri_actions) if not actions: actions.append("no HTML AI meta removed") return out, actions # --------------------------------------------------------------------------- # SVG # --------------------------------------------------------------------------- def inspect_svg(data: bytes) -> tuple[bool, bool, list[str], dict]: findings: list[str] = [] has_c2pa, has_ai, hits = _blob_hits(data) findings.extend(hits) try: text = data.decode("utf-8", errors="replace") if re.search(r"]", text, re.I): findings.append("svg present") has_ai = True # often XMP; treat as inspect signal if re.search(r"xmpmeta|rdf:RDF|contentcredentials", text, re.I): has_ai = True findings.append("XMP/RDF-like content in SVG") if re.search(r"c2pa|jumbf", text, re.I): has_c2pa = True uri_c2pa, uri_ai, uri_findings = _inspect_embedded_data_uris(text) if uri_c2pa: has_c2pa = True if uri_ai: has_ai = True findings.extend(uri_findings) except Exception as e: findings.append(f"svg decode note: {e}") return has_c2pa, has_ai or has_c2pa, findings, {} def clean_svg(data: bytes) -> tuple[bytes, list[str]]: actions: list[str] = [] text = data.decode("utf-8", errors="surrogateescape") # Drop metadata blocks new, n = re.subn( r"]*>.*?", "", text, flags=re.I | re.DOTALL, ) if n: actions.append(f"drop x{n}") text = new # Drop adobe xmp packets new, n = re.subn( r"]*>.*?", "", text, flags=re.I | re.DOTALL, ) if n: actions.append(f"drop xmpmeta x{n}") text = new # Drop comments that look like provenance def _cmt(m: re.Match[str]) -> str: body = m.group(0) if AI_META_NAME_RE.search(body): actions.append("drop SVG comment with AI markers") return "" return body text = re.sub(r"", _cmt, text, flags=re.DOTALL) # Clean embedded data URIs text, uri_actions = _clean_embedded_data_uris(text) if uri_actions: actions.extend(uri_actions) if not actions: # still strip generator attribute on root if present new, n = re.subn( r'\s(inkscape:version|sodipodi:docname|generator)\s*=\s*"[^"]*"', "", text, flags=re.I, ) if n: actions.append(f"drop generator-like attrs x{n}") text = new if not actions: actions.append("no SVG metadata removed") return text.encode("utf-8", errors="surrogateescape"), actions # --------------------------------------------------------------------------- # DOCX / ODT (zip + XML) # --------------------------------------------------------------------------- DOCX_META_PARTS = ( "docProps/core.xml", "docProps/app.xml", "docProps/custom.xml", ) DOCX_CUSTOM_PREFIXES = ( "customXml/", "docProps/", ) # Provenance fields in docProps/core.xml and docProps/app.xml that always come # out empty. dc:title is deliberately not listed: it is the document's own # heading, not provenance. DOCX_SCRUB_FIELDS = ( ("dc:creator", "dc:creator"), ("cp:lastModifiedBy", "cp:lastModifiedBy"), ("dc:description", "dc:description"), ("cp:keywords", "cp:keywords"), ("dc:subject", "dc:subject"), ("cp:category", "cp:category"), ("Application", "Application"), ("AppVersion", "AppVersion"), ("Company", "Company"), ("Manager", "Manager"), ) def _zip_namelist(data: bytes) -> list[str]: with zipfile.ZipFile(io.BytesIO(data)) as zf: return zf.namelist() MAX_ZIP_DECOMPRESSED_BYTES = 128 * 1024 * 1024 def _check_zip_budget(info: zipfile.ZipInfo, budget: list[int]) -> None: """Reject zip bombs before decompression (ZipInfo.file_size is stored).""" budget[0] += info.file_size if budget[0] > MAX_ZIP_DECOMPRESSED_BYTES: raise ValueError( "zip decompressed size exceeds cap " f"({MAX_ZIP_DECOMPRESSED_BYTES} bytes); refusing to process" ) def _is_docx_meta_part(name: str) -> bool: """Return True for DOCX/XLSX/PPTX parts that carry provenance, not visible content.""" return name.startswith(("docProps/", "customXml/")) def _inspect_ooxml_zip(data: bytes, fmt: str) -> tuple[bool, bool, list[str], dict]: findings: list[str] = [] has_c2pa = False has_ai = False parts: list[str] = [] budget = [0] try: with zipfile.ZipFile(io.BytesIO(data)) as zf: parts = zf.namelist() for info in zf.infolist(): _check_zip_budget(info, budget) name = info.filename # Check media parts for C2PA/AI metadata if re.search( r"^(?:word|xl|ppt)/media/.+\.(png|jpe?g|webp|avif|heic|svg)$", name, re.I ): raw = zf.read(name) img_fmt = detect_image_format(raw) sub_c2pa, sub_ai, sub_findings = False, False, [] if img_fmt == "png": sub_c2pa, sub_ai, sub_findings = inspect_png(raw) elif img_fmt == "jpeg": sub_c2pa, sub_ai, sub_findings = inspect_jpeg(raw) elif img_fmt == "webp": sub_c2pa, sub_ai, sub_findings = inspect_webp(raw) elif img_fmt in ("avif", "heic"): sub_c2pa, sub_ai, sub_findings = inspect_isobmff(raw, img_fmt) elif name.lower().endswith(".svg") or raw.lstrip().startswith(b"<"): sub_c2pa, sub_ai, sub_findings, _ = inspect_svg(raw) if sub_c2pa: has_c2pa = True if sub_ai or sub_c2pa: has_ai = True for sf in sub_findings: findings.append(f"{name}: {sf}") continue # Only metadata/provenance parts carry AI markers. The visible # body (word/*.xml, xl/*.xml, ppt/*.xml) may legitimately mention # vendor names such as "Claude" without being AI-generated metadata. if not _is_docx_meta_part(name): continue raw = zf.read(name) c2, ai, hits = _blob_hits(raw) if c2 or ai: if c2: has_c2pa = True if ai: has_ai = True findings.append(f"{name}: {', '.join(hits[:6])}") # always flag customXml presence lightly custom = [n for n in parts if n.startswith("customXml/")] if custom: findings.append(f"customXml parts: {len(custom)}") except zipfile.BadZipFile: return False, False, [f"not a valid {fmt.upper()} zip"], {} return has_c2pa, has_ai or has_c2pa, findings, {"parts": len(parts)} def inspect_docx(data: bytes) -> tuple[bool, bool, list[str], dict]: return _inspect_ooxml_zip(data, "docx") def inspect_xlsx(data: bytes) -> tuple[bool, bool, list[str], dict]: return _inspect_ooxml_zip(data, "xlsx") def inspect_pptx(data: bytes) -> tuple[bool, bool, list[str], dict]: return _inspect_ooxml_zip(data, "pptx") def _scrub_docx_text(xml_text: str) -> tuple[str, int, int]: """Run Layer A over the ```` text runs of a DOCX part. Only ``w:t`` nodes are touched: field codes (``w:instrText``), run/paragraph properties and the surrounding XML are left byte-identical. If leading or trailing whitespace survives the clean, the node keeps ``xml:space="preserve"`` so Word does not trim it. """ from text_unicode import clean_text # local import to avoid cycles removed = 0 replaced = 0 def _repl(m: re.Match[str]) -> str: nonlocal removed, replaced open_tag, inner, close_tag = m.group(1), m.group(2), m.group(3) new_inner, stats = clean_text(inner) if not (stats["removed_count"] or stats["replaced_count"]): return m.group(0) removed += stats["removed_count"] replaced += stats["replaced_count"] if (new_inner[:1].isspace() or new_inner[-1:].isspace()) and "xml:space" not in open_tag: open_tag = open_tag[:-1] + ' xml:space="preserve">' return open_tag + new_inner + close_tag new = re.sub(r"(]*>)(.*?)()", _repl, xml_text, flags=re.S) return new, removed, replaced def _scrub_xlsx_text(xml_text: str) -> tuple[str, int, int]: """Run Layer A over the ```` text elements of an XLSX part.""" from text_unicode import clean_text # local import to avoid cycles removed = 0 replaced = 0 def _repl(m: re.Match[str]) -> str: nonlocal removed, replaced open_tag, inner, close_tag = m.group(1), m.group(2), m.group(3) new_inner, stats = clean_text(inner) if not (stats["removed_count"] or stats["replaced_count"]): return m.group(0) removed += stats["removed_count"] replaced += stats["replaced_count"] if (new_inner[:1].isspace() or new_inner[-1:].isspace()) and "xml:space" not in open_tag: open_tag = open_tag[:-1] + ' xml:space="preserve">' return open_tag + new_inner + close_tag new = re.sub(r"(]*>)(.*?)()", _repl, xml_text, flags=re.S) return new, removed, replaced def _scrub_pptx_text(xml_text: str) -> tuple[str, int, int]: """Run Layer A over the ```` text elements of a PPTX part.""" from text_unicode import clean_text # local import to avoid cycles removed = 0 replaced = 0 def _repl(m: re.Match[str]) -> str: nonlocal removed, replaced open_tag, inner, close_tag = m.group(1), m.group(2), m.group(3) new_inner, stats = clean_text(inner) if not (stats["removed_count"] or stats["replaced_count"]): return m.group(0) removed += stats["removed_count"] replaced += stats["replaced_count"] return open_tag + new_inner + close_tag new = re.sub(r"(]*>)(.*?)()", _repl, xml_text, flags=re.S) return new, removed, replaced def _scrub_odt_text(xml_text: str) -> tuple[str, int, int]: """Run Layer A over ODF paragraph text (``text:p`` content, incl. spans). ``text:span``/``text:tab``/``text:s`` children live inside the paragraph, so cleaning the paragraph content covers the visible text. The markup itself is untouched. """ from text_unicode import clean_text # local import to avoid cycles removed = 0 replaced = 0 def _repl(m: re.Match[str]) -> str: nonlocal removed, replaced open_tag, inner, close_tag = m.group(1), m.group(2), m.group(3) new_inner, stats = clean_text(inner) if not (stats["removed_count"] or stats["replaced_count"]): return m.group(0) removed += stats["removed_count"] replaced += stats["replaced_count"] return open_tag + new_inner + close_tag new = re.sub(r"(]*>)(.*?)()", _repl, xml_text, flags=re.S) return new, removed, replaced def _prune_dangling_relationships( rels_name: str, raw: bytes, kept_names: set[str] ) -> tuple[bytes, int]: """Drop entries whose internal target part no longer exists. Removing a part (e.g. a customXml tree) must also remove the relationships that point at it, or the package is malformed: python-docx refuses to open it and Word offers to repair it. External relationships (``TargetMode``) and the package root (``Target="/"``) are left alone. ``rels_name`` is the archive member like ``word/_rels/document.xml.rels``; ``kept_names`` is the set of archive members that survive cleaning. """ base = posixpath.dirname(posixpath.dirname(rels_name)) text = raw.decode("utf-8", errors="replace") dropped = [0] def _target_attr(tag: str) -> str: m = re.search(r'\bTarget\s*=\s*"([^"]*)"', tag, re.I) return m.group(1) if m else "" def _drop(m: re.Match[str]) -> str: tag = m.group(0) if re.search(r"\bTargetMode\s*=", tag, re.I): return tag # external (http / mailto / ...) — never pruned target = _target_attr(tag) if target.startswith("/"): resolved = posixpath.normpath(target.lstrip("/")) else: resolved = posixpath.normpath(posixpath.join(base, target)) if resolved in ("", "."): return tag # points at the package root if resolved in kept_names: return tag dropped[0] += 1 return "" new = re.sub(r"]*/>", _drop, text, flags=re.I) return new.encode("utf-8"), dropped[0] def _scrub_ooxml_zip( data: bytes, fmt: str, *, also_layer_a_text: bool = True ) -> tuple[bytes, list[str]]: actions: list[str] = [] budget = [0] layer_removed = 0 layer_replaced = 0 kept: list[tuple[zipfile.ZipInfo, bytes]] = [] with zipfile.ZipFile(io.BytesIO(data)) as zin: for info in zin.infolist(): name = info.filename _check_zip_budget(info, budget) raw = zin.read(name) # 1. Clean embedded media (PNG, JPEG, WebP, AVIF, HEIC, SVG) if re.search(r"^(?:word|xl|ppt)/media/.+\.(png|jpe?g|webp|avif|heic|svg)$", name, re.I): img_fmt = detect_image_format(raw) sub_actions: list[str] = [] cleaned_bytes = raw try: if img_fmt == "png": cleaned_bytes, sub_actions = strip_png(raw, strip_all_text=True) elif img_fmt == "jpeg": cleaned_bytes, sub_actions = strip_jpeg(raw, strip_all_app=True) elif img_fmt == "webp": cleaned_bytes, sub_actions = strip_webp(raw, strip_all_metadata=True) elif img_fmt in ("avif", "heic"): cleaned_bytes, sub_actions = strip_isobmff( raw, img_fmt, strip_all_metadata=True ) elif name.lower().endswith(".svg") or raw.lstrip().startswith(b"<"): cleaned_bytes, sub_actions = clean_svg(raw) except Exception: # noqa: S110 - malformed embedded media; keep original part pass if any("drop" in a.lower() for a in sub_actions) and cleaned_bytes != raw: actions.append(f"clean embedded media in {name} ({', '.join(sub_actions[:2])})") raw = cleaned_bytes kept.append((info, raw)) continue # 2. Drop customXml trees if name.startswith("customXml/"): actions.append(f"drop part {name}") continue # 3. docProps/ provenance if name in DOCX_META_PARTS or name.startswith("docProps/"): if name.endswith("custom.xml"): actions.append(f"drop part {name}") continue text = raw.decode("utf-8", errors="replace") new = text for tag, label in DOCX_SCRUB_FIELDS: pat = rf"(<{tag}\b[^>]*>)(.*?)()" def _empty(m: re.Match[str], _label=label, _name=name) -> str: if m.group(2): actions.append(f"scrub {_name} field {_label}") return m.group(1) + m.group(3) new = re.sub(pat, _empty, new, flags=re.I | re.DOTALL) raw = new.encode("utf-8") # 4. [Content_Types].xml overrides if name == "[Content_Types].xml": text = raw.decode("utf-8", errors="replace") new, n = re.subn( r']*PartName="/customXml/[^"]*"[^>]*/>', "", text, ) if n: actions.append(f"drop Content_Types customXml overrides x{n}") raw = new.encode("utf-8") new, n = re.subn( r']*PartName="/docProps/custom\.xml"[^>]*/>', "", raw.decode("utf-8", errors="replace"), ) if n: actions.append(f"drop Content_Types custom.xml override x{n}") raw = new.encode("utf-8") # 5. Layer A text runs if also_layer_a_text and name.endswith(".xml"): if fmt == "docx" and name.startswith("word/"): text = raw.decode("utf-8", errors="replace") new, r, rp = _scrub_docx_text(text) if r or rp: layer_removed += r layer_replaced += rp raw = new.encode("utf-8") elif fmt == "xlsx" and name.startswith("xl/"): text = raw.decode("utf-8", errors="replace") new, r, rp = _scrub_xlsx_text(text) if r or rp: layer_removed += r layer_replaced += rp raw = new.encode("utf-8") elif fmt == "pptx" and name.startswith("ppt/"): text = raw.decode("utf-8", errors="replace") new, r, rp = _scrub_pptx_text(text) if r or rp: layer_removed += r layer_replaced += rp raw = new.encode("utf-8") kept.append((info, raw)) kept_names = {info.filename for info, _ in kept} final: list[tuple[zipfile.ZipInfo, bytes]] = [] for info, raw in kept: part_raw = raw if info.filename.endswith(".rels"): part_raw, n = _prune_dangling_relationships(info.filename, raw, kept_names) if n: actions.append(f"prune dangling relationships x{n} in {info.filename}") final.append((info, part_raw)) out_buf = io.BytesIO() with zipfile.ZipFile(out_buf, "w", compression=zipfile.ZIP_DEFLATED) as zout: for info, raw in final: zout.writestr(info, raw) if layer_removed or layer_replaced: actions.append(f"layer A text: removed={layer_removed} replaced={layer_replaced}") if not actions: actions.append(f"no {fmt.upper()} metadata parts removed") return out_buf.getvalue(), actions def clean_docx(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]: return _scrub_ooxml_zip(data, "docx", also_layer_a_text=also_layer_a_text) def clean_xlsx(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]: return _scrub_ooxml_zip(data, "xlsx", also_layer_a_text=also_layer_a_text) def clean_pptx(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]: return _scrub_ooxml_zip(data, "pptx", also_layer_a_text=also_layer_a_text) def inspect_odt(data: bytes) -> tuple[bool, bool, list[str], dict]: findings: list[str] = [] has_c2pa = False has_ai = False budget = [0] try: with zipfile.ZipFile(io.BytesIO(data)) as zf: for info in zf.infolist(): _check_zip_budget(info, budget) raw = zf.read(info.filename) c2, ai, hits = _blob_hits(raw) if c2 or ai: if c2: has_c2pa = True if ai: has_ai = True findings.append(f"{info.filename}: {', '.join(hits[:6])}") if "meta.xml" in zf.namelist(): meta = zf.read("meta.xml").decode("utf-8", errors="replace") if re.search(r"generator|claude|openai|anthropic|gemini", meta, re.I): has_ai = True findings.append("meta.xml generator-like fields") except zipfile.BadZipFile: return False, False, ["not a valid ODT zip"], {} return has_c2pa, has_ai or has_c2pa, findings, {} def clean_odt(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]: actions: list[str] = [] out_buf = io.BytesIO() budget = [0] layer_removed = 0 layer_replaced = 0 with ( zipfile.ZipFile(io.BytesIO(data)) as zin, zipfile.ZipFile(out_buf, "w", compression=zipfile.ZIP_DEFLATED) as zout, ): for info in zin.infolist(): name = info.filename _check_zip_budget(info, budget) raw = zin.read(name) if name == "meta.xml": text = raw.decode("utf-8", errors="replace") new, n = re.subn( r"]*>.*?", "", text, flags=re.I | re.DOTALL, ) if n: actions.append("drop meta:generator") text = new # scrub creator-like if AI def _creator(m: re.Match[str]) -> str: if AI_META_NAME_RE.search(m.group(0)): actions.append("scrub creator-like meta") return "" return m.group(0) text = re.sub( r"]*>.*?", _creator, text, flags=re.I | re.DOTALL, ) raw = text.encode("utf-8") else: c2, ai, _ = _blob_hits(raw) if (c2 or ai) and name not in ( "content.xml", "styles.xml", "mimetype", "META-INF/manifest.xml", ): actions.append(f"drop part {name} (AI/C2PA markers)") continue # Layer A over the visible paragraph text of the body part. if also_layer_a_text and name == "content.xml": text = raw.decode("utf-8", errors="replace") new, r, rp = _scrub_odt_text(text) if r or rp: layer_removed += r layer_replaced += rp raw = new.encode("utf-8") zout.writestr(info, raw) if layer_removed or layer_replaced: actions.append(f"layer A text: removed={layer_removed} replaced={layer_replaced}") if not actions: actions.append("no ODT metadata removed") return out_buf.getvalue(), actions # --------------------------------------------------------------------------- # PDF # --------------------------------------------------------------------------- _XMP_PACKET_RE = re.compile( rb"<\?xpacket begin.*?<\?xpacket end[^?]*\?>", re.I | re.DOTALL, ) def _pdf_structured_blob(data: bytes) -> bytes: """Return PDF bytes with stream payloads removed, plus XMP packets. Stream payloads are often compressed binary where an AI-marker byte sequence (e.g. "AIGC") can occur by chance. Scanning only dictionaries and XMP packets avoids treating those collisions as metadata findings. """ no_streams = re.sub( rb"stream\r?\n.*?endstream", b"stream endstream", data, flags=re.DOTALL, ) xmp = b"\n".join(_XMP_PACKET_RE.findall(data)) return no_streams + b"\n" + xmp def inspect_pdf(path: Path, data: bytes) -> tuple[bool, bool, list[str], dict]: findings: list[str] = [] has_c2pa, has_ai, hits = _blob_hits(_pdf_structured_blob(data)) findings.extend(f"pdf-structured:{h}" for h in hits) # XMP packet scan xmp_blob = b"\n".join(_XMP_PACKET_RE.findall(data)) if xmp_blob: findings.append("XMP packet present") has_ai = has_ai or bool( re.search( rb"digitalSourceType|trainedAlgorithmicMedia|SoftwareAgent|c2pa", xmp_blob, re.I, ) ) tools = run_optional_tools(path) ct = tools.get("c2patool") or {} if ct.get("has_manifest"): has_c2pa = True findings.append("c2patool reports C2PA-related manifest") return has_c2pa, has_ai or has_c2pa, findings, {"tools": tools} def _pdf_structural_rewrite(dest: Path, actions: list[str]) -> bool: """Rebuild a PDF so unreferenced objects are dropped. exiftool's PDF edits are incremental, so freed metadata objects survive in the byte stream. qpdf re-serializes from the object graph, which is what actually removes them. No-op (with a warning) when qpdf is absent. """ qpdf = which("qpdf") if not qpdf: actions.append( "warning: exiftool PDF edits are incremental — the original metadata " "bytes remain recoverable; install qpdf for a structural rewrite" ) return False tmp = dest.with_name(dest.name + ".qpdf-tmp") try: r = subprocess.run( [qpdf, "--linearize", "--", safe_arg(str(dest)), safe_arg(str(tmp))], capture_output=True, text=True, timeout=120, check=False, preexec_fn=subprocess_preexec_fn, ) except Exception as e: tmp.unlink(missing_ok=True) actions.append(f"qpdf rewrite failed: {e}; metadata bytes may remain recoverable") return False # qpdf exit codes: 0 = clean, 3 = succeeded with warnings (output written). if r.returncode in (0, 3) and tmp.is_file() and tmp.stat().st_size > 0: tmp.replace(dest) actions.append(f"qpdf --linearize structural rewrite (rc={r.returncode})") return True tmp.unlink(missing_ok=True) actions.append( f"qpdf rewrite skipped (rc={r.returncode}); metadata bytes may remain recoverable" ) return False def clean_pdf(path: Path, dest: Path) -> tuple[list[str], dict]: """Best-effort PDF clean. Prefers exiftool; falls back to XMP strip warning.""" actions: list[str] = [] data = path.read_bytes() dest.parent.mkdir(parents=True, exist_ok=True) exiftool = which("exiftool") if exiftool: safe_write_bytes(dest, data) try: r = subprocess.run( [ exiftool, "-all=", "-overwrite_original", safe_arg(str(dest)), ], capture_output=True, text=True, timeout=60, check=False, preexec_fn=subprocess_preexec_fn, ) actions.append(f"exiftool -all= (rc={r.returncode})") except Exception as e: actions.append(f"exiftool failed: {e}") # exiftool writes PDFs *incrementally*: it appends a # %BeginExifToolUpdate block that frees the Info object and drops # /Info from the trailer, but the original metadata bytes stay in the # file verbatim and are trivially recoverable (exiftool itself can # revert them with -PDF-update:all=). A structural rewrite is what # actually drops the now-unreferenced objects. rewritten = _pdf_structural_rewrite(dest, actions) c2patool = which("c2patool") # c2patool does not always strip; leave note if c2patool: actions.append("c2patool available for inspect; strip via exiftool/re-export") return actions, {"mode": "exiftool", "structural_rewrite": rewritten} # Degraded: strip obvious XMP packets between ", b"", text, flags=re.I | re.DOTALL, ) if n: actions.append(f"stripped XMP xpacket x{n} (degraded; may leave offsets broken)") # PDF structural risk: document degraded mode clearly safe_write_bytes(dest, new) actions.append("warning: pure-stdlib PDF strip is best-effort; prefer exiftool") return actions, {"mode": "stdlib-xmp", "degraded": True} safe_write_bytes(dest, data) actions.append( "no PDF cleaner available (install exiftool for reliable metadata strip); copied as-is" ) return actions, {"mode": "copy", "degraded": True} # --------------------------------------------------------------------------- # Unified API # --------------------------------------------------------------------------- def inspect_container(path: Path) -> ContainerInspectReport: data = path.read_bytes() fmt = detect_container_format(path, data) tools: dict[str, Any] = {} details: dict[str, Any] = {} if fmt == "svg": has_c2pa, has_ai, findings, details = inspect_svg(data) elif fmt == "pdf": has_c2pa, has_ai, findings, details = inspect_pdf(path, data) tools = details.pop("tools", {}) elif fmt == "docx": has_c2pa, has_ai, findings, details = inspect_docx(data) elif fmt == "xlsx": has_c2pa, has_ai, findings, details = inspect_xlsx(data) elif fmt == "pptx": has_c2pa, has_ai, findings, details = inspect_pptx(data) elif fmt == "odt": has_c2pa, has_ai, findings, details = inspect_odt(data) elif fmt == "html": # surrogateescape, not replace: clean_container() decodes the same way, # and U+FFFD substitutions would make the two disagree on the counts. body = data.decode("utf-8", errors="surrogateescape") has_c2pa, has_ai, findings, details = inspect_html(body) elif fmt == "markdown": body = data.decode("utf-8", errors="surrogateescape") has_c2pa, has_ai, findings, details = inspect_markdown(body) else: has_c2pa, has_ai, findings = False, False, [f"unsupported container: {fmt}"] details = {"unsupported": True} # Layer A body scan for exactly the formats clean_container() scrubs, so # inspect predicts clean rather than contradicting it. layer_a_total = 0 layer_a_hits: list[dict] = [] if fmt in ("markdown", "html"): from text_unicode import inspect_text # local import to avoid cycles ta = inspect_text(body).to_dict() layer_a_total = ta["suspicious_total"] layer_a_hits = ta["hits"] for h in layer_a_hits: findings.append(f"layer-a: {h['codepoint']} {h['label']} x{h['count']} ({h['kind']})") notes: list[str] = [] if fmt == "pdf": notes.append( "PDF inspection is best-effort; exiftool/c2patool give more reliable metadata detection" ) elif fmt in ("docx", "xlsx", "pptx"): notes.append(f"{fmt.upper()}: metadata/provenance and embedded media are scanned") if "unsupported" in details: notes.append(f"format not fully inspected: {fmt}") if layer_a_total: notes.append( f"layer A: {layer_a_total} invisible/format codepoint(s) in body text; " "clean removes these" ) if fmt in ("svg", "pdf", "docx", "xlsx", "pptx") and not tools: tools = run_optional_tools(path) return ContainerInspectReport( path=str(path), format=fmt, has_c2pa=has_c2pa, has_ai_metadata=has_ai, findings=findings, tools=tools, details=details, notes=notes, layer_a_total=layer_a_total, layer_a_hits=layer_a_hits, ) def clean_container( path: Path, dest: Path, fmt: str | None = None, *, also_layer_a_text: bool = True, ) -> dict[str, Any]: """Clean container metadata; optionally Layer-A scrub text bodies for md/html. ``fmt`` pins the container format when the caller already knows it. This matters for ``--in-place`` flows where *path* is a ``.bak`` copy whose suffix would otherwise make markdown/HTML (which have no magic bytes) classify as ``unknown``. """ from text_unicode import clean_text # local import to avoid cycles data = path.read_bytes() fmt = fmt or detect_container_format(path, data) actions: list[str] = [] dest.parent.mkdir(parents=True, exist_ok=True) meta: dict[str, Any] = {"format": fmt} if fmt == "svg": cleaned, actions = clean_svg(data) safe_write_bytes(dest, cleaned) elif fmt == "pdf": actions, meta_extra = clean_pdf(path, dest) meta.update(meta_extra) elif fmt == "docx": cleaned, actions = clean_docx(data, also_layer_a_text=also_layer_a_text) safe_write_bytes(dest, cleaned) elif fmt == "xlsx": cleaned, actions = clean_xlsx(data, also_layer_a_text=also_layer_a_text) safe_write_bytes(dest, cleaned) elif fmt == "pptx": cleaned, actions = clean_pptx(data, also_layer_a_text=also_layer_a_text) safe_write_bytes(dest, cleaned) elif fmt == "odt": cleaned, actions = clean_odt(data, also_layer_a_text=also_layer_a_text) safe_write_bytes(dest, cleaned) elif fmt == "html": text = data.decode("utf-8", errors="surrogateescape") text, actions = clean_html(text) if also_layer_a_text: text2, stats = clean_text(text) if stats["removed_count"] or stats["replaced_count"]: actions.append( f"layer A text: removed={stats['removed_count']} replaced={stats['replaced_count']}" ) text = text2 safe_write_text(dest, text) elif fmt == "markdown": text = data.decode("utf-8", errors="surrogateescape") text, actions = clean_markdown(text) if also_layer_a_text: text2, stats = clean_text(text) if stats["removed_count"] or stats["replaced_count"]: actions.append( f"layer A text: removed={stats['removed_count']} replaced={stats['replaced_count']}" ) text = text2 safe_write_text(dest, text) else: raise ValueError(f"unsupported container format: {fmt}") after = inspect_container(dest) return { "input": str(path), "output": str(dest), "format": fmt, "actions": actions, "bytes_in": len(data), "bytes_out": dest.stat().st_size, "still_has_c2pa": after.has_c2pa, "still_has_ai_metadata": after.has_ai_metadata, "post_findings": after.findings, "meta": meta, }