Files

1428 lines
51 KiB
Python

"""Inspect/clean AI provenance metadata in non-raster containers.
Formats: SVG, PDF (best-effort), DOCX, ODT, HTML, Markdown frontmatter.
Stdlib-first; PDF prefers optional exiftool/c2patool when present.
"""
import base64
import io
import posixpath
import re
import subprocess
import urllib.parse
import zipfile
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any
from common import (
classify_finding_confidence,
safe_arg,
safe_write_bytes,
safe_write_text,
subprocess_preexec_fn,
which,
)
from image_meta import (
AI_META_HINTS,
C2PA_MARKERS,
inspect_isobmff,
inspect_jpeg,
inspect_png,
inspect_webp,
run_optional_tools,
strip_isobmff,
strip_jpeg,
strip_png,
strip_webp,
)
from image_meta import (
detect_format as detect_image_format,
)
# Frontmatter / meta keys that often carry AI provenance
AI_FRONTMATTER_KEYS = frozenset(
{
"generator",
"ai",
"ai_generated",
"ai-generated",
"claude",
"anthropic",
"openai",
"gemini",
"synthid",
"c2pa",
"content_credentials",
"contentcredentials",
"provenance",
"digital_source_type",
"digitalsourcetype",
"created_with",
"createdwith",
"model",
"llm",
}
)
AI_META_NAME_RE = re.compile(
r"generator|ai[-_ ]?generated|claude|anthropic|openai|gemini|synthid|"
r"c2pa|content.?credential|provenance|digital.?source|aigc",
re.I,
)
SVG_DROP_TAGS = frozenset(
{
"{http://www.w3.org/2000/svg}metadata",
"metadata",
"{http://www.w3.org/1999/02/22-rdf-syntax-ns#}RDF",
"{adobe:ns:meta/}xmpmeta",
}
)
@dataclass
class ContainerInspectReport:
path: str
format: str
has_c2pa: bool
has_ai_metadata: bool
findings: list[str] = field(default_factory=list)
tools: dict[str, Any] = field(default_factory=dict)
details: dict[str, Any] = field(default_factory=dict)
notes: list[str] = field(default_factory=list)
# Layer A (invisible/format Unicode) scan of the text body, populated only
# for the formats clean_container() actually scrubs. Without it, inspect
# reported a markdown/html file carrying invisible carriers as clean while
# clean then went on to remove them.
layer_a_total: int = 0
layer_a_hits: list[dict] = field(default_factory=list)
def to_dict(self) -> dict:
return {
"path": self.path,
"format": self.format,
"has_c2pa": self.has_c2pa,
"has_ai_metadata": self.has_ai_metadata,
"findings": self.findings,
"findings_confidence": [classify_finding_confidence(f) for f in self.findings],
"tools": self.tools,
"details": self.details,
"notes": self.notes,
# Same key as TextInspectReport so every caller — including the
# HTTP server's `suspicious` flag — reads both report kinds alike.
"suspicious_total": self.layer_a_total,
"layer_a_hits": self.layer_a_hits,
}
def detect_container_format(path: Path, data: bytes | None = None) -> str:
ext = path.suffix.lower()
if ext in (".svg",):
return "svg"
if ext in (".pdf",):
return "pdf"
if ext in (".docx",):
return "docx"
if ext in (".xlsx",):
return "xlsx"
if ext in (".pptx",):
return "pptx"
if ext in (".odt",):
return "odt"
if ext in (".html", ".htm"):
return "html"
if ext in (".md", ".markdown", ".mdx"):
return "markdown"
if data is not None:
if data[:4] == b"%PDF":
return "pdf"
if data[:100].lstrip().startswith(b"<") and b"svg" in data[:500].lower():
return "svg"
if data[:2] == b"PK":
# zip-based; sniff
try:
with zipfile.ZipFile(io.BytesIO(data)) as zf:
names = set(zf.namelist())
if "word/document.xml" in names:
return "docx"
if "xl/workbook.xml" in names:
return "xlsx"
if "ppt/presentation.xml" in names:
return "pptx"
if "content.xml" in names and "meta.xml" in names:
return "odt"
except zipfile.BadZipFile:
pass
return "unknown"
def _blob_hits(blob: bytes) -> tuple[bool, bool, list[str]]:
lower = blob.lower()
findings: list[str] = []
has_c2pa = False
has_ai = False
for n in C2PA_MARKERS:
if n.lower() in lower:
has_c2pa = True
findings.append(f"marker:{n.decode('ascii', errors='replace')}")
for n in AI_META_HINTS:
if n.lower() in lower:
has_ai = True
label = n.decode("ascii", errors="replace")
if label not in {f.split(":", 1)[-1] for f in findings}:
findings.append(f"ai:{label}")
return has_c2pa, has_ai or has_c2pa, findings[:30]
RE_DATA_IMAGE_URI = re.compile(
r"data:image\/(?P<mime>[a-zA-Z0-9\+\-\.]+)(?P<params>;[^\s\"'\)<>]+)?,(?P<payload>[A-Za-z0-9+/=\s%]+)",
re.I,
)
def _inspect_embedded_data_uris(text: str) -> tuple[bool, bool, list[str]]:
has_c2pa = False
has_ai = False
findings: list[str] = []
for m in RE_DATA_IMAGE_URI.finditer(text):
mime = m.group("mime").lower()
params = (m.group("params") or "").lower()
payload = m.group("payload")
is_b64 = "base64" in params
try:
if is_b64:
raw_b64 = re.sub(r"\s+", "", payload)
pad = len(raw_b64) % 4
if pad:
raw_b64 += "=" * (4 - pad)
data = base64.b64decode(raw_b64)
else:
data = urllib.parse.unquote_to_bytes(payload)
except Exception: # noqa: S112 - malformed data URI; skip to next match
continue
if not data:
continue
fmt = detect_image_format(data)
if fmt == "png":
sub_c2pa, sub_ai, sub_findings = inspect_png(data)
elif fmt == "jpeg":
sub_c2pa, sub_ai, sub_findings = inspect_jpeg(data)
elif fmt == "webp":
sub_c2pa, sub_ai, sub_findings = inspect_webp(data)
elif fmt in ("avif", "heic"):
sub_c2pa, sub_ai, sub_findings = inspect_isobmff(data, fmt)
elif "svg" in mime or data.lstrip().startswith(b"<"):
sub_c2pa, sub_ai, sub_findings, _ = inspect_svg(data)
else:
sub_c2pa, sub_ai, sub_findings = _blob_hits(data)
if sub_c2pa:
has_c2pa = True
if sub_ai or sub_c2pa:
has_ai = True
for f in sub_findings:
findings.append(f"embedded data:image/{mime}: {f}")
return has_c2pa, has_ai, findings
def _clean_embedded_data_uris(
text: str, *, strip_all_metadata: bool = True
) -> tuple[str, list[str]]:
actions: list[str] = []
def _replace_uri(m: re.Match[str]) -> str:
full_match = m.group(0)
mime = m.group("mime")
params = m.group("params") or ""
payload = m.group("payload")
is_b64 = "base64" in params.lower()
try:
if is_b64:
raw_b64 = re.sub(r"\s+", "", payload)
pad = len(raw_b64) % 4
if pad:
raw_b64 += "=" * (4 - pad)
data = base64.b64decode(raw_b64)
else:
data = urllib.parse.unquote_to_bytes(payload)
except Exception:
return full_match
if not data:
return full_match
fmt = detect_image_format(data)
sub_actions: list[str] = []
cleaned_bytes = data
try:
if fmt == "png":
cleaned_bytes, sub_actions = strip_png(data, strip_all_text=strip_all_metadata)
elif fmt == "jpeg":
cleaned_bytes, sub_actions = strip_jpeg(data, strip_all_app=strip_all_metadata)
elif fmt == "webp":
cleaned_bytes, sub_actions = strip_webp(data, strip_all_metadata=strip_all_metadata)
elif fmt in ("avif", "heic"):
cleaned_bytes, sub_actions = strip_isobmff(
data, fmt, strip_all_metadata=strip_all_metadata
)
elif "svg" in mime.lower() or data.lstrip().startswith(b"<"):
cleaned_bytes, sub_actions = clean_svg(data)
except Exception:
return full_match
if not any("drop" in a.lower() for a in sub_actions) or cleaned_bytes == data:
return full_match
actions.append(f"cleaned embedded data:image/{mime} ({', '.join(sub_actions[:2])})")
if is_b64:
new_b64 = base64.b64encode(cleaned_bytes).decode("ascii")
return f"data:image/{mime}{params},{new_b64}"
else:
new_payload = urllib.parse.quote_from_bytes(cleaned_bytes)
return f"data:image/{mime}{params},{new_payload}"
out = RE_DATA_IMAGE_URI.sub(_replace_uri, text)
return out, actions
# ---------------------------------------------------------------------------
# Markdown frontmatter
# ---------------------------------------------------------------------------
_FM_RE = re.compile(r"\A---\r?\n(.*?)\r?\n---\r?\n?", re.DOTALL)
def _parse_simple_yaml_keys(block: str) -> list[tuple[str, str, int]]:
"""Return list of (key, full_line, line_index) for top-level keys only."""
rows: list[tuple[str, str, int]] = []
for i, line in enumerate(block.splitlines()):
if not line.strip() or line.strip().startswith("#"):
continue
if line[0] in (" ", "\t", "-"):
continue # nested / list — leave alone
m = re.match(r"^([A-Za-z0-9_.-]+)\s*:", line)
if m:
rows.append((m.group(1), line, i))
return rows
def inspect_markdown(text: str) -> tuple[bool, bool, list[str], dict]:
findings: list[str] = []
has_ai = False
has_fm = False
keys = []
m = _FM_RE.match(text)
if m:
has_fm = True
block = m.group(1)
for key, _line, _i in _parse_simple_yaml_keys(block):
keys.append(key)
if key.lower() in AI_FRONTMATTER_KEYS or AI_META_NAME_RE.search(key):
has_ai = True
findings.append(f"frontmatter key: {key}")
# also check value
val = _line.split(":", 1)[1] if ":" in _line else ""
if AI_META_NAME_RE.search(val):
has_ai = True
findings.append(f"frontmatter value hit on {key}")
uri_c2pa, uri_ai, uri_findings = _inspect_embedded_data_uris(text)
if uri_c2pa:
has_ai = True
if uri_ai:
has_ai = True
findings.extend(uri_findings)
c2pa = uri_c2pa or any("c2pa" in f.lower() or "content" in f.lower() for f in findings)
return c2pa, has_ai, findings, {"has_frontmatter": has_fm, "keys": keys}
def clean_markdown(text: str) -> tuple[str, list[str]]:
actions: list[str] = []
m = _FM_RE.match(text)
if m:
block = m.group(1)
body = text[m.end() :]
kept: list[str] = []
dropping = False # inside the nested block of a dropped top-level key
for line in block.splitlines():
stripped = line.strip()
# Blank lines and comments belong to whichever block we are inside.
if not stripped or stripped.startswith("#"):
if not dropping:
kept.append(line)
continue
# Continuation lines (nested mappings, list items) follow their parent.
if line[0] in (" ", "\t", "-"):
if not dropping:
kept.append(line)
continue
km = re.match(r"^([A-Za-z0-9_.-]+)\s*:", line)
if not km:
dropping = False
kept.append(line)
continue
key = km.group(1)
val = line.split(":", 1)[1] if ":" in line else ""
if key.lower() in AI_FRONTMATTER_KEYS or AI_META_NAME_RE.search(key):
actions.append(f"drop frontmatter key: {key}")
dropping = True
continue
if AI_META_NAME_RE.search(val):
actions.append(f"drop frontmatter key (value hit): {key}")
dropping = True
continue
dropping = False
kept.append(line)
new_block = "\n".join(kept).strip("\n")
if new_block:
out = f"---\n{new_block}\n---\n{body}"
else:
out = body.lstrip("\n")
actions.append("removed empty frontmatter block")
else:
out = text
out, uri_actions = _clean_embedded_data_uris(out)
if uri_actions:
actions.extend(uri_actions)
if not actions:
actions.append("no AI frontmatter keys or embedded data URIs removed")
return out, actions
# ---------------------------------------------------------------------------
# HTML
# ---------------------------------------------------------------------------
_META_TAG_RE = re.compile(
r"<meta\b[^>]*>",
re.I,
)
_META_ATTR_RE = re.compile(
r"""(name|property|content|generator)\s*=\s*["']([^"']*)["']""",
re.I,
)
# Known AI vendor names for the "generator" meta tag. A plain CMS generator
# (WordPress, Elementor) is CMS provenance, not AI-generator metadata.
_GENERATOR_AI_RE = re.compile(
r"claude|anthropic|openai|chatgpt|gemini|synthid|copilot|midjourney|dall.?e|stable.?diffusion",
re.I,
)
def _meta_attrs(tag: str) -> dict[str, str]:
return {name.lower(): value for name, value in _META_ATTR_RE.findall(tag)}
def _is_cms_generator_meta(tag: str) -> bool:
"""Return True for a generator meta tag that is CMS provenance, not AI."""
attrs = _meta_attrs(tag)
name_or_prop = (
attrs.get("name") or attrs.get("property") or attrs.get("generator") or ""
).lower()
if name_or_prop != "generator":
return False
return not (_GENERATOR_AI_RE.search(attrs.get("content", "")) or _GENERATOR_AI_RE.search(tag))
_JSONLD_RE = re.compile(
r"<script\b[^>]*type\s*=\s*[\"']application/ld\+json[\"'][^>]*>.*?</script>",
re.I | re.DOTALL,
)
def inspect_html(text: str) -> tuple[bool, bool, list[str], dict]:
findings: list[str] = []
has_ai = False
has_c2pa = False
for tag in _META_TAG_RE.findall(text):
if re.search(r"c2pa|content.?credential", tag, re.I):
has_c2pa = True
if _is_cms_generator_meta(tag):
findings.append(f"info: cms generator: {tag[:120]}")
continue
if AI_META_NAME_RE.search(tag) or any(
h.decode("ascii", "ignore").lower() in tag.lower() for h in AI_META_HINTS[:12]
):
has_ai = True
findings.append(f"meta: {tag[:120]}")
for m in _JSONLD_RE.finditer(text):
blob = m.group(0)
if AI_META_NAME_RE.search(blob) or re.search(
r"DigitalSourceType|trainedAlgorithmicMedia|SoftwareAgent", blob, re.I
):
has_ai = True
findings.append("json-ld provenance-like block")
if re.search(r"c2pa|contentcredential", blob, re.I):
has_c2pa = True
# data-ai* attributes
for m in re.finditer(r"\bdata-ai[\w-]*\s*=\s*[\"'][^\"']*[\"']", text, re.I):
has_ai = True
findings.append(f"attr: {m.group(0)[:80]}")
uri_c2pa, uri_ai, uri_findings = _inspect_embedded_data_uris(text)
if uri_c2pa:
has_c2pa = True
if uri_ai:
has_ai = True
findings.extend(uri_findings)
return has_c2pa, has_ai, findings, {}
def clean_html(text: str) -> tuple[str, list[str]]:
actions: list[str] = []
def _meta_sub(m: re.Match[str]) -> str:
tag = m.group(0)
if _is_cms_generator_meta(tag):
return tag
if AI_META_NAME_RE.search(tag) or re.search(
r"generator|claude|anthropic|openai|gemini|synthid|c2pa|aigc", tag, re.I
):
actions.append(f"drop meta: {tag[:80]}")
return ""
return tag
out = _META_TAG_RE.sub(_meta_sub, text)
def _jsonld_sub(m: re.Match[str]) -> str:
blob = m.group(0)
if AI_META_NAME_RE.search(blob) or re.search(
r"DigitalSourceType|trainedAlgorithmicMedia|SoftwareAgent", blob, re.I
):
actions.append("drop json-ld provenance-like script")
return ""
return blob
out = _JSONLD_RE.sub(_jsonld_sub, out)
out2, n = re.subn(r"\sdata-ai[\w-]*\s*=\s*[\"'][^\"']*[\"']", "", out, flags=re.I)
if n:
actions.append(f"drop data-ai* attributes x{n}")
out = out2
out, uri_actions = _clean_embedded_data_uris(out)
if uri_actions:
actions.extend(uri_actions)
if not actions:
actions.append("no HTML AI meta removed")
return out, actions
# ---------------------------------------------------------------------------
# SVG
# ---------------------------------------------------------------------------
def inspect_svg(data: bytes) -> tuple[bool, bool, list[str], dict]:
findings: list[str] = []
has_c2pa, has_ai, hits = _blob_hits(data)
findings.extend(hits)
try:
text = data.decode("utf-8", errors="replace")
if re.search(r"<metadata[\s>]", text, re.I):
findings.append("svg <metadata> present")
has_ai = True # often XMP; treat as inspect signal
if re.search(r"xmpmeta|rdf:RDF|contentcredentials", text, re.I):
has_ai = True
findings.append("XMP/RDF-like content in SVG")
if re.search(r"c2pa|jumbf", text, re.I):
has_c2pa = True
uri_c2pa, uri_ai, uri_findings = _inspect_embedded_data_uris(text)
if uri_c2pa:
has_c2pa = True
if uri_ai:
has_ai = True
findings.extend(uri_findings)
except Exception as e:
findings.append(f"svg decode note: {e}")
return has_c2pa, has_ai or has_c2pa, findings, {}
def clean_svg(data: bytes) -> tuple[bytes, list[str]]:
actions: list[str] = []
text = data.decode("utf-8", errors="surrogateescape")
# Drop metadata blocks
new, n = re.subn(
r"<metadata\b[^>]*>.*?</metadata\s*>",
"",
text,
flags=re.I | re.DOTALL,
)
if n:
actions.append(f"drop <metadata> x{n}")
text = new
# Drop adobe xmp packets
new, n = re.subn(
r"<x:xmpmeta\b[^>]*>.*?</x:xmpmeta\s*>",
"",
text,
flags=re.I | re.DOTALL,
)
if n:
actions.append(f"drop xmpmeta x{n}")
text = new
# Drop comments that look like provenance
def _cmt(m: re.Match[str]) -> str:
body = m.group(0)
if AI_META_NAME_RE.search(body):
actions.append("drop SVG comment with AI markers")
return ""
return body
text = re.sub(r"<!--.*?-->", _cmt, text, flags=re.DOTALL)
# Clean embedded data URIs
text, uri_actions = _clean_embedded_data_uris(text)
if uri_actions:
actions.extend(uri_actions)
if not actions:
# still strip generator attribute on root if present
new, n = re.subn(
r'\s(inkscape:version|sodipodi:docname|generator)\s*=\s*"[^"]*"',
"",
text,
flags=re.I,
)
if n:
actions.append(f"drop generator-like attrs x{n}")
text = new
if not actions:
actions.append("no SVG metadata removed")
return text.encode("utf-8", errors="surrogateescape"), actions
# ---------------------------------------------------------------------------
# DOCX / ODT (zip + XML)
# ---------------------------------------------------------------------------
DOCX_META_PARTS = (
"docProps/core.xml",
"docProps/app.xml",
"docProps/custom.xml",
)
DOCX_CUSTOM_PREFIXES = (
"customXml/",
"docProps/",
)
# Provenance fields in docProps/core.xml and docProps/app.xml that always come
# out empty. dc:title is deliberately not listed: it is the document's own
# heading, not provenance.
DOCX_SCRUB_FIELDS = (
("dc:creator", "dc:creator"),
("cp:lastModifiedBy", "cp:lastModifiedBy"),
("dc:description", "dc:description"),
("cp:keywords", "cp:keywords"),
("dc:subject", "dc:subject"),
("cp:category", "cp:category"),
("Application", "Application"),
("AppVersion", "AppVersion"),
("Company", "Company"),
("Manager", "Manager"),
)
def _zip_namelist(data: bytes) -> list[str]:
with zipfile.ZipFile(io.BytesIO(data)) as zf:
return zf.namelist()
MAX_ZIP_DECOMPRESSED_BYTES = 128 * 1024 * 1024
def _check_zip_budget(info: zipfile.ZipInfo, budget: list[int]) -> None:
"""Reject zip bombs before decompression (ZipInfo.file_size is stored)."""
budget[0] += info.file_size
if budget[0] > MAX_ZIP_DECOMPRESSED_BYTES:
raise ValueError(
"zip decompressed size exceeds cap "
f"({MAX_ZIP_DECOMPRESSED_BYTES} bytes); refusing to process"
)
def _is_docx_meta_part(name: str) -> bool:
"""Return True for DOCX/XLSX/PPTX parts that carry provenance, not visible content."""
return name.startswith(("docProps/", "customXml/"))
def _inspect_ooxml_zip(data: bytes, fmt: str) -> tuple[bool, bool, list[str], dict]:
findings: list[str] = []
has_c2pa = False
has_ai = False
parts: list[str] = []
budget = [0]
try:
with zipfile.ZipFile(io.BytesIO(data)) as zf:
parts = zf.namelist()
for info in zf.infolist():
_check_zip_budget(info, budget)
name = info.filename
# Check media parts for C2PA/AI metadata
if re.search(
r"^(?:word|xl|ppt)/media/.+\.(png|jpe?g|webp|avif|heic|svg)$", name, re.I
):
raw = zf.read(name)
img_fmt = detect_image_format(raw)
sub_c2pa, sub_ai, sub_findings = False, False, []
if img_fmt == "png":
sub_c2pa, sub_ai, sub_findings = inspect_png(raw)
elif img_fmt == "jpeg":
sub_c2pa, sub_ai, sub_findings = inspect_jpeg(raw)
elif img_fmt == "webp":
sub_c2pa, sub_ai, sub_findings = inspect_webp(raw)
elif img_fmt in ("avif", "heic"):
sub_c2pa, sub_ai, sub_findings = inspect_isobmff(raw, img_fmt)
elif name.lower().endswith(".svg") or raw.lstrip().startswith(b"<"):
sub_c2pa, sub_ai, sub_findings, _ = inspect_svg(raw)
if sub_c2pa:
has_c2pa = True
if sub_ai or sub_c2pa:
has_ai = True
for sf in sub_findings:
findings.append(f"{name}: {sf}")
continue
# Only metadata/provenance parts carry AI markers. The visible
# body (word/*.xml, xl/*.xml, ppt/*.xml) may legitimately mention
# vendor names such as "Claude" without being AI-generated metadata.
if not _is_docx_meta_part(name):
continue
raw = zf.read(name)
c2, ai, hits = _blob_hits(raw)
if c2 or ai:
if c2:
has_c2pa = True
if ai:
has_ai = True
findings.append(f"{name}: {', '.join(hits[:6])}")
# always flag customXml presence lightly
custom = [n for n in parts if n.startswith("customXml/")]
if custom:
findings.append(f"customXml parts: {len(custom)}")
except zipfile.BadZipFile:
return False, False, [f"not a valid {fmt.upper()} zip"], {}
return has_c2pa, has_ai or has_c2pa, findings, {"parts": len(parts)}
def inspect_docx(data: bytes) -> tuple[bool, bool, list[str], dict]:
return _inspect_ooxml_zip(data, "docx")
def inspect_xlsx(data: bytes) -> tuple[bool, bool, list[str], dict]:
return _inspect_ooxml_zip(data, "xlsx")
def inspect_pptx(data: bytes) -> tuple[bool, bool, list[str], dict]:
return _inspect_ooxml_zip(data, "pptx")
def _scrub_docx_text(xml_text: str) -> tuple[str, int, int]:
"""Run Layer A over the ``<w:t>`` text runs of a DOCX part.
Only ``w:t`` nodes are touched: field codes (``w:instrText``), run/paragraph
properties and the surrounding XML are left byte-identical. If leading or
trailing whitespace survives the clean, the node keeps
``xml:space="preserve"`` so Word does not trim it.
"""
from text_unicode import clean_text # local import to avoid cycles
removed = 0
replaced = 0
def _repl(m: re.Match[str]) -> str:
nonlocal removed, replaced
open_tag, inner, close_tag = m.group(1), m.group(2), m.group(3)
new_inner, stats = clean_text(inner)
if not (stats["removed_count"] or stats["replaced_count"]):
return m.group(0)
removed += stats["removed_count"]
replaced += stats["replaced_count"]
if (new_inner[:1].isspace() or new_inner[-1:].isspace()) and "xml:space" not in open_tag:
open_tag = open_tag[:-1] + ' xml:space="preserve">'
return open_tag + new_inner + close_tag
new = re.sub(r"(<w:t\b[^>]*>)(.*?)(</w:t>)", _repl, xml_text, flags=re.S)
return new, removed, replaced
def _scrub_xlsx_text(xml_text: str) -> tuple[str, int, int]:
"""Run Layer A over the ``<t>`` text elements of an XLSX part."""
from text_unicode import clean_text # local import to avoid cycles
removed = 0
replaced = 0
def _repl(m: re.Match[str]) -> str:
nonlocal removed, replaced
open_tag, inner, close_tag = m.group(1), m.group(2), m.group(3)
new_inner, stats = clean_text(inner)
if not (stats["removed_count"] or stats["replaced_count"]):
return m.group(0)
removed += stats["removed_count"]
replaced += stats["replaced_count"]
if (new_inner[:1].isspace() or new_inner[-1:].isspace()) and "xml:space" not in open_tag:
open_tag = open_tag[:-1] + ' xml:space="preserve">'
return open_tag + new_inner + close_tag
new = re.sub(r"(<t\b[^>]*>)(.*?)(</t>)", _repl, xml_text, flags=re.S)
return new, removed, replaced
def _scrub_pptx_text(xml_text: str) -> tuple[str, int, int]:
"""Run Layer A over the ``<a:t>`` text elements of a PPTX part."""
from text_unicode import clean_text # local import to avoid cycles
removed = 0
replaced = 0
def _repl(m: re.Match[str]) -> str:
nonlocal removed, replaced
open_tag, inner, close_tag = m.group(1), m.group(2), m.group(3)
new_inner, stats = clean_text(inner)
if not (stats["removed_count"] or stats["replaced_count"]):
return m.group(0)
removed += stats["removed_count"]
replaced += stats["replaced_count"]
return open_tag + new_inner + close_tag
new = re.sub(r"(<a:t\b[^>]*>)(.*?)(</a:t>)", _repl, xml_text, flags=re.S)
return new, removed, replaced
def _scrub_odt_text(xml_text: str) -> tuple[str, int, int]:
"""Run Layer A over ODF paragraph text (``text:p`` content, incl. spans).
``text:span``/``text:tab``/``text:s`` children live inside the paragraph,
so cleaning the paragraph content covers the visible text. The markup
itself is untouched.
"""
from text_unicode import clean_text # local import to avoid cycles
removed = 0
replaced = 0
def _repl(m: re.Match[str]) -> str:
nonlocal removed, replaced
open_tag, inner, close_tag = m.group(1), m.group(2), m.group(3)
new_inner, stats = clean_text(inner)
if not (stats["removed_count"] or stats["replaced_count"]):
return m.group(0)
removed += stats["removed_count"]
replaced += stats["replaced_count"]
return open_tag + new_inner + close_tag
new = re.sub(r"(<text:p\b[^>]*>)(.*?)(</text:p>)", _repl, xml_text, flags=re.S)
return new, removed, replaced
def _prune_dangling_relationships(
rels_name: str, raw: bytes, kept_names: set[str]
) -> tuple[bytes, int]:
"""Drop <Relationship> entries whose internal target part no longer exists.
Removing a part (e.g. a customXml tree) must also remove the relationships
that point at it, or the package is malformed: python-docx refuses to open
it and Word offers to repair it. External relationships (``TargetMode``)
and the package root (``Target="/"``) are left alone. ``rels_name`` is the
archive member like ``word/_rels/document.xml.rels``; ``kept_names`` is the
set of archive members that survive cleaning.
"""
base = posixpath.dirname(posixpath.dirname(rels_name))
text = raw.decode("utf-8", errors="replace")
dropped = [0]
def _target_attr(tag: str) -> str:
m = re.search(r'\bTarget\s*=\s*"([^"]*)"', tag, re.I)
return m.group(1) if m else ""
def _drop(m: re.Match[str]) -> str:
tag = m.group(0)
if re.search(r"\bTargetMode\s*=", tag, re.I):
return tag # external (http / mailto / ...) — never pruned
target = _target_attr(tag)
if target.startswith("/"):
resolved = posixpath.normpath(target.lstrip("/"))
else:
resolved = posixpath.normpath(posixpath.join(base, target))
if resolved in ("", "."):
return tag # points at the package root
if resolved in kept_names:
return tag
dropped[0] += 1
return ""
new = re.sub(r"<Relationship\b[^>]*/>", _drop, text, flags=re.I)
return new.encode("utf-8"), dropped[0]
def _scrub_ooxml_zip(
data: bytes, fmt: str, *, also_layer_a_text: bool = True
) -> tuple[bytes, list[str]]:
actions: list[str] = []
budget = [0]
layer_removed = 0
layer_replaced = 0
kept: list[tuple[zipfile.ZipInfo, bytes]] = []
with zipfile.ZipFile(io.BytesIO(data)) as zin:
for info in zin.infolist():
name = info.filename
_check_zip_budget(info, budget)
raw = zin.read(name)
# 1. Clean embedded media (PNG, JPEG, WebP, AVIF, HEIC, SVG)
if re.search(r"^(?:word|xl|ppt)/media/.+\.(png|jpe?g|webp|avif|heic|svg)$", name, re.I):
img_fmt = detect_image_format(raw)
sub_actions: list[str] = []
cleaned_bytes = raw
try:
if img_fmt == "png":
cleaned_bytes, sub_actions = strip_png(raw, strip_all_text=True)
elif img_fmt == "jpeg":
cleaned_bytes, sub_actions = strip_jpeg(raw, strip_all_app=True)
elif img_fmt == "webp":
cleaned_bytes, sub_actions = strip_webp(raw, strip_all_metadata=True)
elif img_fmt in ("avif", "heic"):
cleaned_bytes, sub_actions = strip_isobmff(
raw, img_fmt, strip_all_metadata=True
)
elif name.lower().endswith(".svg") or raw.lstrip().startswith(b"<"):
cleaned_bytes, sub_actions = clean_svg(raw)
except Exception: # noqa: S110 - malformed embedded media; keep original part
pass
if any("drop" in a.lower() for a in sub_actions) and cleaned_bytes != raw:
actions.append(f"clean embedded media in {name} ({', '.join(sub_actions[:2])})")
raw = cleaned_bytes
kept.append((info, raw))
continue
# 2. Drop customXml trees
if name.startswith("customXml/"):
actions.append(f"drop part {name}")
continue
# 3. docProps/ provenance
if name in DOCX_META_PARTS or name.startswith("docProps/"):
if name.endswith("custom.xml"):
actions.append(f"drop part {name}")
continue
text = raw.decode("utf-8", errors="replace")
new = text
for tag, label in DOCX_SCRUB_FIELDS:
pat = rf"(<{tag}\b[^>]*>)(.*?)(</{tag}>)"
def _empty(m: re.Match[str], _label=label, _name=name) -> str:
if m.group(2):
actions.append(f"scrub {_name} field {_label}")
return m.group(1) + m.group(3)
new = re.sub(pat, _empty, new, flags=re.I | re.DOTALL)
raw = new.encode("utf-8")
# 4. [Content_Types].xml overrides
if name == "[Content_Types].xml":
text = raw.decode("utf-8", errors="replace")
new, n = re.subn(
r'<Override\b[^>]*PartName="/customXml/[^"]*"[^>]*/>',
"",
text,
)
if n:
actions.append(f"drop Content_Types customXml overrides x{n}")
raw = new.encode("utf-8")
new, n = re.subn(
r'<Override\b[^>]*PartName="/docProps/custom\.xml"[^>]*/>',
"",
raw.decode("utf-8", errors="replace"),
)
if n:
actions.append(f"drop Content_Types custom.xml override x{n}")
raw = new.encode("utf-8")
# 5. Layer A text runs
if also_layer_a_text and name.endswith(".xml"):
if fmt == "docx" and name.startswith("word/"):
text = raw.decode("utf-8", errors="replace")
new, r, rp = _scrub_docx_text(text)
if r or rp:
layer_removed += r
layer_replaced += rp
raw = new.encode("utf-8")
elif fmt == "xlsx" and name.startswith("xl/"):
text = raw.decode("utf-8", errors="replace")
new, r, rp = _scrub_xlsx_text(text)
if r or rp:
layer_removed += r
layer_replaced += rp
raw = new.encode("utf-8")
elif fmt == "pptx" and name.startswith("ppt/"):
text = raw.decode("utf-8", errors="replace")
new, r, rp = _scrub_pptx_text(text)
if r or rp:
layer_removed += r
layer_replaced += rp
raw = new.encode("utf-8")
kept.append((info, raw))
kept_names = {info.filename for info, _ in kept}
final: list[tuple[zipfile.ZipInfo, bytes]] = []
for info, raw in kept:
part_raw = raw
if info.filename.endswith(".rels"):
part_raw, n = _prune_dangling_relationships(info.filename, raw, kept_names)
if n:
actions.append(f"prune dangling relationships x{n} in {info.filename}")
final.append((info, part_raw))
out_buf = io.BytesIO()
with zipfile.ZipFile(out_buf, "w", compression=zipfile.ZIP_DEFLATED) as zout:
for info, raw in final:
zout.writestr(info, raw)
if layer_removed or layer_replaced:
actions.append(f"layer A text: removed={layer_removed} replaced={layer_replaced}")
if not actions:
actions.append(f"no {fmt.upper()} metadata parts removed")
return out_buf.getvalue(), actions
def clean_docx(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]:
return _scrub_ooxml_zip(data, "docx", also_layer_a_text=also_layer_a_text)
def clean_xlsx(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]:
return _scrub_ooxml_zip(data, "xlsx", also_layer_a_text=also_layer_a_text)
def clean_pptx(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]:
return _scrub_ooxml_zip(data, "pptx", also_layer_a_text=also_layer_a_text)
def inspect_odt(data: bytes) -> tuple[bool, bool, list[str], dict]:
findings: list[str] = []
has_c2pa = False
has_ai = False
budget = [0]
try:
with zipfile.ZipFile(io.BytesIO(data)) as zf:
for info in zf.infolist():
_check_zip_budget(info, budget)
raw = zf.read(info.filename)
c2, ai, hits = _blob_hits(raw)
if c2 or ai:
if c2:
has_c2pa = True
if ai:
has_ai = True
findings.append(f"{info.filename}: {', '.join(hits[:6])}")
if "meta.xml" in zf.namelist():
meta = zf.read("meta.xml").decode("utf-8", errors="replace")
if re.search(r"generator|claude|openai|anthropic|gemini", meta, re.I):
has_ai = True
findings.append("meta.xml generator-like fields")
except zipfile.BadZipFile:
return False, False, ["not a valid ODT zip"], {}
return has_c2pa, has_ai or has_c2pa, findings, {}
def clean_odt(data: bytes, *, also_layer_a_text: bool = True) -> tuple[bytes, list[str]]:
actions: list[str] = []
out_buf = io.BytesIO()
budget = [0]
layer_removed = 0
layer_replaced = 0
with (
zipfile.ZipFile(io.BytesIO(data)) as zin,
zipfile.ZipFile(out_buf, "w", compression=zipfile.ZIP_DEFLATED) as zout,
):
for info in zin.infolist():
name = info.filename
_check_zip_budget(info, budget)
raw = zin.read(name)
if name == "meta.xml":
text = raw.decode("utf-8", errors="replace")
new, n = re.subn(
r"<meta:generator\b[^>]*>.*?</meta:generator\s*>",
"",
text,
flags=re.I | re.DOTALL,
)
if n:
actions.append("drop meta:generator")
text = new
# scrub creator-like if AI
def _creator(m: re.Match[str]) -> str:
if AI_META_NAME_RE.search(m.group(0)):
actions.append("scrub creator-like meta")
return ""
return m.group(0)
text = re.sub(
r"<dc:creator\b[^>]*>.*?</dc:creator\s*>",
_creator,
text,
flags=re.I | re.DOTALL,
)
raw = text.encode("utf-8")
else:
c2, ai, _ = _blob_hits(raw)
if (c2 or ai) and name not in (
"content.xml",
"styles.xml",
"mimetype",
"META-INF/manifest.xml",
):
actions.append(f"drop part {name} (AI/C2PA markers)")
continue
# Layer A over the visible paragraph text of the body part.
if also_layer_a_text and name == "content.xml":
text = raw.decode("utf-8", errors="replace")
new, r, rp = _scrub_odt_text(text)
if r or rp:
layer_removed += r
layer_replaced += rp
raw = new.encode("utf-8")
zout.writestr(info, raw)
if layer_removed or layer_replaced:
actions.append(f"layer A text: removed={layer_removed} replaced={layer_replaced}")
if not actions:
actions.append("no ODT metadata removed")
return out_buf.getvalue(), actions
# ---------------------------------------------------------------------------
# PDF
# ---------------------------------------------------------------------------
_XMP_PACKET_RE = re.compile(
rb"<\?xpacket begin.*?<\?xpacket end[^?]*\?>",
re.I | re.DOTALL,
)
def _pdf_structured_blob(data: bytes) -> bytes:
"""Return PDF bytes with stream payloads removed, plus XMP packets.
Stream payloads are often compressed binary where an AI-marker byte
sequence (e.g. "AIGC") can occur by chance. Scanning only dictionaries and
XMP packets avoids treating those collisions as metadata findings.
"""
no_streams = re.sub(
rb"stream\r?\n.*?endstream",
b"stream endstream",
data,
flags=re.DOTALL,
)
xmp = b"\n".join(_XMP_PACKET_RE.findall(data))
return no_streams + b"\n" + xmp
def inspect_pdf(path: Path, data: bytes) -> tuple[bool, bool, list[str], dict]:
findings: list[str] = []
has_c2pa, has_ai, hits = _blob_hits(_pdf_structured_blob(data))
findings.extend(f"pdf-structured:{h}" for h in hits)
# XMP packet scan
xmp_blob = b"\n".join(_XMP_PACKET_RE.findall(data))
if xmp_blob:
findings.append("XMP packet present")
has_ai = has_ai or bool(
re.search(
rb"digitalSourceType|trainedAlgorithmicMedia|SoftwareAgent|c2pa",
xmp_blob,
re.I,
)
)
tools = run_optional_tools(path)
ct = tools.get("c2patool") or {}
if ct.get("has_manifest"):
has_c2pa = True
findings.append("c2patool reports C2PA-related manifest")
return has_c2pa, has_ai or has_c2pa, findings, {"tools": tools}
def _pdf_structural_rewrite(dest: Path, actions: list[str]) -> bool:
"""Rebuild a PDF so unreferenced objects are dropped.
exiftool's PDF edits are incremental, so freed metadata objects survive in
the byte stream. qpdf re-serializes from the object graph, which is what
actually removes them. No-op (with a warning) when qpdf is absent.
"""
qpdf = which("qpdf")
if not qpdf:
actions.append(
"warning: exiftool PDF edits are incremental — the original metadata "
"bytes remain recoverable; install qpdf for a structural rewrite"
)
return False
tmp = dest.with_name(dest.name + ".qpdf-tmp")
try:
r = subprocess.run(
[qpdf, "--linearize", "--", safe_arg(str(dest)), safe_arg(str(tmp))],
capture_output=True,
text=True,
timeout=120,
check=False,
preexec_fn=subprocess_preexec_fn,
)
except Exception as e:
tmp.unlink(missing_ok=True)
actions.append(f"qpdf rewrite failed: {e}; metadata bytes may remain recoverable")
return False
# qpdf exit codes: 0 = clean, 3 = succeeded with warnings (output written).
if r.returncode in (0, 3) and tmp.is_file() and tmp.stat().st_size > 0:
tmp.replace(dest)
actions.append(f"qpdf --linearize structural rewrite (rc={r.returncode})")
return True
tmp.unlink(missing_ok=True)
actions.append(
f"qpdf rewrite skipped (rc={r.returncode}); metadata bytes may remain recoverable"
)
return False
def clean_pdf(path: Path, dest: Path) -> tuple[list[str], dict]:
"""Best-effort PDF clean. Prefers exiftool; falls back to XMP strip warning."""
actions: list[str] = []
data = path.read_bytes()
dest.parent.mkdir(parents=True, exist_ok=True)
exiftool = which("exiftool")
if exiftool:
safe_write_bytes(dest, data)
try:
r = subprocess.run(
[
exiftool,
"-all=",
"-overwrite_original",
safe_arg(str(dest)),
],
capture_output=True,
text=True,
timeout=60,
check=False,
preexec_fn=subprocess_preexec_fn,
)
actions.append(f"exiftool -all= (rc={r.returncode})")
except Exception as e:
actions.append(f"exiftool failed: {e}")
# exiftool writes PDFs *incrementally*: it appends a
# %BeginExifToolUpdate block that frees the Info object and drops
# /Info from the trailer, but the original metadata bytes stay in the
# file verbatim and are trivially recoverable (exiftool itself can
# revert them with -PDF-update:all=). A structural rewrite is what
# actually drops the now-unreferenced objects.
rewritten = _pdf_structural_rewrite(dest, actions)
c2patool = which("c2patool")
# c2patool does not always strip; leave note
if c2patool:
actions.append("c2patool available for inspect; strip via exiftool/re-export")
return actions, {"mode": "exiftool", "structural_rewrite": rewritten}
# Degraded: strip obvious XMP packets between <?xpacket begin and end
text = data
new, n = re.subn(
rb"<\?xpacket begin.*?<\?xpacket end[^?]*\?>",
b"",
text,
flags=re.I | re.DOTALL,
)
if n:
actions.append(f"stripped XMP xpacket x{n} (degraded; may leave offsets broken)")
# PDF structural risk: document degraded mode clearly
safe_write_bytes(dest, new)
actions.append("warning: pure-stdlib PDF strip is best-effort; prefer exiftool")
return actions, {"mode": "stdlib-xmp", "degraded": True}
safe_write_bytes(dest, data)
actions.append(
"no PDF cleaner available (install exiftool for reliable metadata strip); copied as-is"
)
return actions, {"mode": "copy", "degraded": True}
# ---------------------------------------------------------------------------
# Unified API
# ---------------------------------------------------------------------------
def inspect_container(path: Path) -> ContainerInspectReport:
data = path.read_bytes()
fmt = detect_container_format(path, data)
tools: dict[str, Any] = {}
details: dict[str, Any] = {}
if fmt == "svg":
has_c2pa, has_ai, findings, details = inspect_svg(data)
elif fmt == "pdf":
has_c2pa, has_ai, findings, details = inspect_pdf(path, data)
tools = details.pop("tools", {})
elif fmt == "docx":
has_c2pa, has_ai, findings, details = inspect_docx(data)
elif fmt == "xlsx":
has_c2pa, has_ai, findings, details = inspect_xlsx(data)
elif fmt == "pptx":
has_c2pa, has_ai, findings, details = inspect_pptx(data)
elif fmt == "odt":
has_c2pa, has_ai, findings, details = inspect_odt(data)
elif fmt == "html":
# surrogateescape, not replace: clean_container() decodes the same way,
# and U+FFFD substitutions would make the two disagree on the counts.
body = data.decode("utf-8", errors="surrogateescape")
has_c2pa, has_ai, findings, details = inspect_html(body)
elif fmt == "markdown":
body = data.decode("utf-8", errors="surrogateescape")
has_c2pa, has_ai, findings, details = inspect_markdown(body)
else:
has_c2pa, has_ai, findings = False, False, [f"unsupported container: {fmt}"]
details = {"unsupported": True}
# Layer A body scan for exactly the formats clean_container() scrubs, so
# inspect predicts clean rather than contradicting it.
layer_a_total = 0
layer_a_hits: list[dict] = []
if fmt in ("markdown", "html"):
from text_unicode import inspect_text # local import to avoid cycles
ta = inspect_text(body).to_dict()
layer_a_total = ta["suspicious_total"]
layer_a_hits = ta["hits"]
for h in layer_a_hits:
findings.append(f"layer-a: {h['codepoint']} {h['label']} x{h['count']} ({h['kind']})")
notes: list[str] = []
if fmt == "pdf":
notes.append(
"PDF inspection is best-effort; exiftool/c2patool give more reliable metadata detection"
)
elif fmt in ("docx", "xlsx", "pptx"):
notes.append(f"{fmt.upper()}: metadata/provenance and embedded media are scanned")
if "unsupported" in details:
notes.append(f"format not fully inspected: {fmt}")
if layer_a_total:
notes.append(
f"layer A: {layer_a_total} invisible/format codepoint(s) in body text; "
"clean removes these"
)
if fmt in ("svg", "pdf", "docx", "xlsx", "pptx") and not tools:
tools = run_optional_tools(path)
return ContainerInspectReport(
path=str(path),
format=fmt,
has_c2pa=has_c2pa,
has_ai_metadata=has_ai,
findings=findings,
tools=tools,
details=details,
notes=notes,
layer_a_total=layer_a_total,
layer_a_hits=layer_a_hits,
)
def clean_container(
path: Path,
dest: Path,
fmt: str | None = None,
*,
also_layer_a_text: bool = True,
) -> dict[str, Any]:
"""Clean container metadata; optionally Layer-A scrub text bodies for md/html.
``fmt`` pins the container format when the caller already knows it. This
matters for ``--in-place`` flows where *path* is a ``.bak`` copy whose
suffix would otherwise make markdown/HTML (which have no magic bytes)
classify as ``unknown``.
"""
from text_unicode import clean_text # local import to avoid cycles
data = path.read_bytes()
fmt = fmt or detect_container_format(path, data)
actions: list[str] = []
dest.parent.mkdir(parents=True, exist_ok=True)
meta: dict[str, Any] = {"format": fmt}
if fmt == "svg":
cleaned, actions = clean_svg(data)
safe_write_bytes(dest, cleaned)
elif fmt == "pdf":
actions, meta_extra = clean_pdf(path, dest)
meta.update(meta_extra)
elif fmt == "docx":
cleaned, actions = clean_docx(data, also_layer_a_text=also_layer_a_text)
safe_write_bytes(dest, cleaned)
elif fmt == "xlsx":
cleaned, actions = clean_xlsx(data, also_layer_a_text=also_layer_a_text)
safe_write_bytes(dest, cleaned)
elif fmt == "pptx":
cleaned, actions = clean_pptx(data, also_layer_a_text=also_layer_a_text)
safe_write_bytes(dest, cleaned)
elif fmt == "odt":
cleaned, actions = clean_odt(data, also_layer_a_text=also_layer_a_text)
safe_write_bytes(dest, cleaned)
elif fmt == "html":
text = data.decode("utf-8", errors="surrogateescape")
text, actions = clean_html(text)
if also_layer_a_text:
text2, stats = clean_text(text)
if stats["removed_count"] or stats["replaced_count"]:
actions.append(
f"layer A text: removed={stats['removed_count']} replaced={stats['replaced_count']}"
)
text = text2
safe_write_text(dest, text)
elif fmt == "markdown":
text = data.decode("utf-8", errors="surrogateescape")
text, actions = clean_markdown(text)
if also_layer_a_text:
text2, stats = clean_text(text)
if stats["removed_count"] or stats["replaced_count"]:
actions.append(
f"layer A text: removed={stats['removed_count']} replaced={stats['replaced_count']}"
)
text = text2
safe_write_text(dest, text)
else:
raise ValueError(f"unsupported container format: {fmt}")
after = inspect_container(dest)
return {
"input": str(path),
"output": str(dest),
"format": fmt,
"actions": actions,
"bytes_in": len(data),
"bytes_out": dest.stat().st_size,
"still_has_c2pa": after.has_c2pa,
"still_has_ai_metadata": after.has_ai_metadata,
"post_findings": after.findings,
"meta": meta,
}