From 1cf93d0d60d1efa2660e15885f116be012501271 Mon Sep 17 00:00:00 2001 From: yzxcj797 <54314860+yzxcj797@users.noreply.github.com> Date: Thu, 20 Aug 2026 02:59:24 +0800 Subject: [PATCH] fix: keep collected evidence when a later zip member fails to read (#175) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix: keep collected evidence when a later zip member fails to read The OOXML, ODT, and EPUB inspectors accumulated has_c2pa/has_ai/ findings member by member, then discarded all of it if any later member raised: the except returned hardcoded False, False, ["not a valid X zip"] — markers already found in earlier members were thrown away and a container that could not be fully read reported as one that was read and found clean. The pre-commit gate then exited 0 on exactly the file that carried evidence (#164). Keep the accumulated evidence and append a partial-read note naming the exception class; only a wholly-garbage container (nothing accumulated) keeps the not-a-valid-zip shape. EPUB's except is widened from BadZipFile to the shared _ZIP_PARSE_ERRORS tuple, closing the asymmetry where a zlib error crashed the scan while a CRC failure discarded everything. * fix: silence RUF100/RUF059 lint errors in zip partial-evidence tests --------- Co-authored-by: yzxcj797 Co-authored-by: Guillaume Meyer (The Opinionated Man) <1385518+guillaumemeyer@users.noreply.github.com> Co-authored-by: guillaumemeyer --- service/scripts/container_meta.py | 34 ++++++++- tests/test_zip_partial_evidence.py | 111 +++++++++++++++++++++++++++++ 2 files changed, 142 insertions(+), 3 deletions(-) create mode 100644 tests/test_zip_partial_evidence.py diff --git a/service/scripts/container_meta.py b/service/scripts/container_meta.py index 7641bbb..ebad7a8 100644 --- a/service/scripts/container_meta.py +++ b/service/scripts/container_meta.py @@ -882,7 +882,18 @@ def _inspect_ooxml_zip(data: bytes, fmt: str) -> tuple[bool, bool, list[str], di custom = [n for n in parts if n.startswith("customXml/")] if custom: findings.append(f"customXml parts: {len(custom)}") - except _ZIP_PARSE_ERRORS: + except _ZIP_PARSE_ERRORS as exc: + # A member that failed to read must not discard the evidence already + # collected from earlier members, nor read as "opened and found + # clean": keep the flags/findings, append a partial-read note, and + # only a wholly-garbage container (nothing accumulated) keeps the old + # not-a-valid-zip shape (#164). + if findings: + findings.append( + f"partial read of {fmt.upper()} zip ({exc.__class__.__name__}); " + "evidence above survives, later members were not scanned" + ) + return has_c2pa, has_ai or has_c2pa, findings, {"parts": len(parts)} return False, False, [f"not a valid {fmt.upper()} zip"], {} return has_c2pa, has_ai or has_c2pa, findings, {"parts": len(parts)} @@ -1349,7 +1360,13 @@ def inspect_odt(data: bytes) -> tuple[bool, bool, list[str], dict]: if re.search(r"generator|claude|openai|anthropic|gemini", meta, re.I): has_ai = True findings.append("meta.xml generator-like fields") - except _ZIP_PARSE_ERRORS: + except _ZIP_PARSE_ERRORS as exc: + if findings: + findings.append( + f"partial read of ODT zip ({exc.__class__.__name__}); " + "evidence above survives, later members were not scanned" + ) + return has_c2pa, has_ai or has_c2pa, findings, {} return False, False, ["not a valid ODT zip"], {} return has_c2pa, has_ai or has_c2pa, findings, {} @@ -1496,6 +1513,7 @@ def inspect_epub(data: bytes) -> tuple[bool, bool, list[str], dict]: has_ai = False budget = [0] encrypted = _epub_encrypted_parts(data) + names: list[str] = [] try: with zipfile.ZipFile(io.BytesIO(data)) as zf: names = zf.namelist() @@ -1532,7 +1550,17 @@ def inspect_epub(data: bytes) -> tuple[bool, bool, list[str], dict]: has_c2pa = has_c2pa or c2 has_ai = has_ai or ai findings.append(f"{name}: {', '.join(hits[:6])}") - except zipfile.BadZipFile: + except _ZIP_PARSE_ERRORS as exc: + # EPUB previously caught only BadZipFile, so a zlib error propagated + # (crashing the scan) while a CRC failure discarded everything — + # neither is right. Same partial-read contract as the OOXML/ODT + # inspectors (#164). + if findings: + findings.append( + f"partial read of EPUB zip ({exc.__class__.__name__}); " + "evidence above survives, later members were not scanned" + ) + return has_c2pa, has_ai or has_c2pa, findings, {"parts": len(names)} return False, False, ["not a valid EPUB zip"], {} return has_c2pa, has_ai or has_c2pa, findings, {"parts": len(names)} diff --git a/tests/test_zip_partial_evidence.py b/tests/test_zip_partial_evidence.py new file mode 100644 index 0000000..2cd04ef --- /dev/null +++ b/tests/test_zip_partial_evidence.py @@ -0,0 +1,111 @@ +"""Partial zip reads keep already-collected evidence (#164). + +One unreadable member used to discard every C2PA/AI marker found in earlier +members and replace them with a hardcoded clean-looking result. Evidence must +survive, with a note naming the partial read. +""" + +from __future__ import annotations + +import io +import sys +import zipfile +import zlib +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +SCRIPTS = ROOT / "service" / "scripts" +sys.path.insert(0, str(SCRIPTS)) + +import container_meta + +MARKER = b"c2pa contentcredentials OpenAI" + + +def _docx_bytes() -> bytes: + buf = io.BytesIO() + with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf: + zf.writestr("[Content_Types].xml", "") + zf.writestr("word/document.xml", "") + zf.writestr("docProps/core.xml", MARKER) + zf.writestr("customXml/item1.xml", "") + return buf.getvalue() + + +def _fail_read_for(container_meta, fail_member: str): + """Make _read_zip_member raise zlib.error for fail_member (deterministic + member-read failure with an intact archive structure — the real-world + triggers per the issue: corrupt deflate, CRC mismatch, unsupported + compression, encrypted member).""" + orig = container_meta._read_zip_member + + def flaky(zf, info, budget): + if info.filename == fail_member: + raise zlib.error("Error -3 while decompressing data: invalid code lengths set") + return orig(zf, info, budget) + + container_meta._read_zip_member = flaky + return lambda: setattr(container_meta, "_read_zip_member", orig) + + +def test_corrupt_later_member_keeps_earlier_evidence(): + # docProps/core.xml is read before customXml/*; failing the later member + # must not discard the earlier marker evidence. + restore = _fail_read_for(container_meta, "customXml/item1.xml") + try: + has_c2pa, has_ai, findings, _ = container_meta.inspect_docx(_docx_bytes()) + finally: + restore() + assert has_c2pa and has_ai, findings + assert any("docProps/core.xml" in f for f in findings), findings + assert any("partial read" in f.lower() for f in findings), findings + + +def test_odt_partial_read_keeps_evidence(): + # mimetype sorts first; failing content.xml (read after it) keeps the + # mimetype blob evidence and the meta.xml generator finding. + buf = io.BytesIO() + with zipfile.ZipFile(buf, "w") as zf: + zf.writestr("mimetype", "application/vnd.oasis.opendocument.text c2pa") + zf.writestr("content.xml", "") + zf.writestr("meta.xml", "Claude") + restore = _fail_read_for(container_meta, "content.xml") + try: + has_c2pa, _has_ai, findings, _ = container_meta.inspect_odt(buf.getvalue()) + finally: + restore() + assert has_c2pa, findings + assert any("mimetype" in f for f in findings), findings + assert any("partial read" in f.lower() for f in findings), findings + + +def test_epub_partial_read_keeps_evidence(): + buf = io.BytesIO() + with zipfile.ZipFile(buf, "w") as zf: + zf.writestr("mimetype", "application/epub+zip") + zf.writestr("META-INF/container.xml", "") + zf.writestr("content.opf", "Generated by OpenAI") + zf.writestr("chapter1.xhtml", "ok") + restore = _fail_read_for(container_meta, "chapter1.xhtml") + try: + has_ai, findings = None, None + r = container_meta.inspect_epub(buf.getvalue()) + _has_c2pa, has_ai, findings, _ = r + finally: + restore() + assert has_ai, findings + assert any("content.opf" in f for f in findings), findings + assert any("partial read" in f.lower() for f in findings), findings + + +def test_wholly_garbage_bytes_keep_not_a_valid_shape(): + has_c2pa, has_ai, findings, _ = container_meta.inspect_docx(b"not a zip at all") + assert has_c2pa is False and has_ai is False + assert findings == ["not a valid DOCX zip"] + + +def test_intact_docx_finds_markers_without_partial_note(): + has_c2pa, _has_ai, findings, _ = container_meta.inspect_docx(_docx_bytes()) + assert has_c2pa is True + assert any("docProps/core.xml" in f for f in findings) + assert not any("partial read" in f.lower() for f in findings)