From fd46e649e329a52668aa6dd3aa109724cbf9a53c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 24 Sep 2026 00:00:24 +0900 Subject: [PATCH 1/5] fix(pingora): admit OOXML manuscripts by fail-closed structural DOCX evidence late-life-anxiety-reanalysis#257's required OpenCode bootstrap (run 35808203458 / job 107013612421, exit 2) failed inside the trusted central `scripts/ci/pingora_edge_policy.py` with Runtime policy candidate docs/delivery_interim_20260920/ air_render_00cdc51_g7integration_20260921_205341/ manuscript_interim_20260921.docx is not valid UTF-8 `.docx` had no `BINARY_DOCUMENT_MAGIC` entry, so a tracked research manuscript under `docs/` was never a candidate binary documentation asset and reached the ordinary content scan's UTF-8 decode, which fails closed for any genuinely binary file. Reuse the module's existing `_binary_documentation_evidence_confirms` path rather than adding a parallel mechanism: register `.docx` in `BINARY_DOCUMENT_MAGIC` and dispatch it to a new `_is_complete_docx`, a sibling of `_is_complete_hwpx`. Admission is proved structurally -- unprefixed ZIP, exact end record with a consistent comment length, unique members, every part in `DOCX_REQUIRED_PARTS` present, non-empty and unencrypted, and `DOCX_MAIN_DOCUMENT_CONTENT_TYPE` declared in a `[Content_Types].xml` read bounded by `MAX_DOCX_CONTENT_TYPES_BYTES`. Unlike HWPX a conforming `.docx` has no stored `mimetype` member and DEFLATEs every part, so neither is required. Nothing becomes neutral or skipped and no file type is blanket-exempted: a truncated, prefixed, appended-to, encrypted, entryless, duplicated or non-WordprocessingML package returns False and falls through to the same scan as before, which still fails the policy. A file that decodes as valid UTF-8 is still never treated as a binary artifact, and `docs/nginx/*.docx` stays rejected on `_runtime_path_rule`. The research artifact is not deleted, relocated, renamed or excluded. `.docx` under a declared #2193 artifact prefix is now held to this structural proof instead of the UTF-8 complement -- strictly narrower, failing closed in the same direction. Verified offline against the real exact-head bytes of PR#257 (706a81e5a6f88ad74544ab9cf89d4da2b9e6a44d, 95620 bytes, under the 1 MiB Contents API ceiling): `_is_complete_docx` returns True for the artifact and False for its truncated and script-prefixed variants. The fixture in `tests/test_pingora_docx_evidence.py` reproduces that container shape without copying another repository's artifact into this one. Refs: ContextualWisdomLab/late-life-anxiety-reanalysis#257 Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01FvosNg4GVUjaV5UfrimrsX --- scripts/ci/pingora_edge_policy.py | 83 +++++++++++++++-- tests/test_pingora_docx_evidence.py | 136 ++++++++++++++++++++++++++++ 2 files changed, 212 insertions(+), 7 deletions(-) create mode 100644 tests/test_pingora_docx_evidence.py diff --git a/scripts/ci/pingora_edge_policy.py b/scripts/ci/pingora_edge_policy.py index b53a68c6ae..9c6750aee3 100644 --- a/scripts/ci/pingora_edge_policy.py +++ b/scripts/ci/pingora_edge_policy.py @@ -22,7 +22,7 @@ Suffix decision: most research-data formats (``.xlsx``, ``.sav``, ``.rds``, ``.npz``, ...) have no entry in ``BINARY_DOCUMENT_MAGIC``, which only knows -``.hwpx``/``.pdf``/``.png``. Rather than grow that registry for every such +``.docx``/``.hwpx``/``.pdf``/``.png``. Rather than grow that registry for every such format, a file under a declared prefix whose suffix has no magic entry is admitted on the stricter complement of the UTF-8 decode this module already performs for every ordinarily-scanned file: no diff patch available, *and* the @@ -32,8 +32,17 @@ exists to do -- while still admitting genuinely opaque research binaries without maintaining an open-ended magic-byte catalog. A suffix that *does* have a magic entry keeps that entry's existing structural evidence check -(``_is_complete_png``, ``_is_complete_hwpx``, or the raw magic-prefix check for -``.pdf``) even under a declared prefix. +(``_is_complete_png``, ``_is_complete_hwpx``, ``_is_complete_docx``, or the raw +magic-prefix check for ``.pdf``) even under a declared prefix -- so a ``.docx`` +under a declared prefix is now held to the stricter structural proof rather +than the UTF-8 complement, which fails closed in the same direction. + +late-life-anxiety-reanalysis#257 -- structural DOCX admission: a tracked +research manuscript under ``docs/`` was rejected with "is not valid UTF-8" +because ``.docx`` had no magic entry and so reached the ordinary content +scan's UTF-8 decode. ``.docx`` is admitted on ``_is_complete_docx`` container +evidence instead. The artifact is neither relocated nor exempted: an +unreadable, truncated, disguised, or non-WordprocessingML package still fails. """ from __future__ import annotations @@ -76,11 +85,25 @@ # (this org's own "attach the relevant paper PDF" convention) for a reason # that has nothing to do with the Nginx runtime policy this module enforces. BINARY_DOCUMENT_MAGIC = { + ".docx": (b"PK\x03\x04",), ".hwpx": (b"PK\x03\x04",), ".pdf": (b"%PDF-",), ".png": (b"\x89PNG\r\n\x1a\n",), } PNG_SIGNATURE = BINARY_DOCUMENT_MAGIC[".png"][0] +# OOXML WordprocessingML structural admission (late-life-anxiety-reanalysis#257). +# A ``.docx`` is admitted by proving its container structure, never by decoding +# it as text: the exact OPC parts every conforming writer emits, plus the main +# document part's content type, which is what distinguishes a WordprocessingML +# package from an arbitrary ZIP (or an HWPX) renamed to ``.docx``. +DOCX_REQUIRED_PARTS = ("[Content_Types].xml", "_rels/.rels", "word/document.xml") +DOCX_MAIN_DOCUMENT_CONTENT_TYPE = ( + b"application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml" +) +# The content-type declaration is the only part this module reads, and it is +# read bounded: a package whose declaration exceeds this ceiling is rejected +# rather than streamed, so admission cannot be turned into an unbounded read. +MAX_DOCX_CONTENT_TYPES_BYTES = 65_536 SOURCE_TEST_SUFFIXES = frozenset({".py", ".pyi", ".js", ".mjs", ".cjs", ".ts", ".tsx", ".rs"}) LICENSE_NAMES = frozenset({"license", "license.md", "copying", "copyrights", "notice"}) DOCUMENTATION_DIRECTORIES = frozenset({"doc", "docs", "documentation"}) @@ -369,8 +392,8 @@ def _is_binary_documentation_asset(changed: ChangedFile, declared_prefixes: Sequ *declared_prefixes* (issue #2193) is the base-ref-only research/data artifact declaration: it replaces ONLY this function's path-shape test, never the content evidence a caller still confirms below. A file whose - suffix is a recognized ``BINARY_DOCUMENT_MAGIC`` format (``.hwpx``/ - ``.pdf``/``.png``) is admitted under a declared prefix on the exact same + suffix is a recognized ``BINARY_DOCUMENT_MAGIC`` format (``.docx``/ + ``.hwpx``/``.pdf``/``.png``) is admitted under a declared prefix on the exact same format evidence documentation paths already require. A file whose suffix has no magic entry at all (research formats such as ``.xlsx``, ``.sav``, ``.rds``, ``.npz`` have none) can ONLY be admitted through a @@ -606,8 +629,9 @@ def _binary_documentation_evidence_confirms( also omits a patch for a textual diff that exceeds its own rendering limit, well under this module's ``MAX_FILE_BYTES`` content-fetch ceiling. Whenever the file's raw bytes can be fetched at all, this - verifies the declared format's magic prefix instead of trusting - patch-presence alone. Only a file whose content evidently exceeds the + verifies the declared format's structural evidence (``_is_complete_docx`` + for the OOXML manuscript case, ``_is_complete_hwpx``, ``_is_complete_png``) + or magic prefix instead of trusting patch-presence alone. Only a file whose content evidently exceeds the Contents API's size ceiling -- the exact case ``_is_binary_documentation_asset`` exists for, a cited, large research paper -- falls back to trusting the path+suffix convention for oversized PDFs only; every other @@ -636,6 +660,8 @@ def _binary_documentation_evidence_confirms( return _is_complete_png(raw) if suffix == ".hwpx": return _is_complete_hwpx(raw) + if suffix == ".docx": + return _is_complete_docx(raw) if suffix not in BINARY_DOCUMENT_MAGIC: try: raw.decode("utf-8") @@ -645,6 +671,49 @@ def _binary_documentation_evidence_confirms( return raw.startswith(BINARY_DOCUMENT_MAGIC[suffix]) +def _is_complete_docx(raw: bytes) -> bool: + """Confirm a bounded OOXML WordprocessingML container without reading its text. + + Fail-closed structural admission for the research manuscript in + late-life-anxiety-reanalysis#257: require an unprefixed ZIP, its exact end + record, unique members, every part in ``DOCX_REQUIRED_PARTS`` present, + non-empty and unencrypted, and the main document part's content type + declared in a bounded ``[Content_Types].xml``. Unlike HWPX, a conforming + ``.docx`` has no stored ``mimetype`` member and DEFLATEs every part, so + neither is required here. Anything unreadable, truncated, prefixed, + appended to, encrypted, or declaring a different document type returns + ``False`` -- the caller then scans the bytes as it always did, which for + genuinely binary content fails the policy rather than skipping it. + """ + if not raw.startswith(BINARY_DOCUMENT_MAGIC[".docx"][0]): + return False + try: + with zipfile.ZipFile(io.BytesIO(raw)) as archive: + archive_entries = archive.infolist() + member_names = [member_info.filename for member_info in archive_entries] + end_offset = len(raw) - 22 - len(archive.comment) + if end_offset < 0 or raw[end_offset:end_offset + 4] != b"PK\x05\x06": + return False + if int.from_bytes(raw[end_offset + 20:end_offset + 22], "little") != len(archive.comment): + return False + if not archive_entries or archive_entries[0].header_offset != 0: + return False + if len(member_names) != len(set(member_names)): + return False + for part_name in DOCX_REQUIRED_PARTS: + part_info = archive.getinfo(part_name) + if part_info.flag_bits & 1 or part_info.file_size == 0: + return False + content_types_info = archive.getinfo(DOCX_REQUIRED_PARTS[0]) + if content_types_info.file_size > MAX_DOCX_CONTENT_TYPES_BYTES: + return False + with archive.open(content_types_info) as content_types_stream: + declaration = content_types_stream.read(MAX_DOCX_CONTENT_TYPES_BYTES) + return DOCX_MAIN_DOCUMENT_CONTENT_TYPE in declaration + except (KeyError, UnicodeError, OSError, ValueError, NotImplementedError, zipfile.BadZipFile): + return False + + def _is_complete_hwpx(raw: bytes) -> bool: """Confirm a bounded HWPX container without extracting document content. diff --git a/tests/test_pingora_docx_evidence.py b/tests/test_pingora_docx_evidence.py new file mode 100644 index 0000000000..a324dd7599 --- /dev/null +++ b/tests/test_pingora_docx_evidence.py @@ -0,0 +1,136 @@ +"""Exercise DOCX structural admission through the production policy boundary offline.""" +import io +import zipfile + +import pytest +from tests.test_pingora_edge_policy import policy +from tests.test_pingora_hwpx_evidence import RUNTIME_BYTES, RUNTIME_TEXT, evaluate_bytes + +# The real research artifact from late-life-anxiety-reanalysis#257 (exact head +# 706a81e5a6f88ad74544ab9cf89d4da2b9e6a44d) sits at this path. It is a +# DEFLATE-compressed OOXML package whose first member is [Content_Types].xml, +# with no ZIP comment and no prefixed or appended bytes. The fixture below +# reproduces exactly that structure without copying another repository's +# artifact into this one. +REAL_ARTIFACT_PATH = ( + "docs/delivery_interim_20260920/" + "air_render_00cdc51_g7integration_20260921_205341/manuscript_interim_20260921.docx" +) +MAIN_DOCUMENT_CONTENT_TYPE = ( + b"application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml" +) +CONTENT_TYPES_XML = ( + b'' + b'' + b'' + b'' + b"" +) +SPREADSHEET_CONTENT_TYPES_XML = CONTENT_TYPES_XML.replace( + MAIN_DOCUMENT_CONTENT_TYPE, + b"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml", +) + + +def docx_archive( + *, + content_types=CONTENT_TYPES_XML, + package_rels=b'', + document=b'', + compression_type=zipfile.ZIP_DEFLATED, + duplicate_document=False, +): + """Create a deterministic, non-sensitive OOXML boundary fixture.""" + archive_buffer = io.BytesIO() + members = [ + ("[Content_Types].xml", content_types), + ("_rels/.rels", package_rels), + ("word/document.xml", document), + ] + if duplicate_document: + members.append(("word/document.xml", document)) + with zipfile.ZipFile(archive_buffer, "w") as archive_file: + for member_name, member_value in members: + if member_value is None: + continue + member_info = zipfile.ZipInfo(member_name) + member_info.compress_type = compression_type + archive_file.writestr(member_info, member_value) + return archive_buffer.getvalue() + + +@pytest.mark.parametrize("file_path", [REAL_ARTIFACT_PATH, "docs/manuscript.docx", "Docs/MANUSCRIPT.DOCX"]) +def test_valid_docx_is_admitted_at_the_real_research_artifact_path(file_path): + """Admit the #257 manuscript by structure, without moving or excluding it.""" + assert evaluate_bytes(file_path, docx_archive()) == () + + +@pytest.mark.parametrize("compression_type", [zipfile.ZIP_STORED, zipfile.ZIP_DEFLATED]) +def test_valid_docx_is_admitted_for_either_stored_or_deflated_parts(compression_type): + """Real writers DEFLATE every part; stored parts are equally structural.""" + assert evaluate_bytes("docs/manuscript.docx", docx_archive(compression_type=compression_type)) == () + + +@pytest.mark.parametrize("file_bytes", [ + b"PK\x03\x04\xff", + docx_archive()[:-10], + b"#!/bin/sh\n" + RUNTIME_BYTES + docx_archive(), + docx_archive() + b"\n" + RUNTIME_BYTES, + docx_archive()[:-2] + b"\x01\x00", + docx_archive(content_types=None), + docx_archive(package_rels=None), + docx_archive(document=None), + docx_archive(document=b""), + docx_archive(content_types=SPREADSHEET_CONTENT_TYPES_XML), + docx_archive(duplicate_document=True), +]) +def test_corrupt_or_disguised_docx_container_is_not_exempt(file_bytes): + """Unreadable, incomplete, or non-WordprocessingML input still fails closed.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", file_bytes) + + +def test_docx_rejects_an_encrypted_or_entryless_package(): + """An encrypted part or an empty central directory is unverifiable, so it fails.""" + encrypted_bytes = bytearray(docx_archive()) + encrypted_bytes[encrypted_bytes.find(b"PK\x01\x02") + 8] |= 1 + entryless_bytes = bytearray(docx_archive()) + end_offset = len(entryless_bytes) - 22 + entryless_bytes[end_offset + 8:end_offset + 12] = b"\x00\x00\x00\x00" + entryless_bytes[end_offset + 12:end_offset + 16] = b"\x00\x00\x00\x00" + entryless_bytes[end_offset + 16:end_offset + 20] = end_offset.to_bytes(4, "little") + for file_bytes in (bytes(encrypted_bytes), bytes(entryless_bytes)): + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", file_bytes) + + +def test_oversized_content_types_declaration_is_not_read(): + """A declaration beyond the bounded read ceiling is rejected, never streamed.""" + padding = b"" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(content_types=CONTENT_TYPES_XML + padding)) + + +@pytest.mark.parametrize("file_path", ["docs/manuscript.docx", "local/model.zip"]) +def test_non_utf8_bytes_without_a_structural_document_still_fail(file_path): + """Opaque bytes that are not a recognized structural document are never admitted.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes(file_path, b"\x00\x01\x02\xff\xfe" + RUNTIME_BYTES) + + +@pytest.mark.parametrize("patch_value", [None, "+" + RUNTIME_TEXT.rstrip("\n")]) +def test_valid_utf8_named_docx_keeps_todays_runtime_scan(patch_value): + """A file that decodes as UTF-8 is never treated as a binary artifact.""" + violations = evaluate_bytes("docs/manuscript.docx", RUNTIME_BYTES, patch_value=patch_value) + assert [violation.rule for violation in violations] == ["nginx_runtime_path"] + + +def test_docx_in_runtime_path_remains_unavailable(): + """A structural format exception cannot exempt an active runtime location.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/nginx/manuscript.docx", docx_archive()) + + +def test_removed_docx_does_not_load_deleted_content(): + """Deletion does not require unavailable final-head bytes.""" + assert evaluate_bytes("docs/manuscript.docx", b"", file_status="removed") == () From 1091ca79d30f8939c45d080e3cbe7f2ac7d9fce6 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 24 Sep 2026 00:12:04 +0900 Subject: [PATCH 2/5] test(pingora): cover the magic-prefixed shifted DOCX container The existing disguise case prefixes a shell script, which is rejected at the `PK\x03\x04` magic check before `_is_complete_docx` ever evaluates `archive_entries[0].header_offset != 0`. Mirror the HWPX suite's `b"PK\x03\x04" + archive_bytes` case so the claim that a prefixed container is blocked is actually exercised: zipfile opens that file with a non-zero concat offset, so admission must fail on the member offset, not on the magic prefix. Refs: ContextualWisdomLab/late-life-anxiety-reanalysis#257 Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01FvosNg4GVUjaV5UfrimrsX --- tests/test_pingora_docx_evidence.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_pingora_docx_evidence.py b/tests/test_pingora_docx_evidence.py index a324dd7599..f7eccffab3 100644 --- a/tests/test_pingora_docx_evidence.py +++ b/tests/test_pingora_docx_evidence.py @@ -75,6 +75,7 @@ def test_valid_docx_is_admitted_for_either_stored_or_deflated_parts(compression_ b"PK\x03\x04\xff", docx_archive()[:-10], b"#!/bin/sh\n" + RUNTIME_BYTES + docx_archive(), + b"PK\x03\x04" + docx_archive(), docx_archive() + b"\n" + RUNTIME_BYTES, docx_archive()[:-2] + b"\x01\x00", docx_archive(content_types=None), From e1d04ff235bdb41195c2c4e6c77be89ef3be0622 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 24 Sep 2026 00:36:55 +0900 Subject: [PATCH 3/5] fix(pingora): prove DOCX structure by parsed OPC parts, not a MIME substring The first cut ended `_is_complete_docx` with `DOCX_MAIN_DOCUMENT_CONTENT_TYPE in declaration` -- a raw-bytes substring test on `[Content_Types].xml`. Reproduced bypass, 578 bytes: a ZIP with exactly the three `DOCX_REQUIRED_PARTS` where the expected MIME appears only inside an XML comment, the real `Override` for `/word/document.xml` declares `application/octet-stream`, `_rels/.rels` is `not xml at all` and `word/document.xml` is `also not xml`, was admitted and so skipped the content scan. Replace the substring test with bounded structural parsing: - `_docx_part_elements` reads each required part bounded (the read *is* the expansion bound; a declared ZIP size is never trusted), decodes strict UTF-8, refuses U+0000 and the ` Claude-Session: https://claude.ai/code/session_01FvosNg4GVUjaV5UfrimrsX --- scripts/ci/pingora_edge_policy.py | 155 +++++++++++++++-- tests/test_pingora_docx_evidence.py | 249 ++++++++++++++++++++++++---- 2 files changed, 353 insertions(+), 51 deletions(-) diff --git a/scripts/ci/pingora_edge_policy.py b/scripts/ci/pingora_edge_policy.py index 9c6750aee3..258c57c167 100644 --- a/scripts/ci/pingora_edge_policy.py +++ b/scripts/ci/pingora_edge_policy.py @@ -41,7 +41,9 @@ research manuscript under ``docs/`` was rejected with "is not valid UTF-8" because ``.docx`` had no magic entry and so reached the ordinary content scan's UTF-8 decode. ``.docx`` is admitted on ``_is_complete_docx`` container -evidence instead. The artifact is neither relocated nor exempted: an +evidence instead: the OPC parts, an exact main-document content-type +``Override``, and an agreeing ``officeDocument`` relationship, each read under +byte bounds and parsed as DTD-free XML. The artifact is neither relocated nor exempted: an unreadable, truncated, disguised, or non-WordprocessingML package still fails. """ @@ -62,6 +64,7 @@ from urllib.error import HTTPError, URLError from urllib.parse import quote, urlsplit from urllib.request import HTTPRedirectHandler, Request, build_opener +from xml.parsers import expat MAX_FILE_BYTES = 1_048_576 MAX_RESPONSE_BYTES = 16_777_216 @@ -97,13 +100,28 @@ # document part's content type, which is what distinguishes a WordprocessingML # package from an arbitrary ZIP (or an HWPX) renamed to ``.docx``. DOCX_REQUIRED_PARTS = ("[Content_Types].xml", "_rels/.rels", "word/document.xml") +DOCX_MAIN_DOCUMENT_PART = "word/document.xml" +DOCX_MAIN_DOCUMENT_PART_NAME = "/word/document.xml" DOCX_MAIN_DOCUMENT_CONTENT_TYPE = ( - b"application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml" + "application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml" ) -# The content-type declaration is the only part this module reads, and it is -# read bounded: a package whose declaration exceeds this ceiling is rejected -# rather than streamed, so admission cannot be turned into an unbounded read. +DOCX_CONTENT_TYPES_NS = "http://schemas.openxmlformats.org/package/2006/content-types" +DOCX_PACKAGE_RELATIONSHIPS_NS = "http://schemas.openxmlformats.org/package/2006/relationships" +DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPE = ( + "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" +) +# Every required part is read bounded, and the bounded read *is* the expansion +# bound: a part whose decompressed content exceeds its ceiling is rejected +# rather than streamed, so admission cannot be turned into an unbounded read +# or a decompression bomb. A declared ZIP size is never trusted for this -- +# only the length actually read is. MAX_DOCX_CONTENT_TYPES_BYTES = 65_536 +MAX_DOCX_MAIN_PART_BYTES = 8 * 1024 * 1024 +DOCX_PART_CEILINGS = { + "[Content_Types].xml": MAX_DOCX_CONTENT_TYPES_BYTES, + "_rels/.rels": MAX_DOCX_CONTENT_TYPES_BYTES, + "word/document.xml": MAX_DOCX_MAIN_PART_BYTES, +} SOURCE_TEST_SUFFIXES = frozenset({".py", ".pyi", ".js", ".mjs", ".cjs", ".ts", ".tsx", ".rs"}) LICENSE_NAMES = frozenset({"license", "license.md", "copying", "copyrights", "notice"}) DOCUMENTATION_DIRECTORIES = frozenset({"doc", "docs", "documentation"}) @@ -676,9 +694,15 @@ def _is_complete_docx(raw: bytes) -> bool: Fail-closed structural admission for the research manuscript in late-life-anxiety-reanalysis#257: require an unprefixed ZIP, its exact end - record, unique members, every part in ``DOCX_REQUIRED_PARTS`` present, - non-empty and unencrypted, and the main document part's content type - declared in a bounded ``[Content_Types].xml``. Unlike HWPX, a conforming + record, unique members, and every part in ``DOCX_REQUIRED_PARTS`` present, + non-empty, unencrypted and parseable as bounded, DTD-free XML + (``_docx_part_elements``). The main document part must then be declared as + an exact ``Override`` pairing ``/word/document.xml`` with + ``DOCX_MAIN_DOCUMENT_CONTENT_TYPE`` (``_docx_declares_main_document``) and + agreed by the package's single internal ``officeDocument`` relationship + (``_docx_relates_main_document``). A substring of the expected MIME + anywhere in the declaration -- in a comment, or in an unrelated attribute + -- is not a declaration and does not admit. Unlike HWPX, a conforming ``.docx`` has no stored ``mimetype`` member and DEFLATEs every part, so neither is required here. Anything unreadable, truncated, prefixed, appended to, encrypted, or declaring a different document type returns @@ -700,20 +724,121 @@ def _is_complete_docx(raw: bytes) -> bool: return False if len(member_names) != len(set(member_names)): return False + parsed_parts: dict[str, tuple[tuple[str, Mapping[str, str]], ...]] = {} for part_name in DOCX_REQUIRED_PARTS: part_info = archive.getinfo(part_name) if part_info.flag_bits & 1 or part_info.file_size == 0: return False - content_types_info = archive.getinfo(DOCX_REQUIRED_PARTS[0]) - if content_types_info.file_size > MAX_DOCX_CONTENT_TYPES_BYTES: - return False - with archive.open(content_types_info) as content_types_stream: - declaration = content_types_stream.read(MAX_DOCX_CONTENT_TYPES_BYTES) - return DOCX_MAIN_DOCUMENT_CONTENT_TYPE in declaration - except (KeyError, UnicodeError, OSError, ValueError, NotImplementedError, zipfile.BadZipFile): + elements = _docx_part_elements(archive, part_info, DOCX_PART_CEILINGS[part_name]) + if not elements: + return False + parsed_parts[part_name] = elements + return _docx_declares_main_document( + parsed_parts[DOCX_REQUIRED_PARTS[0]] + ) and _docx_relates_main_document(parsed_parts[DOCX_REQUIRED_PARTS[1]]) + except ( + KeyError, + UnicodeError, + OSError, + ValueError, + NotImplementedError, + zipfile.BadZipFile, + zlib.error, + ): return False +def _docx_part_elements( + archive: zipfile.ZipFile, part_info: zipfile.ZipInfo, ceiling: int +) -> tuple[tuple[str, Mapping[str, str]], ...] | None: + """Return one bounded OOXML part's elements, or ``None`` if it is not safe XML. + + The gate's runtime installs nothing -- the required workflow runs this + module with the runner's stock ``python3`` and no ``pip install`` step + (``.github/workflows/opencode-review.yml``'s ``required-workflow-bootstrap`` + job) -- so ``defusedxml`` is not importable here and a module-level import + of it would break the required check in every consumer repository. The + equivalent guarantee is reconstructed on the standard library's expat + parser instead: the bytes are read bounded (the read *is* the expansion + bound; a declared ZIP size is never trusted), decoded as strict UTF-8, and + refused outright if they contain U+0000 -- which rejects every UTF-16/UCS-4 + encoding of the checks below -- or the `` ceiling: + return None + try: + text = data.decode("utf-8") + except UnicodeDecodeError: + return None + if "\x00" in text or " bool: + """Return whether ``[Content_Types].xml`` declares exactly the main document part. + + Requires the OPC content-types root element and exactly one ``Override`` + naming ``/word/document.xml``, whose ``ContentType`` is exactly + ``DOCX_MAIN_DOCUMENT_CONTENT_TYPE``. A ``Default`` extension mapping, a + comment, an ambiguous pair of Overrides for the same part, or any other + declared content type does not satisfy this. + """ + + if elements[0][0] != f"{DOCX_CONTENT_TYPES_NS} Types": + return False + declared = [ + attributes.get("ContentType") + for name, attributes in elements + if name == f"{DOCX_CONTENT_TYPES_NS} Override" + and attributes.get("PartName") == DOCX_MAIN_DOCUMENT_PART_NAME + ] + return declared == [DOCX_MAIN_DOCUMENT_CONTENT_TYPE] + + +def _docx_relates_main_document( + elements: Sequence[tuple[str, Mapping[str, str]]] +) -> bool: + """Return whether ``_rels/.rels`` points its package root at that same part. + + Requires the OPC relationships root element and exactly one relationship of + the ``officeDocument`` type, internal, whose ``Target`` is exactly the main + document part in either permitted spelling. Exact matching is what rejects + a traversal, an absolute or an external target: nothing is resolved or + normalized, so no target outside the package can agree. + """ + + if elements[0][0] != f"{DOCX_PACKAGE_RELATIONSHIPS_NS} Relationships": + return False + roots = [ + (attributes.get("Target"), attributes.get("TargetMode", "Internal")) + for name, attributes in elements + if name == f"{DOCX_PACKAGE_RELATIONSHIPS_NS} Relationship" + and attributes.get("Type") == DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPE + ] + return roots in ( + [(DOCX_MAIN_DOCUMENT_PART, "Internal")], + [(DOCX_MAIN_DOCUMENT_PART_NAME, "Internal")], + ) + + def _is_complete_hwpx(raw: bytes) -> bool: """Confirm a bounded HWPX container without extracting document content. diff --git a/tests/test_pingora_docx_evidence.py b/tests/test_pingora_docx_evidence.py index f7eccffab3..96271bf2ae 100644 --- a/tests/test_pingora_docx_evidence.py +++ b/tests/test_pingora_docx_evidence.py @@ -1,5 +1,6 @@ """Exercise DOCX structural admission through the production policy boundary offline.""" import io +import os import zipfile import pytest @@ -9,49 +10,101 @@ # The real research artifact from late-life-anxiety-reanalysis#257 (exact head # 706a81e5a6f88ad74544ab9cf89d4da2b9e6a44d) sits at this path. It is a # DEFLATE-compressed OOXML package whose first member is [Content_Types].xml, -# with no ZIP comment and no prefixed or appended bytes. The fixture below -# reproduces exactly that structure without copying another repository's -# artifact into this one. +# with no ZIP comment and no prefixed or appended bytes. The fixtures below +# reproduce that structure without copying another repository's artifact into +# this one; test_real_artifact_bytes_are_admitted binds the same assertion to +# the actual bytes when PINGORA_REAL_DOCX_PATH names a local copy. REAL_ARTIFACT_PATH = ( "docs/delivery_interim_20260920/" "air_render_00cdc51_g7integration_20260921_205341/manuscript_interim_20260921.docx" ) -MAIN_DOCUMENT_CONTENT_TYPE = ( - b"application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml" -) -CONTENT_TYPES_XML = ( - b'' - b'' - b'' - b'' - b"" -) -SPREADSHEET_CONTENT_TYPES_XML = CONTENT_TYPES_XML.replace( - MAIN_DOCUMENT_CONTENT_TYPE, - b"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml", +REAL_ARTIFACT_ENV = "PINGORA_REAL_DOCX_PATH" +CONTENT_TYPES_NS = "http://schemas.openxmlformats.org/package/2006/content-types" +PACKAGE_RELATIONSHIPS_NS = "http://schemas.openxmlformats.org/package/2006/relationships" +OFFICE_DOCUMENT_RELATIONSHIP_TYPE = ( + "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" ) +MAIN_DOCUMENT_CONTENT_TYPE = policy.DOCX_MAIN_DOCUMENT_CONTENT_TYPE +WORD_MAIN_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" +STRICT_WORD_MAIN_NS = "http://purl.oclc.org/ooxml/wordprocessingml/main" +XML_DECLARATION = '' +# Distinguishes "omit this part entirely" from "use the conforming default". +OMITTED = object() + + +def content_types_xml( + *, + part_name="/word/document.xml", + content_type=MAIN_DOCUMENT_CONTENT_TYPE, + overrides=1, + default_content_type=None, + root_tag="Types", + comment="", +): + """Render one [Content_Types].xml declaration with an exact Override shape.""" + default_extension = ( + f'' + if default_content_type is not None + else '' + ) + override = f'' + # An unrelated Override always accompanies the main one, so the part-name + # filter is exercised in both directions by every positive fixture. + unrelated = ( + '' + ) + body = default_extension + unrelated + comment + override * overrides + return f"{XML_DECLARATION}<{root_tag} xmlns=\"{CONTENT_TYPES_NS}\">{body}".encode() + + +def package_rels_xml( + *, + target="word/document.xml", + target_mode=None, + relationship_type=OFFICE_DOCUMENT_RELATIONSHIP_TYPE, + relationships=1, + root_tag="Relationships", + padding="", +): + """Render one _rels/.rels package relationship part.""" + mode = f' TargetMode="{target_mode}"' if target_mode is not None else "" + main = f'' + # A real package always carries unrelated relationships alongside the + # officeDocument one, so the relationship-type filter is exercised both ways. + unrelated = ( + '' + ) + body = unrelated + main * relationships + padding + return f"{XML_DECLARATION}<{root_tag} xmlns=\"{PACKAGE_RELATIONSHIPS_NS}\">{body}".encode() + + +def document_xml(*, namespace=WORD_MAIN_NS): + """Render one minimal but well-formed WordprocessingML main document part.""" + return f'{XML_DECLARATION}'.encode() def docx_archive( *, - content_types=CONTENT_TYPES_XML, - package_rels=b'', - document=b'', + content_types=None, + package_rels=None, + document=None, compression_type=zipfile.ZIP_DEFLATED, duplicate_document=False, ): """Create a deterministic, non-sensitive OOXML boundary fixture.""" archive_buffer = io.BytesIO() members = [ - ("[Content_Types].xml", content_types), - ("_rels/.rels", package_rels), - ("word/document.xml", document), + ("[Content_Types].xml", content_types_xml() if content_types is None else content_types), + ("_rels/.rels", package_rels_xml() if package_rels is None else package_rels), + ("word/document.xml", document_xml() if document is None else document), ] if duplicate_document: - members.append(("word/document.xml", document)) + members.append(("word/document.xml", document_xml())) with zipfile.ZipFile(archive_buffer, "w") as archive_file: for member_name, member_value in members: - if member_value is None: + if member_value is OMITTED: continue member_info = zipfile.ZipInfo(member_name) member_info.compress_type = compression_type @@ -59,6 +112,27 @@ def docx_archive( return archive_buffer.getvalue() +def comment_only_mime_counterexample(): + """Build the reported bypass: the MIME only appears inside an XML comment.""" + return docx_archive( + content_types=content_types_xml( + content_type="application/octet-stream", + comment=f"", + ), + package_rels=b"not xml at all", + document=b"also not xml", + ) + + +def deflate_damaged(file_bytes): + """Corrupt the first compressed stream without touching the ZIP structure.""" + damaged = bytearray(file_bytes) + data_offset = damaged.find(b"[Content_Types].xml") + len(b"[Content_Types].xml") + for index in range(data_offset, data_offset + 16): + damaged[index] ^= 0xFF + return bytes(damaged) + + @pytest.mark.parametrize("file_path", [REAL_ARTIFACT_PATH, "docs/manuscript.docx", "Docs/MANUSCRIPT.DOCX"]) def test_valid_docx_is_admitted_at_the_real_research_artifact_path(file_path): """Admit the #257 manuscript by structure, without moving or excluding it.""" @@ -71,6 +145,32 @@ def test_valid_docx_is_admitted_for_either_stored_or_deflated_parts(compression_ assert evaluate_bytes("docs/manuscript.docx", docx_archive(compression_type=compression_type)) == () +@pytest.mark.parametrize("target", ["word/document.xml", "/word/document.xml"]) +@pytest.mark.parametrize("target_mode", [None, "Internal"]) +def test_valid_docx_accepts_either_relationship_target_spelling(target, target_mode): + """OPC permits both target spellings and an explicit Internal target mode.""" + archive_bytes = docx_archive(package_rels=package_rels_xml(target=target, target_mode=target_mode)) + assert evaluate_bytes("docs/manuscript.docx", archive_bytes) == () + + +def test_valid_docx_accepts_the_iso_strict_main_document_namespace(): + """ISO 29500 Strict manuscripts carry the same content type and must pass.""" + archive_bytes = docx_archive(document=document_xml(namespace=STRICT_WORD_MAIN_NS)) + assert evaluate_bytes("docs/manuscript.docx", archive_bytes) == () + + +@pytest.mark.skipif( + not os.environ.get(REAL_ARTIFACT_ENV), + reason=f"{REAL_ARTIFACT_ENV} must name a local copy of the #257 manuscript", +) +def test_real_artifact_bytes_are_admitted(): + """Bind admission to the actual exact-head bytes, not only to the fixture.""" + with open(os.environ[REAL_ARTIFACT_ENV], "rb") as artifact_file: + raw = artifact_file.read() + assert raw.startswith(b"PK\x03\x04") + assert evaluate_bytes(REAL_ARTIFACT_PATH, raw) == () + + @pytest.mark.parametrize("file_bytes", [ b"PK\x03\x04\xff", docx_archive()[:-10], @@ -78,19 +178,103 @@ def test_valid_docx_is_admitted_for_either_stored_or_deflated_parts(compression_ b"PK\x03\x04" + docx_archive(), docx_archive() + b"\n" + RUNTIME_BYTES, docx_archive()[:-2] + b"\x01\x00", - docx_archive(content_types=None), - docx_archive(package_rels=None), - docx_archive(document=None), + docx_archive(content_types=b""), + docx_archive(package_rels=b""), docx_archive(document=b""), - docx_archive(content_types=SPREADSHEET_CONTENT_TYPES_XML), docx_archive(duplicate_document=True), + deflate_damaged(docx_archive()), + comment_only_mime_counterexample(), ]) def test_corrupt_or_disguised_docx_container_is_not_exempt(file_bytes): - """Unreadable, incomplete, or non-WordprocessingML input still fails closed.""" + """Unreadable, incomplete, or non-structural input still fails closed.""" with pytest.raises(policy.PolicyError): evaluate_bytes("docs/manuscript.docx", file_bytes) +@pytest.mark.parametrize("missing_part", ["content_types", "package_rels", "document"]) +def test_docx_missing_a_required_opc_part_is_not_exempt(missing_part): + """Every part in DOCX_REQUIRED_PARTS must actually be present.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(**{missing_part: OMITTED})) + + +@pytest.mark.parametrize("content_types", [ + # The expected MIME as a Default extension mapping, never as the Override. + content_types_xml(content_type="application/octet-stream", default_content_type=MAIN_DOCUMENT_CONTENT_TYPE), + # A spreadsheet package renamed to .docx. + content_types_xml( + content_type="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml" + ), + # No Override for the main document part at all. + content_types_xml(overrides=0), + # Two Overrides for the same part name: an ambiguous declaration. + content_types_xml(overrides=2), + # The Override names a different part. + content_types_xml(part_name="/word/other.xml"), + # Correct Override, wrong root element. + content_types_xml(root_tag="Relationships"), +]) +def test_docx_requires_an_exact_main_document_content_type_override(content_types): + """Only an exact Override pairing the main part with its MIME admits.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(content_types=content_types)) + + +@pytest.mark.parametrize("package_rels", [ + # The officeDocument relationship points somewhere else entirely. + package_rels_xml(target="word/other.xml"), + # A path traversal target is rejected by exact matching. + package_rels_xml(target="../../etc/passwd"), + # An external target is not the packaged main document. + package_rels_xml(target="https://example.invalid/document.xml", target_mode="External"), + # No officeDocument relationship at all. + package_rels_xml(relationships=0), + # Two officeDocument relationships: an ambiguous package root. + package_rels_xml(relationships=2), + # A relationship of a different type cannot stand in for it. + package_rels_xml(relationship_type="http://schemas.openxmlformats.org/package/2006/relationships/x"), + # Correct relationship, wrong root element. + package_rels_xml(root_tag="Types"), +]) +def test_docx_requires_the_office_document_relationship_to_agree(package_rels): + """The package root relationship must resolve to the checked main part.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(package_rels=package_rels)) + + +@pytest.mark.parametrize("part_bytes", [ + b"not xml at all", + XML_DECLARATION.encode() + b"", + b"\xff\xfe", + XML_DECLARATION.encode() + b']>', + (']>').encode("utf-16-le"), + XML_DECLARATION.encode() + b"&undefined;", +]) +@pytest.mark.parametrize("part_name", ["content_types", "package_rels", "document"]) +def test_docx_requires_well_formed_entity_free_xml_parts(part_bytes, part_name): + """Each required part must be well-formed UTF-8 XML with no DTD or entity.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(**{part_name: part_bytes})) + + +@pytest.mark.parametrize("part_name, padding_length", [ + ("content_types", policy.MAX_DOCX_CONTENT_TYPES_BYTES + 1), + ("package_rels", policy.MAX_DOCX_CONTENT_TYPES_BYTES + 1), + ("document", policy.MAX_DOCX_MAIN_PART_BYTES + 1), +]) +def test_oversized_docx_part_is_not_read_beyond_its_ceiling(part_name, padding_length): + """A part beyond its bounded read ceiling is rejected, never streamed.""" + padding = "" + if part_name == "content_types": + part_bytes = content_types_xml(comment=padding) + elif part_name == "package_rels": + part_bytes = package_rels_xml(padding=padding) + else: + part_bytes = document_xml()[:-len(b"")] + padding.encode() + b"" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(**{part_name: part_bytes})) + + def test_docx_rejects_an_encrypted_or_entryless_package(): """An encrypted part or an empty central directory is unverifiable, so it fails.""" encrypted_bytes = bytearray(docx_archive()) @@ -105,13 +289,6 @@ def test_docx_rejects_an_encrypted_or_entryless_package(): evaluate_bytes("docs/manuscript.docx", file_bytes) -def test_oversized_content_types_declaration_is_not_read(): - """A declaration beyond the bounded read ceiling is rejected, never streamed.""" - padding = b"" - with pytest.raises(policy.PolicyError): - evaluate_bytes("docs/manuscript.docx", docx_archive(content_types=CONTENT_TYPES_XML + padding)) - - @pytest.mark.parametrize("file_path", ["docs/manuscript.docx", "local/model.zip"]) def test_non_utf8_bytes_without_a_structural_document_still_fail(file_path): """Opaque bytes that are not a recognized structural document are never admitted.""" From dbbeccd69718d19f61bafc4d74b88d12c8c0ae4b Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Mon, 28 Sep 2026 21:49:15 +0900 Subject: [PATCH 4/5] test(pingora): confine the duplicate DOCX part warning to its fixture The rejection case writes word/document.xml twice on purpose. zipfile warns on that write; keep the warning inside the fixture so the suite stays quiet. --- tests/test_pingora_docx_evidence.py | 24 +++++++++++++++++------- 1 file changed, 17 insertions(+), 7 deletions(-) diff --git a/tests/test_pingora_docx_evidence.py b/tests/test_pingora_docx_evidence.py index 96271bf2ae..0360c4ef15 100644 --- a/tests/test_pingora_docx_evidence.py +++ b/tests/test_pingora_docx_evidence.py @@ -1,6 +1,7 @@ """Exercise DOCX structural admission through the production policy boundary offline.""" import io import os +import warnings import zipfile import pytest @@ -102,13 +103,22 @@ def docx_archive( ] if duplicate_document: members.append(("word/document.xml", document_xml())) - with zipfile.ZipFile(archive_buffer, "w") as archive_file: - for member_name, member_value in members: - if member_value is OMITTED: - continue - member_info = zipfile.ZipInfo(member_name) - member_info.compress_type = compression_type - archive_file.writestr(member_info, member_value) + with warnings.catch_warnings(): + if duplicate_document: + # The duplicate-member rejection fixture writes the same part twice. + # zipfile warns on that write; the warning is the fixture, not a defect. + warnings.filterwarnings( + "ignore", + message=r"Duplicate name: 'word/document.xml'", + category=UserWarning, + ) + with zipfile.ZipFile(archive_buffer, "w") as archive_file: + for member_name, member_value in members: + if member_value is OMITTED: + continue + member_info = zipfile.ZipInfo(member_name) + member_info.compress_type = compression_type + archive_file.writestr(member_info, member_value) return archive_buffer.getvalue() From a3230454894dcf8d256abab449d0214465bd5ebe Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Mon, 28 Sep 2026 21:55:08 +0900 Subject: [PATCH 5/5] fix(pingora): admit ISO 29500 Strict officeDocument relationships Word's Strict Open XML save and the Library of Congress description of ISO/IEC 29500-1 name the same main part with two purl.oclc.org relationship types. Exactly one of those, or the transitional type, may agree. --- scripts/ci/pingora_edge_policy.py | 22 ++++++++++++++++---- tests/test_pingora_docx_evidence.py | 31 +++++++++++++++++++++++++++++ 2 files changed, 49 insertions(+), 4 deletions(-) diff --git a/scripts/ci/pingora_edge_policy.py b/scripts/ci/pingora_edge_policy.py index 135509dd79..33133f2422 100644 --- a/scripts/ci/pingora_edge_policy.py +++ b/scripts/ci/pingora_edge_policy.py @@ -113,6 +113,17 @@ DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPE = ( "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" ) +# ISO/IEC 29500 Strict names the same officeDocument part. Library of Congress +# fdd000400 cites the first URI from ISO/IEC 29500-1:2012 §11.3.10. Microsoft +# Word's Strict Open XML save writes the second (python-openxml/python-docx#693). +DOCX_STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES = frozenset({ + "http://purl.oclc.org/ooxml/relationships/officeDocument", + "http://purl.oclc.org/ooxml/officeDocument/relationships/officeDocument", +}) +DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPES = frozenset({ + DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPE, + *DOCX_STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES, +}) # Every required part is read bounded, and the bounded read *is* the expansion # bound: a part whose decompressed content exceeds its ceiling is rejected # rather than streamed, so admission cannot be turned into an unbounded read @@ -872,9 +883,12 @@ def _docx_relates_main_document( Requires the OPC relationships root element and exactly one relationship of the ``officeDocument`` type, internal, whose ``Target`` is exactly the main - document part in either permitted spelling. Exact matching is what rejects - a traversal, an absolute or an external target: nothing is resolved or - normalized, so no target outside the package can agree. + document part in either permitted spelling. Transitional packages use + ``DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPE``; ISO/IEC 29500 Strict packages + use one URI in ``DOCX_STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES``. One of + each is two relationships and does not agree. Exact matching is what + rejects a traversal, an absolute or an external target: nothing is resolved + or normalized, so no target outside the package can agree. """ if elements[0][0] != f"{DOCX_PACKAGE_RELATIONSHIPS_NS} Relationships": @@ -883,7 +897,7 @@ def _docx_relates_main_document( (attributes.get("Target"), attributes.get("TargetMode", "Internal")) for name, attributes in elements if name == f"{DOCX_PACKAGE_RELATIONSHIPS_NS} Relationship" - and attributes.get("Type") == DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPE + and attributes.get("Type") in DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPES ] return roots in ( [(DOCX_MAIN_DOCUMENT_PART, "Internal")], diff --git a/tests/test_pingora_docx_evidence.py b/tests/test_pingora_docx_evidence.py index 0360c4ef15..7fef8e0f96 100644 --- a/tests/test_pingora_docx_evidence.py +++ b/tests/test_pingora_docx_evidence.py @@ -25,6 +25,13 @@ OFFICE_DOCUMENT_RELATIONSHIP_TYPE = ( "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" ) +# Library of Congress fdd000400 cites ISO/IEC 29500-1:2012 §11.3.10 for the +# first spelling. Microsoft Word's Strict Open XML save writes the second +# (python-openxml/python-docx#693). The OPC relationships namespace is unchanged. +STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES = ( + "http://purl.oclc.org/ooxml/relationships/officeDocument", + "http://purl.oclc.org/ooxml/officeDocument/relationships/officeDocument", +) MAIN_DOCUMENT_CONTENT_TYPE = policy.DOCX_MAIN_DOCUMENT_CONTENT_TYPE WORD_MAIN_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" STRICT_WORD_MAIN_NS = "http://purl.oclc.org/ooxml/wordprocessingml/main" @@ -169,6 +176,30 @@ def test_valid_docx_accepts_the_iso_strict_main_document_namespace(): assert evaluate_bytes("docs/manuscript.docx", archive_bytes) == () +@pytest.mark.parametrize("relationship_type", STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES) +def test_valid_docx_accepts_an_iso_strict_office_document_relationship(relationship_type): + """Strict WordprocessingML still points the package root at the main part.""" + archive_bytes = docx_archive( + package_rels=package_rels_xml(relationship_type=relationship_type), + document=document_xml(namespace=STRICT_WORD_MAIN_NS), + ) + assert evaluate_bytes("docs/manuscript.docx", archive_bytes) == () + + +def test_mixed_transitional_and_strict_office_document_relationships_are_ambiguous(): + """Exactly one officeDocument relationship is allowed, across either spelling.""" + rels = ( + f'{XML_DECLARATION}' + f'' + '' + "" + ).encode() + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(package_rels=rels)) + + @pytest.mark.skipif( not os.environ.get(REAL_ARTIFACT_ENV), reason=f"{REAL_ARTIFACT_ENV} must name a local copy of the #257 manuscript",