diff --git a/scripts/ci/pingora_edge_policy.py b/scripts/ci/pingora_edge_policy.py index 0d3a2c0948..33133f2422 100644 --- a/scripts/ci/pingora_edge_policy.py +++ b/scripts/ci/pingora_edge_policy.py @@ -22,7 +22,7 @@ Suffix decision: most research-data formats (``.xlsx``, ``.sav``, ``.rds``, ``.npz``, ...) have no entry in ``BINARY_DOCUMENT_MAGIC``, which only knows -``.hwpx``/``.pdf``/``.png``. Rather than grow that registry for every such +``.docx``/``.hwpx``/``.pdf``/``.png``. Rather than grow that registry for every such format, a file under a declared prefix whose suffix has no magic entry is admitted only when no diff patch is available, the fetched bytes fail to decode as UTF-8, and their replacement-decoded text contains no prohibited runtime @@ -32,8 +32,20 @@ admitting genuinely opaque research binaries without maintaining an open-ended magic-byte catalog. A suffix that *does* have a magic entry keeps that entry's existing structural evidence check -(``_is_complete_png``, ``_is_complete_hwpx``, or the raw magic-prefix check for -``.pdf``) even under a declared prefix. +(``_is_complete_png``, ``_is_complete_hwpx``, ``_is_complete_docx``, or the raw +magic-prefix check for ``.pdf``) even under a declared prefix -- so a ``.docx`` +under a declared prefix is held to the stricter structural proof rather than +the UTF-8 complement, which fails closed in the same direction. + +late-life-anxiety-reanalysis#257 -- structural DOCX admission: a tracked +research manuscript under ``docs/`` was rejected with "is not valid UTF-8" +because ``.docx`` had no magic entry and so reached the ordinary content +scan's UTF-8 decode. ``.docx`` is admitted on ``_is_complete_docx`` container +evidence instead: the OPC parts, an exact main-document content-type +``Override``, and an agreeing ``officeDocument`` relationship, each read under +byte bounds and parsed as DTD-free XML. The artifact is neither relocated nor +exempted: an unreadable, truncated, disguised, or non-WordprocessingML package +still fails. """ from __future__ import annotations @@ -54,6 +66,7 @@ from urllib.error import HTTPError, URLError from urllib.parse import quote, urlsplit from urllib.request import HTTPRedirectHandler, Request, build_opener +from xml.parsers import expat MAX_FILE_BYTES = 1_048_576 MAX_RESPONSE_BYTES = 16_777_216 @@ -78,11 +91,51 @@ # (this org's own "attach the relevant paper PDF" convention) for a reason # that has nothing to do with the Nginx runtime policy this module enforces. BINARY_DOCUMENT_MAGIC = { + ".docx": (b"PK\x03\x04",), ".hwpx": (b"PK\x03\x04",), ".pdf": (b"%PDF-",), ".png": (b"\x89PNG\r\n\x1a\n",), } PNG_SIGNATURE = BINARY_DOCUMENT_MAGIC[".png"][0] +# OOXML WordprocessingML structural admission (late-life-anxiety-reanalysis#257). +# A ``.docx`` is admitted by proving its container structure, never by decoding +# it as text: the exact OPC parts every conforming writer emits, plus the main +# document part's content type, which is what distinguishes a WordprocessingML +# package from an arbitrary ZIP (or an HWPX) renamed to ``.docx``. +DOCX_REQUIRED_PARTS = ("[Content_Types].xml", "_rels/.rels", "word/document.xml") +DOCX_MAIN_DOCUMENT_PART = "word/document.xml" +DOCX_MAIN_DOCUMENT_PART_NAME = "/word/document.xml" +DOCX_MAIN_DOCUMENT_CONTENT_TYPE = ( + "application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml" +) +DOCX_CONTENT_TYPES_NS = "http://schemas.openxmlformats.org/package/2006/content-types" +DOCX_PACKAGE_RELATIONSHIPS_NS = "http://schemas.openxmlformats.org/package/2006/relationships" +DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPE = ( + "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" +) +# ISO/IEC 29500 Strict names the same officeDocument part. Library of Congress +# fdd000400 cites the first URI from ISO/IEC 29500-1:2012 §11.3.10. Microsoft +# Word's Strict Open XML save writes the second (python-openxml/python-docx#693). +DOCX_STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES = frozenset({ + "http://purl.oclc.org/ooxml/relationships/officeDocument", + "http://purl.oclc.org/ooxml/officeDocument/relationships/officeDocument", +}) +DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPES = frozenset({ + DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPE, + *DOCX_STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES, +}) +# Every required part is read bounded, and the bounded read *is* the expansion +# bound: a part whose decompressed content exceeds its ceiling is rejected +# rather than streamed, so admission cannot be turned into an unbounded read +# or a decompression bomb. A declared ZIP size is never trusted for this -- +# only the length actually read is. +MAX_DOCX_CONTENT_TYPES_BYTES = 65_536 +MAX_DOCX_MAIN_PART_BYTES = 8 * 1024 * 1024 +DOCX_PART_CEILINGS = { + "[Content_Types].xml": MAX_DOCX_CONTENT_TYPES_BYTES, + "_rels/.rels": MAX_DOCX_CONTENT_TYPES_BYTES, + "word/document.xml": MAX_DOCX_MAIN_PART_BYTES, +} SOURCE_TEST_SUFFIXES = frozenset({".py", ".pyi", ".js", ".mjs", ".cjs", ".ts", ".tsx", ".rs"}) LICENSE_NAMES = frozenset({"license", "license.md", "copying", "copyrights", "notice"}) DOCUMENTATION_DIRECTORIES = frozenset({"doc", "docs", "documentation"}) @@ -367,8 +420,8 @@ def _is_binary_documentation_asset(changed: ChangedFile, declared_prefixes: Sequ *declared_prefixes* (issue #2193) is the base-ref-only research/data artifact declaration: it replaces ONLY this function's path-shape test, never the content evidence a caller still confirms below. A file whose - suffix is a recognized ``BINARY_DOCUMENT_MAGIC`` format (``.hwpx``/ - ``.pdf``/``.png``) is admitted under a declared prefix on the exact same + suffix is a recognized ``BINARY_DOCUMENT_MAGIC`` format (``.docx``/ + ``.hwpx``/``.pdf``/``.png``) is admitted under a declared prefix on the exact same format evidence documentation paths already require. A file whose suffix has no magic entry at all (research formats such as ``.xlsx``, ``.sav``, ``.rds``, ``.npz`` have none) can ONLY be admitted through a @@ -653,9 +706,11 @@ def _binary_documentation_evidence_confirms( also omits a patch for a textual diff that exceeds its own rendering limit, well under this module's ``MAX_BLOB_BYTES`` content-fetch ceiling. Whenever the file's raw bytes can be fetched at all, this - verifies the declared format's magic prefix instead of trusting - patch-presence alone. Only a file whose content evidently exceeds the - Git blob API's size ceiling -- the exact case ``_is_binary_documentation_asset`` + verifies the declared format's structural evidence (``_is_complete_docx`` + for the OOXML manuscript case, ``_is_complete_hwpx``, ``_is_complete_png``) + or magic prefix instead of trusting patch-presence alone. Only a file + whose content evidently exceeds the Git blob API's size ceiling -- the + exact case ``_is_binary_documentation_asset`` exists for, a cited, large research paper -- falls back to trusting the path+suffix convention for PDFs over 100 MB only; every other content-evidence failure (a @@ -685,6 +740,8 @@ def _binary_documentation_evidence_confirms( return _is_complete_png(raw) if suffix == ".hwpx": return _is_complete_hwpx(raw) + if suffix == ".docx": + return _is_complete_docx(raw) if suffix not in BINARY_DOCUMENT_MAGIC: try: raw.decode("utf-8") @@ -695,6 +752,159 @@ def _binary_documentation_evidence_confirms( return raw.startswith(BINARY_DOCUMENT_MAGIC[suffix]) +def _is_complete_docx(raw: bytes) -> bool: + """Confirm a bounded OOXML WordprocessingML container without reading its text. + + Fail-closed structural admission for the research manuscript in + late-life-anxiety-reanalysis#257: require an unprefixed ZIP, its exact end + record, unique members, and every part in ``DOCX_REQUIRED_PARTS`` present, + non-empty, unencrypted and parseable as bounded, DTD-free XML + (``_docx_part_elements``). The main document part must then be declared as + an exact ``Override`` pairing ``/word/document.xml`` with + ``DOCX_MAIN_DOCUMENT_CONTENT_TYPE`` (``_docx_declares_main_document``) and + agreed by the package's single internal ``officeDocument`` relationship + (``_docx_relates_main_document``). A substring of the expected MIME + anywhere in the declaration -- in a comment, or in an unrelated attribute + -- is not a declaration and does not admit. Unlike HWPX, a conforming + ``.docx`` has no stored ``mimetype`` member and DEFLATEs every part, so + neither is required here. Anything unreadable, truncated, prefixed, + appended to, encrypted, or declaring a different document type returns + ``False`` -- the caller then scans the bytes as it always did, which for + genuinely binary content fails the policy rather than skipping it. + """ + if not raw.startswith(BINARY_DOCUMENT_MAGIC[".docx"][0]): + return False + try: + with zipfile.ZipFile(io.BytesIO(raw)) as archive: + archive_entries = archive.infolist() + member_names = [member_info.filename for member_info in archive_entries] + end_offset = len(raw) - 22 - len(archive.comment) + if end_offset < 0 or raw[end_offset:end_offset + 4] != b"PK\x05\x06": + return False + if int.from_bytes(raw[end_offset + 20:end_offset + 22], "little") != len(archive.comment): + return False + if not archive_entries or archive_entries[0].header_offset != 0: + return False + if len(member_names) != len(set(member_names)): + return False + parsed_parts: dict[str, tuple[tuple[str, Mapping[str, str]], ...]] = {} + for part_name in DOCX_REQUIRED_PARTS: + part_info = archive.getinfo(part_name) + if part_info.flag_bits & 1 or part_info.file_size == 0: + return False + elements = _docx_part_elements(archive, part_info, DOCX_PART_CEILINGS[part_name]) + if not elements: + return False + parsed_parts[part_name] = elements + return _docx_declares_main_document( + parsed_parts[DOCX_REQUIRED_PARTS[0]] + ) and _docx_relates_main_document(parsed_parts[DOCX_REQUIRED_PARTS[1]]) + except ( + KeyError, + UnicodeError, + OSError, + ValueError, + NotImplementedError, + zipfile.BadZipFile, + zlib.error, + ): + return False + + +def _docx_part_elements( + archive: zipfile.ZipFile, part_info: zipfile.ZipInfo, ceiling: int +) -> tuple[tuple[str, Mapping[str, str]], ...] | None: + """Return one bounded OOXML part's elements, or ``None`` if it is not safe XML. + + The gate's runtime installs nothing -- the required workflow runs this + module with the runner's stock ``python3`` and no ``pip install`` step + (``.github/workflows/opencode-review.yml``'s ``required-workflow-bootstrap`` + job) -- so ``defusedxml`` is not importable here and a module-level import + of it would break the required check in every consumer repository. The + equivalent guarantee is reconstructed on the standard library's expat + parser instead: the bytes are read bounded (the read *is* the expansion + bound; a declared ZIP size is never trusted), decoded as strict UTF-8, and + refused outright if they contain U+0000 -- which rejects every UTF-16/UCS-4 + encoding of the checks below -- or the `` ceiling: + return None + try: + text = data.decode("utf-8") + except UnicodeDecodeError: + return None + if "\x00" in text or " bool: + """Return whether ``[Content_Types].xml`` declares exactly the main document part. + + Requires the OPC content-types root element and exactly one ``Override`` + naming ``/word/document.xml``, whose ``ContentType`` is exactly + ``DOCX_MAIN_DOCUMENT_CONTENT_TYPE``. A ``Default`` extension mapping, a + comment, an ambiguous pair of Overrides for the same part, or any other + declared content type does not satisfy this. + """ + + if elements[0][0] != f"{DOCX_CONTENT_TYPES_NS} Types": + return False + declared = [ + attributes.get("ContentType") + for name, attributes in elements + if name == f"{DOCX_CONTENT_TYPES_NS} Override" + and attributes.get("PartName") == DOCX_MAIN_DOCUMENT_PART_NAME + ] + return declared == [DOCX_MAIN_DOCUMENT_CONTENT_TYPE] + + +def _docx_relates_main_document( + elements: Sequence[tuple[str, Mapping[str, str]]] +) -> bool: + """Return whether ``_rels/.rels`` points its package root at that same part. + + Requires the OPC relationships root element and exactly one relationship of + the ``officeDocument`` type, internal, whose ``Target`` is exactly the main + document part in either permitted spelling. Transitional packages use + ``DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPE``; ISO/IEC 29500 Strict packages + use one URI in ``DOCX_STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES``. One of + each is two relationships and does not agree. Exact matching is what + rejects a traversal, an absolute or an external target: nothing is resolved + or normalized, so no target outside the package can agree. + """ + + if elements[0][0] != f"{DOCX_PACKAGE_RELATIONSHIPS_NS} Relationships": + return False + roots = [ + (attributes.get("Target"), attributes.get("TargetMode", "Internal")) + for name, attributes in elements + if name == f"{DOCX_PACKAGE_RELATIONSHIPS_NS} Relationship" + and attributes.get("Type") in DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPES + ] + return roots in ( + [(DOCX_MAIN_DOCUMENT_PART, "Internal")], + [(DOCX_MAIN_DOCUMENT_PART_NAME, "Internal")], + ) + + def _is_complete_hwpx(raw: bytes) -> bool: """Confirm a bounded HWPX container without extracting document content. @@ -953,11 +1163,12 @@ def evaluate_pull_request( # check ahead of _needs_content_scan's patch-presence-only signal: # a missing patch does not by itself prove binary content (GitHub # also omits one for an oversized textual diff), so this confirms - # the format's magic prefix whenever the bytes can be fetched at - # all, falling back to the path+suffix convention only when the - # content genuinely exceeds the Git blob API's size ceiling. A - # removed file has no head content to fetch at all -- _needs_content_scan - # already special-cases this the same way for every other file. + # structural evidence or the format's magic prefix whenever the + # bytes can be fetched at all, falling back to the path+suffix + # convention only when the content genuinely exceeds the Git blob + # API's size ceiling. A removed file has no head content to fetch + # at all -- _needs_content_scan already special-cases this the same + # way for every other file. if changed.status != "removed" and _is_binary_documentation_asset(changed, declared_prefixes): if _binary_documentation_evidence_confirms( changed, diff --git a/tests/test_pingora_docx_evidence.py b/tests/test_pingora_docx_evidence.py new file mode 100644 index 0000000000..7fef8e0f96 --- /dev/null +++ b/tests/test_pingora_docx_evidence.py @@ -0,0 +1,355 @@ +"""Exercise DOCX structural admission through the production policy boundary offline.""" +import io +import os +import warnings +import zipfile + +import pytest +from tests.test_pingora_edge_policy import policy +from tests.test_pingora_hwpx_evidence import RUNTIME_BYTES, RUNTIME_TEXT, evaluate_bytes + +# The real research artifact from late-life-anxiety-reanalysis#257 (exact head +# 706a81e5a6f88ad74544ab9cf89d4da2b9e6a44d) sits at this path. It is a +# DEFLATE-compressed OOXML package whose first member is [Content_Types].xml, +# with no ZIP comment and no prefixed or appended bytes. The fixtures below +# reproduce that structure without copying another repository's artifact into +# this one; test_real_artifact_bytes_are_admitted binds the same assertion to +# the actual bytes when PINGORA_REAL_DOCX_PATH names a local copy. +REAL_ARTIFACT_PATH = ( + "docs/delivery_interim_20260920/" + "air_render_00cdc51_g7integration_20260921_205341/manuscript_interim_20260921.docx" +) +REAL_ARTIFACT_ENV = "PINGORA_REAL_DOCX_PATH" +CONTENT_TYPES_NS = "http://schemas.openxmlformats.org/package/2006/content-types" +PACKAGE_RELATIONSHIPS_NS = "http://schemas.openxmlformats.org/package/2006/relationships" +OFFICE_DOCUMENT_RELATIONSHIP_TYPE = ( + "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" +) +# Library of Congress fdd000400 cites ISO/IEC 29500-1:2012 §11.3.10 for the +# first spelling. Microsoft Word's Strict Open XML save writes the second +# (python-openxml/python-docx#693). The OPC relationships namespace is unchanged. +STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES = ( + "http://purl.oclc.org/ooxml/relationships/officeDocument", + "http://purl.oclc.org/ooxml/officeDocument/relationships/officeDocument", +) +MAIN_DOCUMENT_CONTENT_TYPE = policy.DOCX_MAIN_DOCUMENT_CONTENT_TYPE +WORD_MAIN_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" +STRICT_WORD_MAIN_NS = "http://purl.oclc.org/ooxml/wordprocessingml/main" +XML_DECLARATION = '' +# Distinguishes "omit this part entirely" from "use the conforming default". +OMITTED = object() + + +def content_types_xml( + *, + part_name="/word/document.xml", + content_type=MAIN_DOCUMENT_CONTENT_TYPE, + overrides=1, + default_content_type=None, + root_tag="Types", + comment="", +): + """Render one [Content_Types].xml declaration with an exact Override shape.""" + default_extension = ( + f'' + if default_content_type is not None + else '' + ) + override = f'' + # An unrelated Override always accompanies the main one, so the part-name + # filter is exercised in both directions by every positive fixture. + unrelated = ( + '' + ) + body = default_extension + unrelated + comment + override * overrides + return f"{XML_DECLARATION}<{root_tag} xmlns=\"{CONTENT_TYPES_NS}\">{body}".encode() + + +def package_rels_xml( + *, + target="word/document.xml", + target_mode=None, + relationship_type=OFFICE_DOCUMENT_RELATIONSHIP_TYPE, + relationships=1, + root_tag="Relationships", + padding="", +): + """Render one _rels/.rels package relationship part.""" + mode = f' TargetMode="{target_mode}"' if target_mode is not None else "" + main = f'' + # A real package always carries unrelated relationships alongside the + # officeDocument one, so the relationship-type filter is exercised both ways. + unrelated = ( + '' + ) + body = unrelated + main * relationships + padding + return f"{XML_DECLARATION}<{root_tag} xmlns=\"{PACKAGE_RELATIONSHIPS_NS}\">{body}".encode() + + +def document_xml(*, namespace=WORD_MAIN_NS): + """Render one minimal but well-formed WordprocessingML main document part.""" + return f'{XML_DECLARATION}'.encode() + + +def docx_archive( + *, + content_types=None, + package_rels=None, + document=None, + compression_type=zipfile.ZIP_DEFLATED, + duplicate_document=False, +): + """Create a deterministic, non-sensitive OOXML boundary fixture.""" + archive_buffer = io.BytesIO() + members = [ + ("[Content_Types].xml", content_types_xml() if content_types is None else content_types), + ("_rels/.rels", package_rels_xml() if package_rels is None else package_rels), + ("word/document.xml", document_xml() if document is None else document), + ] + if duplicate_document: + members.append(("word/document.xml", document_xml())) + with warnings.catch_warnings(): + if duplicate_document: + # The duplicate-member rejection fixture writes the same part twice. + # zipfile warns on that write; the warning is the fixture, not a defect. + warnings.filterwarnings( + "ignore", + message=r"Duplicate name: 'word/document.xml'", + category=UserWarning, + ) + with zipfile.ZipFile(archive_buffer, "w") as archive_file: + for member_name, member_value in members: + if member_value is OMITTED: + continue + member_info = zipfile.ZipInfo(member_name) + member_info.compress_type = compression_type + archive_file.writestr(member_info, member_value) + return archive_buffer.getvalue() + + +def comment_only_mime_counterexample(): + """Build the reported bypass: the MIME only appears inside an XML comment.""" + return docx_archive( + content_types=content_types_xml( + content_type="application/octet-stream", + comment=f"", + ), + package_rels=b"not xml at all", + document=b"also not xml", + ) + + +def deflate_damaged(file_bytes): + """Corrupt the first compressed stream without touching the ZIP structure.""" + damaged = bytearray(file_bytes) + data_offset = damaged.find(b"[Content_Types].xml") + len(b"[Content_Types].xml") + for index in range(data_offset, data_offset + 16): + damaged[index] ^= 0xFF + return bytes(damaged) + + +@pytest.mark.parametrize("file_path", [REAL_ARTIFACT_PATH, "docs/manuscript.docx", "Docs/MANUSCRIPT.DOCX"]) +def test_valid_docx_is_admitted_at_the_real_research_artifact_path(file_path): + """Admit the #257 manuscript by structure, without moving or excluding it.""" + assert evaluate_bytes(file_path, docx_archive()) == () + + +@pytest.mark.parametrize("compression_type", [zipfile.ZIP_STORED, zipfile.ZIP_DEFLATED]) +def test_valid_docx_is_admitted_for_either_stored_or_deflated_parts(compression_type): + """Real writers DEFLATE every part; stored parts are equally structural.""" + assert evaluate_bytes("docs/manuscript.docx", docx_archive(compression_type=compression_type)) == () + + +@pytest.mark.parametrize("target", ["word/document.xml", "/word/document.xml"]) +@pytest.mark.parametrize("target_mode", [None, "Internal"]) +def test_valid_docx_accepts_either_relationship_target_spelling(target, target_mode): + """OPC permits both target spellings and an explicit Internal target mode.""" + archive_bytes = docx_archive(package_rels=package_rels_xml(target=target, target_mode=target_mode)) + assert evaluate_bytes("docs/manuscript.docx", archive_bytes) == () + + +def test_valid_docx_accepts_the_iso_strict_main_document_namespace(): + """ISO 29500 Strict manuscripts carry the same content type and must pass.""" + archive_bytes = docx_archive(document=document_xml(namespace=STRICT_WORD_MAIN_NS)) + assert evaluate_bytes("docs/manuscript.docx", archive_bytes) == () + + +@pytest.mark.parametrize("relationship_type", STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES) +def test_valid_docx_accepts_an_iso_strict_office_document_relationship(relationship_type): + """Strict WordprocessingML still points the package root at the main part.""" + archive_bytes = docx_archive( + package_rels=package_rels_xml(relationship_type=relationship_type), + document=document_xml(namespace=STRICT_WORD_MAIN_NS), + ) + assert evaluate_bytes("docs/manuscript.docx", archive_bytes) == () + + +def test_mixed_transitional_and_strict_office_document_relationships_are_ambiguous(): + """Exactly one officeDocument relationship is allowed, across either spelling.""" + rels = ( + f'{XML_DECLARATION}' + f'' + '' + "" + ).encode() + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(package_rels=rels)) + + +@pytest.mark.skipif( + not os.environ.get(REAL_ARTIFACT_ENV), + reason=f"{REAL_ARTIFACT_ENV} must name a local copy of the #257 manuscript", +) +def test_real_artifact_bytes_are_admitted(): + """Bind admission to the actual exact-head bytes, not only to the fixture.""" + with open(os.environ[REAL_ARTIFACT_ENV], "rb") as artifact_file: + raw = artifact_file.read() + assert raw.startswith(b"PK\x03\x04") + assert evaluate_bytes(REAL_ARTIFACT_PATH, raw) == () + + +@pytest.mark.parametrize("file_bytes", [ + b"PK\x03\x04\xff", + docx_archive()[:-10], + b"#!/bin/sh\n" + RUNTIME_BYTES + docx_archive(), + b"PK\x03\x04" + docx_archive(), + docx_archive() + b"\n" + RUNTIME_BYTES, + docx_archive()[:-2] + b"\x01\x00", + docx_archive(content_types=b""), + docx_archive(package_rels=b""), + docx_archive(document=b""), + docx_archive(duplicate_document=True), + deflate_damaged(docx_archive()), + comment_only_mime_counterexample(), +]) +def test_corrupt_or_disguised_docx_container_is_not_exempt(file_bytes): + """Unreadable, incomplete, or non-structural input still fails closed.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", file_bytes) + + +@pytest.mark.parametrize("missing_part", ["content_types", "package_rels", "document"]) +def test_docx_missing_a_required_opc_part_is_not_exempt(missing_part): + """Every part in DOCX_REQUIRED_PARTS must actually be present.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(**{missing_part: OMITTED})) + + +@pytest.mark.parametrize("content_types", [ + # The expected MIME as a Default extension mapping, never as the Override. + content_types_xml(content_type="application/octet-stream", default_content_type=MAIN_DOCUMENT_CONTENT_TYPE), + # A spreadsheet package renamed to .docx. + content_types_xml( + content_type="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml" + ), + # No Override for the main document part at all. + content_types_xml(overrides=0), + # Two Overrides for the same part name: an ambiguous declaration. + content_types_xml(overrides=2), + # The Override names a different part. + content_types_xml(part_name="/word/other.xml"), + # Correct Override, wrong root element. + content_types_xml(root_tag="Relationships"), +]) +def test_docx_requires_an_exact_main_document_content_type_override(content_types): + """Only an exact Override pairing the main part with its MIME admits.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(content_types=content_types)) + + +@pytest.mark.parametrize("package_rels", [ + # The officeDocument relationship points somewhere else entirely. + package_rels_xml(target="word/other.xml"), + # A path traversal target is rejected by exact matching. + package_rels_xml(target="../../etc/passwd"), + # An external target is not the packaged main document. + package_rels_xml(target="https://example.invalid/document.xml", target_mode="External"), + # No officeDocument relationship at all. + package_rels_xml(relationships=0), + # Two officeDocument relationships: an ambiguous package root. + package_rels_xml(relationships=2), + # A relationship of a different type cannot stand in for it. + package_rels_xml(relationship_type="http://schemas.openxmlformats.org/package/2006/relationships/x"), + # Correct relationship, wrong root element. + package_rels_xml(root_tag="Types"), +]) +def test_docx_requires_the_office_document_relationship_to_agree(package_rels): + """The package root relationship must resolve to the checked main part.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(package_rels=package_rels)) + + +@pytest.mark.parametrize("part_bytes", [ + b"not xml at all", + XML_DECLARATION.encode() + b"", + b"\xff\xfe", + XML_DECLARATION.encode() + b']>', + (']>').encode("utf-16-le"), + XML_DECLARATION.encode() + b"&undefined;", +]) +@pytest.mark.parametrize("part_name", ["content_types", "package_rels", "document"]) +def test_docx_requires_well_formed_entity_free_xml_parts(part_bytes, part_name): + """Each required part must be well-formed UTF-8 XML with no DTD or entity.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(**{part_name: part_bytes})) + + +@pytest.mark.parametrize("part_name, padding_length", [ + ("content_types", policy.MAX_DOCX_CONTENT_TYPES_BYTES + 1), + ("package_rels", policy.MAX_DOCX_CONTENT_TYPES_BYTES + 1), + ("document", policy.MAX_DOCX_MAIN_PART_BYTES + 1), +]) +def test_oversized_docx_part_is_not_read_beyond_its_ceiling(part_name, padding_length): + """A part beyond its bounded read ceiling is rejected, never streamed.""" + padding = "" + if part_name == "content_types": + part_bytes = content_types_xml(comment=padding) + elif part_name == "package_rels": + part_bytes = package_rels_xml(padding=padding) + else: + part_bytes = document_xml()[:-len(b"")] + padding.encode() + b"" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", docx_archive(**{part_name: part_bytes})) + + +def test_docx_rejects_an_encrypted_or_entryless_package(): + """An encrypted part or an empty central directory is unverifiable, so it fails.""" + encrypted_bytes = bytearray(docx_archive()) + encrypted_bytes[encrypted_bytes.find(b"PK\x01\x02") + 8] |= 1 + entryless_bytes = bytearray(docx_archive()) + end_offset = len(entryless_bytes) - 22 + entryless_bytes[end_offset + 8:end_offset + 12] = b"\x00\x00\x00\x00" + entryless_bytes[end_offset + 12:end_offset + 16] = b"\x00\x00\x00\x00" + entryless_bytes[end_offset + 16:end_offset + 20] = end_offset.to_bytes(4, "little") + for file_bytes in (bytes(encrypted_bytes), bytes(entryless_bytes)): + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/manuscript.docx", file_bytes) + + +@pytest.mark.parametrize("file_path", ["docs/manuscript.docx", "local/model.zip"]) +def test_non_utf8_bytes_without_a_structural_document_still_fail(file_path): + """Opaque bytes that are not a recognized structural document are never admitted.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes(file_path, b"\x00\x01\x02\xff\xfe" + RUNTIME_BYTES) + + +@pytest.mark.parametrize("patch_value", [None, "+" + RUNTIME_TEXT.rstrip("\n")]) +def test_valid_utf8_named_docx_keeps_todays_runtime_scan(patch_value): + """A file that decodes as UTF-8 is never treated as a binary artifact.""" + violations = evaluate_bytes("docs/manuscript.docx", RUNTIME_BYTES, patch_value=patch_value) + assert [violation.rule for violation in violations] == ["nginx_runtime_path"] + + +def test_docx_in_runtime_path_remains_unavailable(): + """A structural format exception cannot exempt an active runtime location.""" + with pytest.raises(policy.PolicyError): + evaluate_bytes("docs/nginx/manuscript.docx", docx_archive()) + + +def test_removed_docx_does_not_load_deleted_content(): + """Deletion does not require unavailable final-head bytes.""" + assert evaluate_bytes("docs/manuscript.docx", b"", file_status="removed") == ()