diff --git a/scripts/ci/pingora_edge_policy.py b/scripts/ci/pingora_edge_policy.py
index 0d3a2c0948..33133f2422 100644
--- a/scripts/ci/pingora_edge_policy.py
+++ b/scripts/ci/pingora_edge_policy.py
@@ -22,7 +22,7 @@
Suffix decision: most research-data formats (``.xlsx``, ``.sav``, ``.rds``,
``.npz``, ...) have no entry in ``BINARY_DOCUMENT_MAGIC``, which only knows
-``.hwpx``/``.pdf``/``.png``. Rather than grow that registry for every such
+``.docx``/``.hwpx``/``.pdf``/``.png``. Rather than grow that registry for every such
format, a file under a declared prefix whose suffix has no magic entry is
admitted only when no diff patch is available, the fetched bytes fail to decode
as UTF-8, and their replacement-decoded text contains no prohibited runtime
@@ -32,8 +32,20 @@
admitting genuinely opaque research binaries without maintaining an open-ended
magic-byte catalog. A suffix that *does* have a magic entry keeps that entry's
existing structural evidence check
-(``_is_complete_png``, ``_is_complete_hwpx``, or the raw magic-prefix check for
-``.pdf``) even under a declared prefix.
+(``_is_complete_png``, ``_is_complete_hwpx``, ``_is_complete_docx``, or the raw
+magic-prefix check for ``.pdf``) even under a declared prefix -- so a ``.docx``
+under a declared prefix is held to the stricter structural proof rather than
+the UTF-8 complement, which fails closed in the same direction.
+
+late-life-anxiety-reanalysis#257 -- structural DOCX admission: a tracked
+research manuscript under ``docs/`` was rejected with "is not valid UTF-8"
+because ``.docx`` had no magic entry and so reached the ordinary content
+scan's UTF-8 decode. ``.docx`` is admitted on ``_is_complete_docx`` container
+evidence instead: the OPC parts, an exact main-document content-type
+``Override``, and an agreeing ``officeDocument`` relationship, each read under
+byte bounds and parsed as DTD-free XML. The artifact is neither relocated nor
+exempted: an unreadable, truncated, disguised, or non-WordprocessingML package
+still fails.
"""
from __future__ import annotations
@@ -54,6 +66,7 @@
from urllib.error import HTTPError, URLError
from urllib.parse import quote, urlsplit
from urllib.request import HTTPRedirectHandler, Request, build_opener
+from xml.parsers import expat
MAX_FILE_BYTES = 1_048_576
MAX_RESPONSE_BYTES = 16_777_216
@@ -78,11 +91,51 @@
# (this org's own "attach the relevant paper PDF" convention) for a reason
# that has nothing to do with the Nginx runtime policy this module enforces.
BINARY_DOCUMENT_MAGIC = {
+ ".docx": (b"PK\x03\x04",),
".hwpx": (b"PK\x03\x04",),
".pdf": (b"%PDF-",),
".png": (b"\x89PNG\r\n\x1a\n",),
}
PNG_SIGNATURE = BINARY_DOCUMENT_MAGIC[".png"][0]
+# OOXML WordprocessingML structural admission (late-life-anxiety-reanalysis#257).
+# A ``.docx`` is admitted by proving its container structure, never by decoding
+# it as text: the exact OPC parts every conforming writer emits, plus the main
+# document part's content type, which is what distinguishes a WordprocessingML
+# package from an arbitrary ZIP (or an HWPX) renamed to ``.docx``.
+DOCX_REQUIRED_PARTS = ("[Content_Types].xml", "_rels/.rels", "word/document.xml")
+DOCX_MAIN_DOCUMENT_PART = "word/document.xml"
+DOCX_MAIN_DOCUMENT_PART_NAME = "/word/document.xml"
+DOCX_MAIN_DOCUMENT_CONTENT_TYPE = (
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml"
+)
+DOCX_CONTENT_TYPES_NS = "http://schemas.openxmlformats.org/package/2006/content-types"
+DOCX_PACKAGE_RELATIONSHIPS_NS = "http://schemas.openxmlformats.org/package/2006/relationships"
+DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPE = (
+ "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument"
+)
+# ISO/IEC 29500 Strict names the same officeDocument part. Library of Congress
+# fdd000400 cites the first URI from ISO/IEC 29500-1:2012 §11.3.10. Microsoft
+# Word's Strict Open XML save writes the second (python-openxml/python-docx#693).
+DOCX_STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES = frozenset({
+ "http://purl.oclc.org/ooxml/relationships/officeDocument",
+ "http://purl.oclc.org/ooxml/officeDocument/relationships/officeDocument",
+})
+DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPES = frozenset({
+ DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPE,
+ *DOCX_STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES,
+})
+# Every required part is read bounded, and the bounded read *is* the expansion
+# bound: a part whose decompressed content exceeds its ceiling is rejected
+# rather than streamed, so admission cannot be turned into an unbounded read
+# or a decompression bomb. A declared ZIP size is never trusted for this --
+# only the length actually read is.
+MAX_DOCX_CONTENT_TYPES_BYTES = 65_536
+MAX_DOCX_MAIN_PART_BYTES = 8 * 1024 * 1024
+DOCX_PART_CEILINGS = {
+ "[Content_Types].xml": MAX_DOCX_CONTENT_TYPES_BYTES,
+ "_rels/.rels": MAX_DOCX_CONTENT_TYPES_BYTES,
+ "word/document.xml": MAX_DOCX_MAIN_PART_BYTES,
+}
SOURCE_TEST_SUFFIXES = frozenset({".py", ".pyi", ".js", ".mjs", ".cjs", ".ts", ".tsx", ".rs"})
LICENSE_NAMES = frozenset({"license", "license.md", "copying", "copyrights", "notice"})
DOCUMENTATION_DIRECTORIES = frozenset({"doc", "docs", "documentation"})
@@ -367,8 +420,8 @@ def _is_binary_documentation_asset(changed: ChangedFile, declared_prefixes: Sequ
*declared_prefixes* (issue #2193) is the base-ref-only research/data
artifact declaration: it replaces ONLY this function's path-shape test,
never the content evidence a caller still confirms below. A file whose
- suffix is a recognized ``BINARY_DOCUMENT_MAGIC`` format (``.hwpx``/
- ``.pdf``/``.png``) is admitted under a declared prefix on the exact same
+ suffix is a recognized ``BINARY_DOCUMENT_MAGIC`` format (``.docx``/
+ ``.hwpx``/``.pdf``/``.png``) is admitted under a declared prefix on the exact same
format evidence documentation paths already require. A file whose
suffix has no magic entry at all (research formats such as ``.xlsx``,
``.sav``, ``.rds``, ``.npz`` have none) can ONLY be admitted through a
@@ -653,9 +706,11 @@ def _binary_documentation_evidence_confirms(
also omits a patch for a textual diff that exceeds its own rendering
limit, well under this module's ``MAX_BLOB_BYTES`` content-fetch
ceiling. Whenever the file's raw bytes can be fetched at all, this
- verifies the declared format's magic prefix instead of trusting
- patch-presence alone. Only a file whose content evidently exceeds the
- Git blob API's size ceiling -- the exact case ``_is_binary_documentation_asset``
+ verifies the declared format's structural evidence (``_is_complete_docx``
+ for the OOXML manuscript case, ``_is_complete_hwpx``, ``_is_complete_png``)
+ or magic prefix instead of trusting patch-presence alone. Only a file
+ whose content evidently exceeds the Git blob API's size ceiling -- the
+ exact case ``_is_binary_documentation_asset``
exists for, a cited, large research paper -- falls back to trusting the
path+suffix convention for PDFs over 100 MB only; every other
content-evidence failure (a
@@ -685,6 +740,8 @@ def _binary_documentation_evidence_confirms(
return _is_complete_png(raw)
if suffix == ".hwpx":
return _is_complete_hwpx(raw)
+ if suffix == ".docx":
+ return _is_complete_docx(raw)
if suffix not in BINARY_DOCUMENT_MAGIC:
try:
raw.decode("utf-8")
@@ -695,6 +752,159 @@ def _binary_documentation_evidence_confirms(
return raw.startswith(BINARY_DOCUMENT_MAGIC[suffix])
+def _is_complete_docx(raw: bytes) -> bool:
+ """Confirm a bounded OOXML WordprocessingML container without reading its text.
+
+ Fail-closed structural admission for the research manuscript in
+ late-life-anxiety-reanalysis#257: require an unprefixed ZIP, its exact end
+ record, unique members, and every part in ``DOCX_REQUIRED_PARTS`` present,
+ non-empty, unencrypted and parseable as bounded, DTD-free XML
+ (``_docx_part_elements``). The main document part must then be declared as
+ an exact ``Override`` pairing ``/word/document.xml`` with
+ ``DOCX_MAIN_DOCUMENT_CONTENT_TYPE`` (``_docx_declares_main_document``) and
+ agreed by the package's single internal ``officeDocument`` relationship
+ (``_docx_relates_main_document``). A substring of the expected MIME
+ anywhere in the declaration -- in a comment, or in an unrelated attribute
+ -- is not a declaration and does not admit. Unlike HWPX, a conforming
+ ``.docx`` has no stored ``mimetype`` member and DEFLATEs every part, so
+ neither is required here. Anything unreadable, truncated, prefixed,
+ appended to, encrypted, or declaring a different document type returns
+ ``False`` -- the caller then scans the bytes as it always did, which for
+ genuinely binary content fails the policy rather than skipping it.
+ """
+ if not raw.startswith(BINARY_DOCUMENT_MAGIC[".docx"][0]):
+ return False
+ try:
+ with zipfile.ZipFile(io.BytesIO(raw)) as archive:
+ archive_entries = archive.infolist()
+ member_names = [member_info.filename for member_info in archive_entries]
+ end_offset = len(raw) - 22 - len(archive.comment)
+ if end_offset < 0 or raw[end_offset:end_offset + 4] != b"PK\x05\x06":
+ return False
+ if int.from_bytes(raw[end_offset + 20:end_offset + 22], "little") != len(archive.comment):
+ return False
+ if not archive_entries or archive_entries[0].header_offset != 0:
+ return False
+ if len(member_names) != len(set(member_names)):
+ return False
+ parsed_parts: dict[str, tuple[tuple[str, Mapping[str, str]], ...]] = {}
+ for part_name in DOCX_REQUIRED_PARTS:
+ part_info = archive.getinfo(part_name)
+ if part_info.flag_bits & 1 or part_info.file_size == 0:
+ return False
+ elements = _docx_part_elements(archive, part_info, DOCX_PART_CEILINGS[part_name])
+ if not elements:
+ return False
+ parsed_parts[part_name] = elements
+ return _docx_declares_main_document(
+ parsed_parts[DOCX_REQUIRED_PARTS[0]]
+ ) and _docx_relates_main_document(parsed_parts[DOCX_REQUIRED_PARTS[1]])
+ except (
+ KeyError,
+ UnicodeError,
+ OSError,
+ ValueError,
+ NotImplementedError,
+ zipfile.BadZipFile,
+ zlib.error,
+ ):
+ return False
+
+
+def _docx_part_elements(
+ archive: zipfile.ZipFile, part_info: zipfile.ZipInfo, ceiling: int
+) -> tuple[tuple[str, Mapping[str, str]], ...] | None:
+ """Return one bounded OOXML part's elements, or ``None`` if it is not safe XML.
+
+ The gate's runtime installs nothing -- the required workflow runs this
+ module with the runner's stock ``python3`` and no ``pip install`` step
+ (``.github/workflows/opencode-review.yml``'s ``required-workflow-bootstrap``
+ job) -- so ``defusedxml`` is not importable here and a module-level import
+ of it would break the required check in every consumer repository. The
+ equivalent guarantee is reconstructed on the standard library's expat
+ parser instead: the bytes are read bounded (the read *is* the expansion
+ bound; a declared ZIP size is never trusted), decoded as strict UTF-8, and
+ refused outright if they contain U+0000 -- which rejects every UTF-16/UCS-4
+ encoding of the checks below -- or the `` ceiling:
+ return None
+ try:
+ text = data.decode("utf-8")
+ except UnicodeDecodeError:
+ return None
+ if "\x00" in text or " bool:
+ """Return whether ``[Content_Types].xml`` declares exactly the main document part.
+
+ Requires the OPC content-types root element and exactly one ``Override``
+ naming ``/word/document.xml``, whose ``ContentType`` is exactly
+ ``DOCX_MAIN_DOCUMENT_CONTENT_TYPE``. A ``Default`` extension mapping, a
+ comment, an ambiguous pair of Overrides for the same part, or any other
+ declared content type does not satisfy this.
+ """
+
+ if elements[0][0] != f"{DOCX_CONTENT_TYPES_NS} Types":
+ return False
+ declared = [
+ attributes.get("ContentType")
+ for name, attributes in elements
+ if name == f"{DOCX_CONTENT_TYPES_NS} Override"
+ and attributes.get("PartName") == DOCX_MAIN_DOCUMENT_PART_NAME
+ ]
+ return declared == [DOCX_MAIN_DOCUMENT_CONTENT_TYPE]
+
+
+def _docx_relates_main_document(
+ elements: Sequence[tuple[str, Mapping[str, str]]]
+) -> bool:
+ """Return whether ``_rels/.rels`` points its package root at that same part.
+
+ Requires the OPC relationships root element and exactly one relationship of
+ the ``officeDocument`` type, internal, whose ``Target`` is exactly the main
+ document part in either permitted spelling. Transitional packages use
+ ``DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPE``; ISO/IEC 29500 Strict packages
+ use one URI in ``DOCX_STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES``. One of
+ each is two relationships and does not agree. Exact matching is what
+ rejects a traversal, an absolute or an external target: nothing is resolved
+ or normalized, so no target outside the package can agree.
+ """
+
+ if elements[0][0] != f"{DOCX_PACKAGE_RELATIONSHIPS_NS} Relationships":
+ return False
+ roots = [
+ (attributes.get("Target"), attributes.get("TargetMode", "Internal"))
+ for name, attributes in elements
+ if name == f"{DOCX_PACKAGE_RELATIONSHIPS_NS} Relationship"
+ and attributes.get("Type") in DOCX_OFFICE_DOCUMENT_RELATIONSHIP_TYPES
+ ]
+ return roots in (
+ [(DOCX_MAIN_DOCUMENT_PART, "Internal")],
+ [(DOCX_MAIN_DOCUMENT_PART_NAME, "Internal")],
+ )
+
+
def _is_complete_hwpx(raw: bytes) -> bool:
"""Confirm a bounded HWPX container without extracting document content.
@@ -953,11 +1163,12 @@ def evaluate_pull_request(
# check ahead of _needs_content_scan's patch-presence-only signal:
# a missing patch does not by itself prove binary content (GitHub
# also omits one for an oversized textual diff), so this confirms
- # the format's magic prefix whenever the bytes can be fetched at
- # all, falling back to the path+suffix convention only when the
- # content genuinely exceeds the Git blob API's size ceiling. A
- # removed file has no head content to fetch at all -- _needs_content_scan
- # already special-cases this the same way for every other file.
+ # structural evidence or the format's magic prefix whenever the
+ # bytes can be fetched at all, falling back to the path+suffix
+ # convention only when the content genuinely exceeds the Git blob
+ # API's size ceiling. A removed file has no head content to fetch
+ # at all -- _needs_content_scan already special-cases this the same
+ # way for every other file.
if changed.status != "removed" and _is_binary_documentation_asset(changed, declared_prefixes):
if _binary_documentation_evidence_confirms(
changed,
diff --git a/tests/test_pingora_docx_evidence.py b/tests/test_pingora_docx_evidence.py
new file mode 100644
index 0000000000..7fef8e0f96
--- /dev/null
+++ b/tests/test_pingora_docx_evidence.py
@@ -0,0 +1,355 @@
+"""Exercise DOCX structural admission through the production policy boundary offline."""
+import io
+import os
+import warnings
+import zipfile
+
+import pytest
+from tests.test_pingora_edge_policy import policy
+from tests.test_pingora_hwpx_evidence import RUNTIME_BYTES, RUNTIME_TEXT, evaluate_bytes
+
+# The real research artifact from late-life-anxiety-reanalysis#257 (exact head
+# 706a81e5a6f88ad74544ab9cf89d4da2b9e6a44d) sits at this path. It is a
+# DEFLATE-compressed OOXML package whose first member is [Content_Types].xml,
+# with no ZIP comment and no prefixed or appended bytes. The fixtures below
+# reproduce that structure without copying another repository's artifact into
+# this one; test_real_artifact_bytes_are_admitted binds the same assertion to
+# the actual bytes when PINGORA_REAL_DOCX_PATH names a local copy.
+REAL_ARTIFACT_PATH = (
+ "docs/delivery_interim_20260920/"
+ "air_render_00cdc51_g7integration_20260921_205341/manuscript_interim_20260921.docx"
+)
+REAL_ARTIFACT_ENV = "PINGORA_REAL_DOCX_PATH"
+CONTENT_TYPES_NS = "http://schemas.openxmlformats.org/package/2006/content-types"
+PACKAGE_RELATIONSHIPS_NS = "http://schemas.openxmlformats.org/package/2006/relationships"
+OFFICE_DOCUMENT_RELATIONSHIP_TYPE = (
+ "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument"
+)
+# Library of Congress fdd000400 cites ISO/IEC 29500-1:2012 §11.3.10 for the
+# first spelling. Microsoft Word's Strict Open XML save writes the second
+# (python-openxml/python-docx#693). The OPC relationships namespace is unchanged.
+STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES = (
+ "http://purl.oclc.org/ooxml/relationships/officeDocument",
+ "http://purl.oclc.org/ooxml/officeDocument/relationships/officeDocument",
+)
+MAIN_DOCUMENT_CONTENT_TYPE = policy.DOCX_MAIN_DOCUMENT_CONTENT_TYPE
+WORD_MAIN_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
+STRICT_WORD_MAIN_NS = "http://purl.oclc.org/ooxml/wordprocessingml/main"
+XML_DECLARATION = ''
+# Distinguishes "omit this part entirely" from "use the conforming default".
+OMITTED = object()
+
+
+def content_types_xml(
+ *,
+ part_name="/word/document.xml",
+ content_type=MAIN_DOCUMENT_CONTENT_TYPE,
+ overrides=1,
+ default_content_type=None,
+ root_tag="Types",
+ comment="",
+):
+ """Render one [Content_Types].xml declaration with an exact Override shape."""
+ default_extension = (
+ f''
+ if default_content_type is not None
+ else ''
+ )
+ override = f''
+ # An unrelated Override always accompanies the main one, so the part-name
+ # filter is exercised in both directions by every positive fixture.
+ unrelated = (
+ ''
+ )
+ body = default_extension + unrelated + comment + override * overrides
+ return f"{XML_DECLARATION}<{root_tag} xmlns=\"{CONTENT_TYPES_NS}\">{body}{root_tag}>".encode()
+
+
+def package_rels_xml(
+ *,
+ target="word/document.xml",
+ target_mode=None,
+ relationship_type=OFFICE_DOCUMENT_RELATIONSHIP_TYPE,
+ relationships=1,
+ root_tag="Relationships",
+ padding="",
+):
+ """Render one _rels/.rels package relationship part."""
+ mode = f' TargetMode="{target_mode}"' if target_mode is not None else ""
+ main = f''
+ # A real package always carries unrelated relationships alongside the
+ # officeDocument one, so the relationship-type filter is exercised both ways.
+ unrelated = (
+ ''
+ )
+ body = unrelated + main * relationships + padding
+ return f"{XML_DECLARATION}<{root_tag} xmlns=\"{PACKAGE_RELATIONSHIPS_NS}\">{body}{root_tag}>".encode()
+
+
+def document_xml(*, namespace=WORD_MAIN_NS):
+ """Render one minimal but well-formed WordprocessingML main document part."""
+ return f'{XML_DECLARATION}'.encode()
+
+
+def docx_archive(
+ *,
+ content_types=None,
+ package_rels=None,
+ document=None,
+ compression_type=zipfile.ZIP_DEFLATED,
+ duplicate_document=False,
+):
+ """Create a deterministic, non-sensitive OOXML boundary fixture."""
+ archive_buffer = io.BytesIO()
+ members = [
+ ("[Content_Types].xml", content_types_xml() if content_types is None else content_types),
+ ("_rels/.rels", package_rels_xml() if package_rels is None else package_rels),
+ ("word/document.xml", document_xml() if document is None else document),
+ ]
+ if duplicate_document:
+ members.append(("word/document.xml", document_xml()))
+ with warnings.catch_warnings():
+ if duplicate_document:
+ # The duplicate-member rejection fixture writes the same part twice.
+ # zipfile warns on that write; the warning is the fixture, not a defect.
+ warnings.filterwarnings(
+ "ignore",
+ message=r"Duplicate name: 'word/document.xml'",
+ category=UserWarning,
+ )
+ with zipfile.ZipFile(archive_buffer, "w") as archive_file:
+ for member_name, member_value in members:
+ if member_value is OMITTED:
+ continue
+ member_info = zipfile.ZipInfo(member_name)
+ member_info.compress_type = compression_type
+ archive_file.writestr(member_info, member_value)
+ return archive_buffer.getvalue()
+
+
+def comment_only_mime_counterexample():
+ """Build the reported bypass: the MIME only appears inside an XML comment."""
+ return docx_archive(
+ content_types=content_types_xml(
+ content_type="application/octet-stream",
+ comment=f"",
+ ),
+ package_rels=b"not xml at all",
+ document=b"also not xml",
+ )
+
+
+def deflate_damaged(file_bytes):
+ """Corrupt the first compressed stream without touching the ZIP structure."""
+ damaged = bytearray(file_bytes)
+ data_offset = damaged.find(b"[Content_Types].xml") + len(b"[Content_Types].xml")
+ for index in range(data_offset, data_offset + 16):
+ damaged[index] ^= 0xFF
+ return bytes(damaged)
+
+
+@pytest.mark.parametrize("file_path", [REAL_ARTIFACT_PATH, "docs/manuscript.docx", "Docs/MANUSCRIPT.DOCX"])
+def test_valid_docx_is_admitted_at_the_real_research_artifact_path(file_path):
+ """Admit the #257 manuscript by structure, without moving or excluding it."""
+ assert evaluate_bytes(file_path, docx_archive()) == ()
+
+
+@pytest.mark.parametrize("compression_type", [zipfile.ZIP_STORED, zipfile.ZIP_DEFLATED])
+def test_valid_docx_is_admitted_for_either_stored_or_deflated_parts(compression_type):
+ """Real writers DEFLATE every part; stored parts are equally structural."""
+ assert evaluate_bytes("docs/manuscript.docx", docx_archive(compression_type=compression_type)) == ()
+
+
+@pytest.mark.parametrize("target", ["word/document.xml", "/word/document.xml"])
+@pytest.mark.parametrize("target_mode", [None, "Internal"])
+def test_valid_docx_accepts_either_relationship_target_spelling(target, target_mode):
+ """OPC permits both target spellings and an explicit Internal target mode."""
+ archive_bytes = docx_archive(package_rels=package_rels_xml(target=target, target_mode=target_mode))
+ assert evaluate_bytes("docs/manuscript.docx", archive_bytes) == ()
+
+
+def test_valid_docx_accepts_the_iso_strict_main_document_namespace():
+ """ISO 29500 Strict manuscripts carry the same content type and must pass."""
+ archive_bytes = docx_archive(document=document_xml(namespace=STRICT_WORD_MAIN_NS))
+ assert evaluate_bytes("docs/manuscript.docx", archive_bytes) == ()
+
+
+@pytest.mark.parametrize("relationship_type", STRICT_OFFICE_DOCUMENT_RELATIONSHIP_TYPES)
+def test_valid_docx_accepts_an_iso_strict_office_document_relationship(relationship_type):
+ """Strict WordprocessingML still points the package root at the main part."""
+ archive_bytes = docx_archive(
+ package_rels=package_rels_xml(relationship_type=relationship_type),
+ document=document_xml(namespace=STRICT_WORD_MAIN_NS),
+ )
+ assert evaluate_bytes("docs/manuscript.docx", archive_bytes) == ()
+
+
+def test_mixed_transitional_and_strict_office_document_relationships_are_ambiguous():
+ """Exactly one officeDocument relationship is allowed, across either spelling."""
+ rels = (
+ f'{XML_DECLARATION}'
+ f''
+ ''
+ ""
+ ).encode()
+ with pytest.raises(policy.PolicyError):
+ evaluate_bytes("docs/manuscript.docx", docx_archive(package_rels=rels))
+
+
+@pytest.mark.skipif(
+ not os.environ.get(REAL_ARTIFACT_ENV),
+ reason=f"{REAL_ARTIFACT_ENV} must name a local copy of the #257 manuscript",
+)
+def test_real_artifact_bytes_are_admitted():
+ """Bind admission to the actual exact-head bytes, not only to the fixture."""
+ with open(os.environ[REAL_ARTIFACT_ENV], "rb") as artifact_file:
+ raw = artifact_file.read()
+ assert raw.startswith(b"PK\x03\x04")
+ assert evaluate_bytes(REAL_ARTIFACT_PATH, raw) == ()
+
+
+@pytest.mark.parametrize("file_bytes", [
+ b"PK\x03\x04\xff",
+ docx_archive()[:-10],
+ b"#!/bin/sh\n" + RUNTIME_BYTES + docx_archive(),
+ b"PK\x03\x04" + docx_archive(),
+ docx_archive() + b"\n" + RUNTIME_BYTES,
+ docx_archive()[:-2] + b"\x01\x00",
+ docx_archive(content_types=b""),
+ docx_archive(package_rels=b""),
+ docx_archive(document=b""),
+ docx_archive(duplicate_document=True),
+ deflate_damaged(docx_archive()),
+ comment_only_mime_counterexample(),
+])
+def test_corrupt_or_disguised_docx_container_is_not_exempt(file_bytes):
+ """Unreadable, incomplete, or non-structural input still fails closed."""
+ with pytest.raises(policy.PolicyError):
+ evaluate_bytes("docs/manuscript.docx", file_bytes)
+
+
+@pytest.mark.parametrize("missing_part", ["content_types", "package_rels", "document"])
+def test_docx_missing_a_required_opc_part_is_not_exempt(missing_part):
+ """Every part in DOCX_REQUIRED_PARTS must actually be present."""
+ with pytest.raises(policy.PolicyError):
+ evaluate_bytes("docs/manuscript.docx", docx_archive(**{missing_part: OMITTED}))
+
+
+@pytest.mark.parametrize("content_types", [
+ # The expected MIME as a Default extension mapping, never as the Override.
+ content_types_xml(content_type="application/octet-stream", default_content_type=MAIN_DOCUMENT_CONTENT_TYPE),
+ # A spreadsheet package renamed to .docx.
+ content_types_xml(
+ content_type="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml"
+ ),
+ # No Override for the main document part at all.
+ content_types_xml(overrides=0),
+ # Two Overrides for the same part name: an ambiguous declaration.
+ content_types_xml(overrides=2),
+ # The Override names a different part.
+ content_types_xml(part_name="/word/other.xml"),
+ # Correct Override, wrong root element.
+ content_types_xml(root_tag="Relationships"),
+])
+def test_docx_requires_an_exact_main_document_content_type_override(content_types):
+ """Only an exact Override pairing the main part with its MIME admits."""
+ with pytest.raises(policy.PolicyError):
+ evaluate_bytes("docs/manuscript.docx", docx_archive(content_types=content_types))
+
+
+@pytest.mark.parametrize("package_rels", [
+ # The officeDocument relationship points somewhere else entirely.
+ package_rels_xml(target="word/other.xml"),
+ # A path traversal target is rejected by exact matching.
+ package_rels_xml(target="../../etc/passwd"),
+ # An external target is not the packaged main document.
+ package_rels_xml(target="https://example.invalid/document.xml", target_mode="External"),
+ # No officeDocument relationship at all.
+ package_rels_xml(relationships=0),
+ # Two officeDocument relationships: an ambiguous package root.
+ package_rels_xml(relationships=2),
+ # A relationship of a different type cannot stand in for it.
+ package_rels_xml(relationship_type="http://schemas.openxmlformats.org/package/2006/relationships/x"),
+ # Correct relationship, wrong root element.
+ package_rels_xml(root_tag="Types"),
+])
+def test_docx_requires_the_office_document_relationship_to_agree(package_rels):
+ """The package root relationship must resolve to the checked main part."""
+ with pytest.raises(policy.PolicyError):
+ evaluate_bytes("docs/manuscript.docx", docx_archive(package_rels=package_rels))
+
+
+@pytest.mark.parametrize("part_bytes", [
+ b"not xml at all",
+ XML_DECLARATION.encode() + b"",
+ b"\xff\xfe",
+ XML_DECLARATION.encode() + b']>',
+ (']>').encode("utf-16-le"),
+ XML_DECLARATION.encode() + b"&undefined;",
+])
+@pytest.mark.parametrize("part_name", ["content_types", "package_rels", "document"])
+def test_docx_requires_well_formed_entity_free_xml_parts(part_bytes, part_name):
+ """Each required part must be well-formed UTF-8 XML with no DTD or entity."""
+ with pytest.raises(policy.PolicyError):
+ evaluate_bytes("docs/manuscript.docx", docx_archive(**{part_name: part_bytes}))
+
+
+@pytest.mark.parametrize("part_name, padding_length", [
+ ("content_types", policy.MAX_DOCX_CONTENT_TYPES_BYTES + 1),
+ ("package_rels", policy.MAX_DOCX_CONTENT_TYPES_BYTES + 1),
+ ("document", policy.MAX_DOCX_MAIN_PART_BYTES + 1),
+])
+def test_oversized_docx_part_is_not_read_beyond_its_ceiling(part_name, padding_length):
+ """A part beyond its bounded read ceiling is rejected, never streamed."""
+ padding = ""
+ if part_name == "content_types":
+ part_bytes = content_types_xml(comment=padding)
+ elif part_name == "package_rels":
+ part_bytes = package_rels_xml(padding=padding)
+ else:
+ part_bytes = document_xml()[:-len(b"")] + padding.encode() + b""
+ with pytest.raises(policy.PolicyError):
+ evaluate_bytes("docs/manuscript.docx", docx_archive(**{part_name: part_bytes}))
+
+
+def test_docx_rejects_an_encrypted_or_entryless_package():
+ """An encrypted part or an empty central directory is unverifiable, so it fails."""
+ encrypted_bytes = bytearray(docx_archive())
+ encrypted_bytes[encrypted_bytes.find(b"PK\x01\x02") + 8] |= 1
+ entryless_bytes = bytearray(docx_archive())
+ end_offset = len(entryless_bytes) - 22
+ entryless_bytes[end_offset + 8:end_offset + 12] = b"\x00\x00\x00\x00"
+ entryless_bytes[end_offset + 12:end_offset + 16] = b"\x00\x00\x00\x00"
+ entryless_bytes[end_offset + 16:end_offset + 20] = end_offset.to_bytes(4, "little")
+ for file_bytes in (bytes(encrypted_bytes), bytes(entryless_bytes)):
+ with pytest.raises(policy.PolicyError):
+ evaluate_bytes("docs/manuscript.docx", file_bytes)
+
+
+@pytest.mark.parametrize("file_path", ["docs/manuscript.docx", "local/model.zip"])
+def test_non_utf8_bytes_without_a_structural_document_still_fail(file_path):
+ """Opaque bytes that are not a recognized structural document are never admitted."""
+ with pytest.raises(policy.PolicyError):
+ evaluate_bytes(file_path, b"\x00\x01\x02\xff\xfe" + RUNTIME_BYTES)
+
+
+@pytest.mark.parametrize("patch_value", [None, "+" + RUNTIME_TEXT.rstrip("\n")])
+def test_valid_utf8_named_docx_keeps_todays_runtime_scan(patch_value):
+ """A file that decodes as UTF-8 is never treated as a binary artifact."""
+ violations = evaluate_bytes("docs/manuscript.docx", RUNTIME_BYTES, patch_value=patch_value)
+ assert [violation.rule for violation in violations] == ["nginx_runtime_path"]
+
+
+def test_docx_in_runtime_path_remains_unavailable():
+ """A structural format exception cannot exempt an active runtime location."""
+ with pytest.raises(policy.PolicyError):
+ evaluate_bytes("docs/nginx/manuscript.docx", docx_archive())
+
+
+def test_removed_docx_does_not_load_deleted_content():
+ """Deletion does not require unavailable final-head bytes."""
+ assert evaluate_bytes("docs/manuscript.docx", b"", file_status="removed") == ()