From c6483b8f0b382d019f462fb50e68579cf58dc208 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 19 Sep 2026 03:31:36 +0000 Subject: [PATCH 1/2] test(properties): cover sensitivity enforcement (#755 item 6) Adds the four properties #755 item 6 asks for, plus a guard-the-guard, over `context.sensitivity.apply_sensitivity_filter`: * the floor is INCLUSIVE -- an item whose sensitivity equals the floor is enforced, not passed; * everything below the floor survives untouched and in input order; * the text of an at-or-above item reaches model-visible output under NEITHER action; * redact keeps the slot, masks the payload, and clears `artifact_ref`. The issue warns specifically against "an inverted floor assumption", so the ladder is restated in the test rather than imported from `_SENSITIVITY_ORDER`. Importing it would make the properties agree with the implementation by construction: reorder the ladder in the source and the tests would reorder with it and still pass. The second property is the counter-property. Without it, an implementation returning `[]` for every input would satisfy "nothing disallowed survives" perfectly. Mutation-checked, each mutation read back before trusting the result: >= floor -> > floor (the inverted floor) 4 failed drop the else that keeps permitted items 2 failed redact appends the original, not the mask 2 failed MaskRedactionHook stops clearing artifact_ref 1 failed stop counting drops 1 failed mask leaks item.text into the placeholder 1 failed permitted items reordered (insert(0, item)) 2 failed The artifact_ref mutation initially PASSED. `ContextItem.artifact_ref` defaults to None, so `assert result.artifact_ref is None` was vacuous until the strategy was taught to generate refs. `find()` now pins that the corpus still reaches that state, so the assertion cannot go vacuous again silently. Every mutation was re-run after that strategy change. CI cost: tests/test_properties.py 7.11s -> 7.59s; the five new tests run in 0.92s alone. Honest limitation, recorded rather than glossed: the existing example suite (tests/test_sensitivity.py, tests/test_sensitivity_fixtures.py) catches all seven mutations too, usually more loudly. `sensitivity.py` was already at 100% line coverage and stays there. These properties add input-space breadth and state the contract as a universally quantified claim; they are not demonstrated to catch a defect the examples miss. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01XhJp1Rnr6mF4vdSLkj14jU --- tests/test_properties.py | 172 ++++++++++++++++++++++++++++++++++++++- 1 file changed, 170 insertions(+), 2 deletions(-) diff --git a/tests/test_properties.py b/tests/test_properties.py index d141902..b98f358 100644 --- a/tests/test_properties.py +++ b/tests/test_properties.py @@ -11,6 +11,9 @@ * ``tests/fixtures._normalize.to_canonical_json`` — round-trip idempotence. * ``context.consolidation.cluster_episodes`` — determinism, order-independence, and idempotence (the clustering is documented as stable for identical input). +* ``context.sensitivity.apply_sensitivity_filter`` — the floor is inclusive, no + item at or above it reaches model-visible output, and everything below it + survives untouched and in order. Kept fast and deterministic: no external I/O, bounded example sizes. """ @@ -20,12 +23,14 @@ import json import pytest -from hypothesis import given +from hypothesis import find, given from hypothesis import strategies as st +from contextweaver.config import ContextPolicy from contextweaver.context.candidates import resolve_dependency_closure from contextweaver.context.consolidation import cluster_episodes from contextweaver.context.dedup import deduplicate_candidates +from contextweaver.context.sensitivity import apply_sensitivity_filter from contextweaver.envelope import ( CHOICE_CARD_KINDS, CHOICE_CARD_NAME_MAX_LEN, @@ -41,7 +46,13 @@ from contextweaver.secrets import DEFAULT_SECRET_MASK, scrub_secrets from contextweaver.store.episodic import Episode from contextweaver.store.event_log import InMemoryEventLog -from contextweaver.types import ContextItem, ItemKind, SelectableItem, Sensitivity +from contextweaver.types import ( + ArtifactRef, + ContextItem, + ItemKind, + SelectableItem, + Sensitivity, +) from tests.fixtures._normalize import to_canonical_json # --------------------------------------------------------------------------- @@ -707,3 +718,160 @@ def test_pack_is_deterministic( assert second is not None assert [card.to_dict() for card in first] == [card.to_dict() for card in second] + + +# --------------------------------------------------------------------------- +# context.sensitivity.apply_sensitivity_filter — #755 item 6 +# --------------------------------------------------------------------------- + +# The severity ladder, restated here deliberately rather than imported from the +# module under test. Importing ``_SENSITIVITY_ORDER`` would make every property +# below agree with the implementation by construction: reorder the ladder in the +# source and the tests would reorder with it and still pass. +_LADDER: list[Sensitivity] = [ + Sensitivity.public, + Sensitivity.internal, + Sensitivity.confidential, + Sensitivity.restricted, +] + + +def _rank(level: Sensitivity) -> int: + return _LADDER.index(level) + + +@st.composite +def _sensitive_items(draw: st.DrawFn) -> list[ContextItem]: + """Items across the whole ladder, with broad text including Unicode.""" + size = draw(st.integers(min_value=0, max_value=8)) + items: list[ContextItem] = [] + for index in range(size): + # Some items carry an artifact_ref. Without this the #451 assertion + # ("redaction clears the ref") is vacuous: the field defaults to None, + # so `is None` holds whether or not the hook clears it. Deleting + # `artifact_ref=None` from MaskRedactionHook left the suite green until + # these refs were generated. + has_ref = draw(st.booleans()) + items.append( + ContextItem( + id=f"i{index}", + kind=draw(st.sampled_from(list(ItemKind))), + text=draw(st.text(max_size=40)), + sensitivity=draw(st.sampled_from(_LADDER)), + metadata={"n": index}, + artifact_ref=ArtifactRef( + handle=f"h{index}", media_type="text/plain", size_bytes=index + ) + if has_ref + else None, + ) + ) + return items + + +@given(_sensitive_items(), st.sampled_from(_LADDER)) +def test_drop_removes_everything_at_or_above_the_floor( + items: list[ContextItem], floor: Sensitivity +) -> None: + """#755 item 6: the floor is INCLUSIVE — at-or-above is enforced, not allowed. + + This is the "inverted floor assumption" the issue warns about. An item whose + sensitivity equals the floor is removed, not passed through, so the property + is ``rank < floor`` for survivors and never ``rank <= floor``. + """ + policy = ContextPolicy(sensitivity_floor=floor, sensitivity_action="drop") + kept, dropped = apply_sensitivity_filter(items, policy) + + assert all(_rank(item.sensitivity) < _rank(floor) for item in kept) + assert dropped == sum(1 for item in items if _rank(item.sensitivity) >= _rank(floor)) + + +@given(_sensitive_items(), st.sampled_from(_LADDER)) +def test_drop_preserves_everything_below_the_floor_in_order( + items: list[ContextItem], floor: Sensitivity +) -> None: + """The counter-property: enforcement must not be over-broad. + + Without this, an implementation that returned ``[]`` for every input would + satisfy the "nothing disallowed survives" property perfectly. Pinning the + survivors — same objects, same relative order — is what makes that property + mean something. + """ + policy = ContextPolicy(sensitivity_floor=floor, sensitivity_action="drop") + kept, _ = apply_sensitivity_filter(items, policy) + + expected = [item for item in items if _rank(item.sensitivity) < _rank(floor)] + assert kept == expected + + +@given(_sensitive_items(), st.sampled_from(_LADDER)) +def test_disallowed_text_never_reaches_model_visible_output( + items: list[ContextItem], floor: Sensitivity +) -> None: + """#755 item 6, stated as the security contract rather than as a count. + + Both actions are checked against the same corpus, because "dropped" and + "redacted" are two ways of making the same promise: the *text* of an item at + or above the floor is not in what the model sees. + + Texts that also occur on a permitted item are excluded from the assertion — + their presence in the output says nothing about whether the sensitive item + leaked, and asserting on them would make the property wrong rather than + strict. + """ + allowed_texts = {item.text for item in items if _rank(item.sensitivity) < _rank(floor)} + disallowed_texts = { + item.text for item in items if _rank(item.sensitivity) >= _rank(floor) + } - allowed_texts + + for action in ("drop", "redact"): + policy = ContextPolicy( + sensitivity_floor=floor, + sensitivity_action=action, # type: ignore[arg-type] + ) + kept, _ = apply_sensitivity_filter(items, policy) + visible = [item.text for item in kept] + for secret in disallowed_texts: + assert secret not in visible, f"{action}: sensitive text survived" + + +@given(_sensitive_items(), st.sampled_from(_LADDER)) +def test_redact_keeps_the_item_but_masks_its_payload( + items: list[ContextItem], floor: Sensitivity +) -> None: + """Redaction is structural: the slot stays, the payload and the ref do not. + + ``artifact_ref`` is asserted cleared because keeping it would let the + drilldown path re-fetch the original bytes and undo the redaction (#451) — + the mask alone is not the contract. + """ + policy = ContextPolicy(sensitivity_floor=floor, sensitivity_action="redact") + kept, dropped = apply_sensitivity_filter(items, policy) + + assert dropped == 0, "redact mode keeps items, so nothing is dropped" + assert len(kept) == len(items), "redact mode must not change the item count" + + for original, result in zip(items, kept, strict=True): + assert result.id == original.id + if _rank(original.sensitivity) >= _rank(floor): + assert result.text == f"[REDACTED: {original.sensitivity.value}]" + assert result.artifact_ref is None + assert result.metadata["redacted"] is True + else: + assert result == original + + +def test_generated_items_actually_carry_artifact_refs() -> None: + """Guard-the-guard: the #451 assertion above is only meaningful with refs. + + ``artifact_ref`` defaults to ``None``, so if the strategy stopped producing + refs, ``assert result.artifact_ref is None`` would pass no matter what the + redaction hook did — which is exactly how deleting ``artifact_ref=None`` + from ``MaskRedactionHook`` first went undetected here. ``find`` raises + ``NoSuchExample`` if the corpus can no longer reach that state. + """ + example = find( + _sensitive_items(), + lambda items: any(item.artifact_ref is not None for item in items), + ) + assert any(item.artifact_ref is not None for item in example) From 0d2409330863e7a5b7730ddd2b78093978fcb458 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 19 Sep 2026 03:32:10 +0000 Subject: [PATCH 2/2] docs(changelog): record the #755 item 6 property coverage Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01XhJp1Rnr6mF4vdSLkj14jU --- CHANGELOG.md | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index de180e0..1444cde 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -18,6 +18,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- Property coverage for item 6 of #755's phase-2 list, on sensitivity + enforcement: the floor is inclusive (an item *at* the floor is enforced, which + is the "inverted floor assumption" the issue warns against), everything below + it survives untouched and in input order, the text of an at-or-above item + reaches model-visible output under neither action, and `redact` keeps the slot + while masking the payload and clearing `artifact_ref` (#451). The severity + ladder is restated in the test rather than imported from `_SENSITIVITY_ORDER`, + so reordering it in the source cannot silently reorder the assertions too. + `sensitivity.py` was already at 100% line coverage and stays there; these add + input-space breadth, not lines. - Property coverage for items 1, 2 and 7 of #755's phase-2 list, all on the card-rendering and packing stage: the packer's cumulative budget is honoured or exactly one card is emitted (the documented soft cap never drops the first