diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 4d8beab4..393ae5ff 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -97,7 +97,7 @@ The engine is the orchestration layer that coordinates the full pipeline. It: The engine is capability-agnostic. It does not know what a "grammar" or "rule" does — it only knows that grammars produce span-bearing recognition matches and rules produce candidates. -Before the recognition phase, the engine composes each capability's shipped grammars and rules with any community extensions registered for that capability (see "Community Extensions" below). Composition is guarded: duplicate names fail fast, and every rule's declared `target_grammars` must resolve within the composed set. +Before the recognition phase, the engine composes each capability's shipped grammars and rules with any community extensions registered for that capability (see "Community Extensions" below). Composition is guarded: duplicate names fail fast, and every rule's declared `target_semantics` must resolve within the composed set. ### Public API @@ -171,9 +171,9 @@ Capabilities are closed for modification but open for extension. Community contr A contract opts a registered grammar in by naming it in `extra_grammars`, a base `CapabilityContract` field surfaced on every `create_contract` factory. The engine composes the shipped active set with the opted-in extras, deduplicating names while preserving order — shipped slots first, extras after (unknown extra names are silently skipped). The shipped slots are `contract.active_grammars` when the contract implements it (the gated capabilities), or every shipped grammar in `get_grammars()` order when it returns `None` (the base default). Opt-in preserves determinism: a contract that names no extras composes to exactly the shipped set, so non-opt-in behavior is byte-identical. -Community rules follow the same opt-in discipline: a registered rule runs only when the contract names one of its `target_grammars` in `extra_grammars`. An un-opted community rule — even one targeting a shipped grammar — never affects results, so a default contract resolves with shipped rules only. +Community rules follow the same opt-in discipline: a registered rule runs only when the contract's `extra_grammars` resolve to one of its `target_semantics` ids. An un-opted community rule — even one targeting a shipped grammar's semantics — never affects results, so a default contract resolves with shipped rules only. -Composition is guarded at pipeline start: a community grammar name colliding with a shipped name raises `CapabilityError`, and an opted-in community rule whose `target_grammars` names a missing grammar raises `ContractError` — failing fast rather than producing a silently wrong result. Community grammars and rules are pure functions of their inputs, and the composed set is fixed once the registries freeze, so the determinism guarantees of "Determinism by Construction" extend unchanged. +Composition is guarded at pipeline start: a community grammar name colliding with a shipped name raises `CapabilityError`, and an opted-in community rule whose `target_semantics` names an id no grammar claims raises `ContractError` — failing fast rather than producing a silently wrong result. Community grammars and rules are pure functions of their inputs, and the composed set is fixed once the registries freeze, so the determinism guarantees of "Determinism by Construction" extend unchanged. --- diff --git a/CONTEXT.md b/CONTEXT.md index 4da7da5f..4cba1350 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -137,7 +137,7 @@ class Section341AddrSpec(Rule[EmailNotation]): strategy = RuleStrategy.REGEX provenance = PUBLICATION citation = "Section 3.4.1 (addr-spec)" # Human-readable citation - target_grammars = frozenset({"standard_recognition", "obfuscated_recognition"}) + target_semantics = frozenset({"rfc5322_addr_spec"}) requires_features = frozenset() # authority features this rule gates on def matches(self, notation: EmailNotation, contract: Contract) -> bool: @@ -152,7 +152,7 @@ class Section341AddrSpec(Rule[EmailNotation]): return f"{notation.local_part.lower()}@{notation.domain_part.lower()}" ``` -Every rule declares six metadata attrs — `name`, `strategy`, `provenance`, `citation`, `target_grammars` (non-empty `frozenset[str]`), `requires_features` (`frozenset[str]`) — enforced at import time by `Rule.__init_subclass__`. Rules never raise and never read `output_format`. +Every rule declares six metadata attrs — `name`, `strategy`, `provenance`, `citation`, `target_semantics` (non-empty `frozenset[str]`), `requires_features` (`frozenset[str]`) — enforced at import time by `Rule.__init_subclass__`. Every grammar declares a `semantics` string — the meaning id its recognized notations carry (its identity id by default; a coalesced id shared by same-meaning grammars, e.g. the standard and obfuscated Email grammars both declare `"rfc5322_addr_spec"`) — enforced as a non-empty `str` at import time by `Grammar.__init_subclass__`. Rules never raise and never read `output_format`. ### Notation Purpose Notation exists for **placement-sensitive rules**: @@ -173,13 +173,16 @@ The resolver **consumes notation** and outputs a canonical_value (not notation). ```python # capabilities/Currency/rules/iso_4217_ed2015.py + class SectionCode(Rule[CurrencyNotation]): """ISO 4217 Section 3 - Currency and funds codes""" name = "Section 3-code" strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION - target_grammars = frozenset({"code_recognition", "symbol_recognition", "word_recognition"}) + target_semantics = frozenset( + {"code_recognition", "symbol_recognition", "word_recognition"} + ) requires_features = frozenset() def matches(self, notation: CurrencyNotation, contract: Contract) -> bool: @@ -204,7 +207,7 @@ class Section431CalendarDate(Rule[DateNotation]): name = "Section 4.3.1-calendar-date" strategy = RuleStrategy.PARSER provenance = PUBLICATION - target_grammars = frozenset({"iso8601_recognition"}) + target_semantics = frozenset({"iso8601_calendar_date"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: @@ -244,6 +247,7 @@ class StandardEmailGrammar(Grammar[EmailNotation]): """Standard email recognition: user@domain.tld""" name = "standard_recognition" + semantics = "rfc5322_addr_spec" # coalesced — shared with obfuscated_recognition def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: """Extract span-bearing email matches from text.""" diff --git a/HOW_TO_ADD_NEW_CAPABILITY.md b/HOW_TO_ADD_NEW_CAPABILITY.md index 278770c9..b2750d88 100644 --- a/HOW_TO_ADD_NEW_CAPABILITY.md +++ b/HOW_TO_ADD_NEW_CAPABILITY.md @@ -169,6 +169,7 @@ class StandardMyDomainGrammar(Grammar[MyDomainNotation]): """Standard recognition for the MyDomain capability.""" name = "standard_recognition" + semantics = "standard_recognition" def recognize(self, text: str) -> list[RecognitionMatch[MyDomainNotation]]: """Extract span-bearing matches from raw text. @@ -289,7 +290,7 @@ Codebase examples: IP's IPv4 grammar is a loose regex (`\d{1,3}` octets) and `rf 3. Set `provenance` to the `PUBLICATION` constant defined above 4. Set `citation` to a human-readable citation (e.g., "Section 3.4.1 (addr-spec)") -5. Set `target_grammars` to the `frozenset[str]` of grammar names whose notations this rule validates (e.g., `frozenset({"standard_recognition"})`) +5. Set `target_semantics` to the `frozenset[str]` of grammar semantics whose notations this rule validates (e.g., `frozenset({"standard_recognition"})`) 6. Set `requires_features` to the `frozenset[str]` of contract fields that must be truthy for the rule to run (`frozenset()` when it always runs) All six attributes are enforced by `Rule.__init_subclass__` at class-definition time; see the Rule metadata section in Step 7. @@ -438,7 +439,7 @@ Capability-specific parameters come after the common block. Every capability sat **3. Rule metadata** -Every `Rule` subclass must declare six class attributes: `name`, `strategy`, `provenance`, `citation`, `target_grammars`, and `requires_features`. `Rule.__init_subclass__` enforces this at class-definition time, and a subclass missing any of them fails to import with a `TypeError`: +Every `Rule` subclass must declare six class attributes: `name`, `strategy`, `provenance`, `citation`, `target_semantics`, and `requires_features`. `Rule.__init_subclass__` enforces this at class-definition time, and a subclass missing any of them fails to import with a `TypeError`: ```python class SectionYourRule(Rule[YourDomainNotation]): @@ -446,11 +447,11 @@ class SectionYourRule(Rule[YourDomainNotation]): strategy = RuleStrategy.REGEX provenance = PUBLICATION citation = "Section 1 (your-rule)" - target_grammars = frozenset({"your_recognition"}) + target_semantics = frozenset({"your_recognition"}) requires_features = frozenset() ``` -- **`target_grammars: ClassVar[frozenset[str]]`** is the non-empty set of grammar names whose notations this rule validates. The engine uses it for affinity routing: each recognition is validated only by rules whose `target_grammars` includes the producing grammar's name, and a rule declaring a grammar the capability does not have fails fast with a `ContractError` before any candidate is produced. `Rule.__init_subclass__` also rejects an empty set at import time, since such a rule could never match a recognition. +- **`target_semantics: ClassVar[frozenset[str]]`** is the non-empty set of grammar `semantics` ids whose notations this rule validates. The engine uses it for affinity routing: each recognition is validated only by rules whose `target_semantics` includes the producing grammar's `semantics`, and a rule declaring a semantics id no grammar claims fails fast with a `ContractError` before any candidate is produced. `Rule.__init_subclass__` also rejects an empty set at import time, since such a rule could never match a recognition. - **`requires_features: ClassVar[frozenset[str]]`** is the set of Contract field names that must be truthy for the rule to run. An empty set is valid and is the common case: it means the rule always runs once selected. The engine validates that every named feature exists on the contract (a missing name raises `ContractError`) and applies the final feature filter *after* pinning, exclusion, and year selection: a rule whose required feature is present but `False` is dropped. **Feature gating has two loci, and they produce different `Resolution` statuses:** @@ -569,7 +570,7 @@ The key invariant: `None`, `"default"`, and the default format string are **trea Example — Date input `"01/02/2026"` is recognized by both the US and European grammars and validated by both rules, yielding two distinct canonical values (`2026-01-02` and `2026-02-01`). The result is `AMBIGUOUS` regardless of `output_format`. `output_format="US"` merely renders those two values as `01/02/2026` and `02/01/2026`; it cannot and must not decide which interpretation is "correct". -> Note: the grammar→rule routing decision (which rule validates which recognized notation) is an entirely separate concern from `output_format`. Routing is declared on the rule (e.g. `Rule.target_grammars`); it operates in the recognition→validation stage and never touches formatting. Keep the two orthogonal. +> Note: the grammar→rule routing decision (which rule validates which recognized notation) is an entirely separate concern from `output_format`. Routing is declared on the rule (`Rule.target_semantics`) and matched against each grammar's `semantics`; it operates in the recognition→validation stage and never touches formatting. Keep the two orthogonal. Example wiring — inherited from `CapabilityContract`, you only set the class variables: @@ -1032,7 +1033,7 @@ If your rule needs to read a capability-specific parameter (like `two_digit_base A shipped capability is closed for modification but open for extension: you can add recognition and validation without touching the capability package. -1. **Author** a `Grammar` subclass (Step 4) and a `Rule` subclass (Step 5) for the capability's notation, exactly as you would for a new capability — the same contracts apply, including span-bearing `RecognitionMatch` output, `target_grammars`, and `requires_features`. +1. **Author** a `Grammar` subclass (Step 4) and a `Rule` subclass (Step 5) for the capability's notation, exactly as you would for a new capability — the same contracts apply, including span-bearing `RecognitionMatch` output, `semantics` on the grammar, `target_semantics` on the rule, and `requires_features`. 2. **Register** them before the first `canonicalize()` call: ```python @@ -1053,7 +1054,7 @@ Semantics to rely on: - The extension registries freeze with the capability registry — registration after the first pipeline run raises `CapabilityError`. - Opt-in only: an un-named registered grammar never affects results, keeping shipped behavior byte-identical for non-opt-in contracts. -- Community rules are opt-in too: a registered rule runs only when the contract names one of its `target_grammars` in `extra_grammars`; an un-opted rule — even one targeting a shipped grammar — never affects results. +- Community rules are opt-in too: a registered rule runs only when the contract's `extra_grammars` resolve to one of its `target_semantics`; an un-opted rule — even one targeting a shipped grammar's semantics — never affects results. - Unknown `extra_grammars` names are silently skipped; shipped names listed in `extra_grammars` are deduplicated. - Composition is guarded: a grammar name colliding with a shipped name, or an opted-in community rule naming a missing grammar, fails fast at pipeline start. @@ -1066,7 +1067,7 @@ Use this checklist to verify your capability is complete: - [ ] Notation is a frozen dataclass with `as_list()` method - [ ] Each grammar extends `Grammar[YourDomainNotation]` and implements `recognize(text) -> list[RecognitionMatch[YourDomainNotation]]` - [ ] Each rule extends `Rule[YourDomainNotation]` and implements `matches(notation, contract) -> bool` and `normalize(notation, contract) -> str` -- [ ] Each rule declares `target_grammars` (non-empty `frozenset[str]`) and `requires_features` (`frozenset()` when the rule always runs) +- [ ] Each rule declares `target_semantics` (non-empty `frozenset[str]`) and `requires_features` (`frozenset()` when the rule always runs) - [ ] Each rule file has a `PUBLICATION` provenance constant - [ ] Capability extends `Capability` and implements `get_grammars()` and `get_rules()` - [ ] Contract inherits `CapabilityContract` (frozen dataclass, no `slots=True`) and satisfies the `Contract` protocol diff --git a/HOW_TO_ADD_NEW_GRAMMAR.md b/HOW_TO_ADD_NEW_GRAMMAR.md index e4e3dad4..173d7e56 100644 --- a/HOW_TO_ADD_NEW_GRAMMAR.md +++ b/HOW_TO_ADD_NEW_GRAMMAR.md @@ -18,9 +18,10 @@ Before starting, understand these concepts (all defined in depth in HOW_TO_ADD_N - **Rule** — a validation unit that checks a notation against an authoritative specification and produces the canonical value with provenance. Semantics. - **Notation** — the intermediate token grammars produce and rules consume. - **`active_grammars`** — the *optional* contract property naming which grammars run. A contract that does not implement it runs every shipped grammar returned by `get_grammars()`; only the gated capabilities (Email, IP, ISBN) implement it to name a subset. -- **`target_grammars`** — the rule metadata declaring which grammar(s) a rule validates. A recognition only routes to rules that name its producing grammar. +- **`semantics`** — the grammar metadata declaring the *meaning* the grammar assigns to its recognized notations: an identity id by default, or a coalesced id shared with grammars that carry the same meaning (e.g. both the ISO and slash-ISO Date grammars declare `"iso8601_calendar_date"`). +- **`target_semantics`** — the rule metadata declaring which grammar semantics a rule validates. A recognition only routes to rules whose `target_semantics` includes its producing grammar's `semantics`. -**The one sentence that matters:** a new grammar changes behavior only when it is (1) returned by `get_grammars()`, and (2) targeted by at least one rule via `target_grammars` — plus (3) named in `active_grammars`, but **only for the gated capabilities (Email, IP, ISBN) that implement it**. For the other six capabilities, the contract has no `active_grammars` and the engine runs every shipped grammar, so `get_grammars()` alone activates the new grammar. Miss any condition and the grammar silently never runs — so a shipped grammar ships with a test that proves the difference (Step 5). +**The one sentence that matters:** a new grammar changes behavior only when it is (1) returned by `get_grammars()`, and (2) its `semantics` is claimed by at least one rule via `target_semantics` — plus (3) named in `active_grammars`, but **only for the gated capabilities (Email, IP, ISBN) that implement it**. For the other six capabilities, the contract has no `active_grammars` and the engine runs every shipped grammar, so `get_grammars()` alone activates the new grammar. Miss condition (1) or (3) and the grammar never runs — input matching only it stays `MISSING`. Miss condition (2) and the grammar still runs, but its recognitions route to no rules, so input matching only it becomes `INVALID` (recognized, no authority rule validates) instead of resolving — never a candidate. Either way the resolved output is unchanged, which is why a shipped grammar ships with a test that proves the difference (Step 5). --- @@ -30,7 +31,7 @@ Before writing code, answer these questions: 1. **What new representation are you recognizing?** Write down examples of real human input, including edge cases (`2024/1/5`, not just `2024/01/01`). 2. **Does an existing notation already fit it?** If the representation decomposes into the same fields as an existing grammar (e.g., Date's `DateNotation(N1, N2, N3)`), reuse it. Only extend the notation when the new format genuinely carries different components. -3. **Does an existing rule already assign the same meaning?** If the new format means the same thing and normalizes the same way as an already-validated format (e.g., `2024/01/01` *is* ISO 8601's calendar date), you may only need to extend an existing rule's `target_grammars` (Step 4, option A). +3. **Does an existing rule already assign the same meaning?** If the new format means the same thing and normalizes the same way as an already-validated format (e.g., `2024/01/01` *is* ISO 8601's calendar date), declare that meaning's shipped `semantics` id on the new grammar and stop — no rule edit (Step 4, option A). 4. **Which recognition strategy fits the representation?** Every grammar follows one of two core strategies (see HOW_TO_ADD_NEW_CAPABILITY.md Step 4 for the extended set): @@ -90,7 +91,7 @@ Create `paxman/capabilities//grammar/_recognition.py`: 1. Import `Grammar` and `RecognitionMatch` from `paxman.core.domain`, and the capability's notation. 2. Compile the regex once at module scope — never inside `recognize()` (it runs for every input). -3. Define a class extending `Grammar[Notation]` with a `name` of the form `{format}_recognition` (snake_case, unique within the capability). +3. Define a class extending `Grammar[Notation]` with a `name` of the form `{format}_recognition` (snake_case, unique within the capability) and a non-empty `semantics` string declaring the meaning its notations carry. 4. Implement `recognize(text) -> list[RecognitionMatch[Notation]]` — one `RecognitionMatch` per `finditer()` hit, carrying the notation, the half-open `[start, end)` span, and `raw_text`. ```python @@ -106,13 +107,14 @@ import re from paxman.capabilities.Date.notation import DateNotation from paxman.core.domain import Grammar, RecognitionMatch -_SLASH_ISO_PATTERN = re.compile(r"(\d{4})/(\d{1,2})/(\d{1,2})") +_SLASH_ISO_PATTERN = re.compile(r"(? list[RecognitionMatch[DateNotation]]: """Extract YYYY/MM/DD date patterns from text.""" @@ -130,7 +132,7 @@ class SlashISODateGrammar(Grammar[DateNotation]): **The `recognize()` contract is enforced by the engine.** Every match must satisfy `0 <= start <= end <= len(text)` and `raw_text == text[start:end]`; a grammar returning a malformed match raises `RecognitionError` naming the grammar (see `_recognize` in `paxman/engine/orchestrator.py`). The engine owns within-grammar containment dedup and total recognition ordering — the grammar only extracts and emits spans. -**Guard boundaries against sibling grammars.** When two grammars could claim the same span, use lookarounds so each claims only its own representation. The slash-ISO pattern is naturally disjoint from the US/European grammars (a leading 4-digit year vs. a leading 1–2-digit month/day) — verify that with a negative test like `test_does_not_match_us_or_european_order`. +**Guard boundaries against sibling grammars.** When two grammars could claim the same span, use lookarounds so each claims only its own representation. The slash-ISO pattern is naturally disjoint from the US/European grammars (a leading 4-digit year vs. a leading 1–2-digit month/day) — verify that with a negative test like `test_does_not_match_us_or_european_order`. Digit lookarounds (`(? list[RecognitionMatch[DateNotation]]: @@ -455,7 +456,7 @@ class DotDateRule(Rule[DateNotation]): publication_year=2019, ) citation = "Section 4.3.1 (calendar date)" - target_grammars = frozenset({"dot_date_recognition"}) + target_semantics = frozenset({"dot_date_recognition"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: @@ -482,10 +483,10 @@ print(result.canonicalized_value) # → "2024-01-01" Rules of the seam: - **Register before the first `canonicalize()` call** — the extension registries freeze with the capability registry. -- **Opt-in only** — a registered grammar runs only when named in `extra_grammars` (available on every `create_contract` factory), and a registered rule runs only when the contract names one of its `target_grammars` there; un-named grammars and un-opted rules never affect results. -- **Unknown `extra_grammars` names are silently skipped**, so a contract naming an uninstalled grammar still runs byte-identically. +- **Opt-in only** — a registered grammar runs only when named in `extra_grammars` (available on every `create_contract` factory), and a registered rule runs only when the contract's `extra_grammars` resolve to one of its `target_semantics` ids; un-named grammars and un-opted rules never affect results. +- **Unknown `extra_grammars` names are silently skipped** for grammar activation, so a contract naming an uninstalled grammar still runs byte-identically; the unknown name is kept as-is as the semantics key for rule activation. A name that is not a grammar name but matches a known semantics id therefore activates that semantics's rules without opting in a grammar — those rules fire only on recognitions carrying that semantics (fail-fast `ContractError` applies only to ids no grammar claims). - **Names must be unique in the composed set** — a community grammar colliding with a shipped name fails fast with `CapabilityError`. -- **Community rules declare `target_grammars`** and activate only when the contract names one of them in `extra_grammars`; an opted-in rule naming a missing grammar fails fast with `ContractError`. +- **Community rules declare `target_semantics`** and activate only when the contract's `extra_grammars` resolve to one of those ids; a rule opted in via an id that no grammar claims fails fast with `ContractError`, while a rule that is not opted in stays inert regardless of any dangling targets. --- diff --git a/capability_homogeneity_audit.md b/capability_homogeneity_audit.md index 4170d2f0..a763d613 100644 --- a/capability_homogeneity_audit.md +++ b/capability_homogeneity_audit.md @@ -60,11 +60,11 @@ validate it." Each capability invented its own self-filter: This contradicts `ARCHITECTURE.md:201` ("Each grammar's notation flows to its corresponding validation rule"). -**Unanimous ideal:** declare affinity on the rule — `Rule.target_grammars: +**Unanimous ideal:** declare affinity on the rule — `Rule.target_semantics: frozenset[str]` (ClassVar, enforced by the existing `__init_subclass__` metadata -check); orchestrator adds one line `if grammar_name not in rule.target_grammars: -continue`. Engine stays capability-agnostic (reads declared names, no shape -knowledge). Replay-safe *if* `_collect_candidates` also dedups identical +check); orchestrator adds one line `if semantics_by_name[grammar_name] not in +rule.target_semantics: continue`. Engine stays capability-agnostic (reads declared +semantics ids, no shape knowledge). Replay-safe *if* `_collect_candidates` also dedups identical `(value, recognition_rule, validation_rule)` tuples — otherwise Date candidate multiplicity changes the hash (semantics/status unchanged). Preserves Date ambiguity (each rule sees only its grammar's notation → 2 candidates → @@ -230,7 +230,7 @@ Ranked defects (from the rule-comparison agent): ## Unanimous ideals — recommended build order -1. **`Rule.target_grammars`** + one-line orchestrator filter (fixes F1; makes +1. **`Rule.target_semantics`** + one-line orchestrator filter (fixes F1; makes `ARCHITECTURE.md:201` true; deterministic output with candidate dedup). 2. **Move `include_*` feature-gating** to engine-enforced declared metadata, split by feature kind (fixes F2). **[DONE 2026-08-03]** — enforced via @@ -293,7 +293,7 @@ formatter seam described in the centralize-output-format addendum below. ### B. F1 candidate multiset is identical — byte-identical output requires ordered comparison The audit's *Watch out for* warns that moving to grammar→rule affinity could change -the candidate multiset via candidate multiplicity. In practice `target_grammars` was set equal +the candidate multiset via candidate multiplicity. In practice `target_semantics` was set equal to each rule's *effective acceptance domain* (the affinity map in F1), so the candidate multiset is identical to the cartesian product for every capability (Email / Date / Country / IP / Phone). That multiset equality alone does not prove byte-identical output: @@ -304,7 +304,7 @@ rules and provenance — as asserted by the repeated-run determinism tests (e.g. `ExecutionResult` equality). Where only multiset equality is established, the claim is limited to multiset equality, not byte-identity. The `_dedup_candidates` step is a pure safety net for future over-declaration, not a behavior change. The plan's Step 6.7 -determinism gate was satisfied *structurally* (target_grammars == effective domain ⇒ +determinism gate was satisfied *structurally* (target_semantics == effective domain ⇒ identical multiset) rather than by captured pre-change constants; the full 782-test suite passing is the empirical confirmation. diff --git a/docs/adr/0003-semantic-affinity-routing.md b/docs/adr/0003-semantic-affinity-routing.md new file mode 100644 index 00000000..3cacf74b --- /dev/null +++ b/docs/adr/0003-semantic-affinity-routing.md @@ -0,0 +1,260 @@ +# ADR-0003: Semantic Affinity — Route Rules by Meaning, Not Grammar Name + +## Status + +Accepted + +## Context + +Paxman separates Recognition from Validation: grammars produce span-bearing +`RecognitionMatch`es carrying a notation; rules validate that notation against +an authoritative specification and emit the canonical value with provenance. +The two layers are joined by **grammar-name affinity**: `Rule.target_grammars` +declares the grammar names whose output the rule understands, and the engine +routes a recognition only to rules that name its producing grammar. + +This affinity is enforced at three engine sites and one domain site: + +- `Rule.__init_subclass__` requires a non-empty `target_grammars` + `frozenset[str]` at class-definition time (`paxman/core/domain.py`). +- `_validate_affinity` fails fast with `ContractError` when a rule names a + grammar absent from the composed set (`paxman/engine/orchestrator.py`). +- `_collect_candidates` routes each recognition to rules via + `grammar_name not in rule.target_grammars`. +- `_activated_rules` opts a community rule in only when the contract names one + of its `target_grammars` in `extra_grammars`. + +The affinity exists to prevent the F1 defect (the cartesian product: every rule +validating every grammar's matches regardless of whether the rule's authority +covers that representation), which the homogeneity audit ranked as defect #1 +and which PR #19 fixed via `target_grammars` + the one-line orchestrator +filter. + +The name-based binding has a structural cost: **a new grammar is effective +only when a rule file is edited.** The shipped proof is the slash-ISO grammar +(`SlashISODateGrammar`, `YYYY/MM/DD`): it shares ISO 8601's position mapping +and canonical form, but activating it required extending the ISO rule's +`target_grammars` frozenset by hand. The grammar author's work was complete at +`RecognitionMatch`; the rule edit exists only to declare what the grammar +cannot declare today — *what its notation means*. + +This violates the responsibility boundary the grammar is otherwise held to. +Grammars are prohibited from importing rule-layer data and from assigning +canonical meaning; they are required to end at the span-bearing match. But the +one thing that would let a grammar stand alone — a statement of its own +semantics — has no home in the `Grammar` ABC, which carries only `name`. + +The same-notation-type hazard that motivates affinity is real and must +survive any redesign: `DateNotation(N1, N2, N3)` means `(year, month, day)` +from `iso8601_recognition` but `(month, day, year)` from `us_recognition`. +Affinity must remain *explicit*, it must remain fail-fast on dangling +declarations, and it must remain deterministic per contract. + +## Decision + +Replace grammar-name affinity with **semantic affinity**: a rule targets the +*meaning* of the notations it validates, and each grammar declares that +meaning. `Rule.target_grammars` is **replaced** by +`Rule.target_semantics`; `Grammar` gains a required `semantics` identifier. + +### 1. `Grammar.semantics` — the meaning claim + +`Grammar` (in `paxman/core/domain.py`) gains a required class attribute: + +```python +class Grammar(ABC, Generic[NotationT]): + """Base class for recognition grammars.""" + + name: str + semantics: ClassVar[str] +``` + +- `semantics` is a stable identifier for the *meaning* the grammar's notation + carries — what the notation says, not how it is written. +- It is enforced by a new `Grammar.__init_subclass__` mirroring `Rule`'s + metadata enforcement: non-empty `str`, required at class-definition time. +- It is a **claim**, not an interpretation: the grammar still never imports + rule-layer data, never validates, and never maps tokens to canonical values. + The semantic purity boundary is unchanged; `semantics` only names the + meaning the grammar already produces. +- Example: `iso8601_recognition` and `slash_iso_recognition` both declare + `semantics = "iso8601_calendar_date"`; `us_recognition` declares + `semantics = "us_calendar_date"`. + +### 2. `Rule.target_semantics` — replace `target_grammars` + +`Rule.target_grammars: ClassVar[frozenset[str]]` becomes +`Rule.target_semantics: ClassVar[frozenset[str]]` with identical enforcement +(non-empty `frozenset[str]`, checked by `Rule.__init_subclass__`). The ISO +rule's declaration collapses from two grammar names to one meaning: + +```python +# before +target_grammars = frozenset({"iso8601_recognition", "slash_iso_recognition"}) +# after +target_semantics = frozenset({"iso8601_calendar_date"}) +``` + +### 3. Engine routing — three one-line sites + +- `_validate_affinity`: validate `rule.target_semantics` against the composed + **semantics set** `{g.semantics for g in all_grammars}` instead of the + grammar-name set. Dangling semantics still fail fast with `ContractError`. +- `_collect_candidates`: route via + `recognition.grammar.semantics not in rule.target_semantics`. The engine + builds a `semantics_by_name` map at composition time so the producing + grammar's semantics is resolvable from the `RecognizedRep`'s + `GrammarRule.grammar_name`. +- `_activated_rules`: a community rule activates when the contract names, in + `extra_grammars`, any grammar whose `semantics` is in the rule's + `target_semantics` — the opt-in discipline is unchanged, keyed on meaning. + +### 4. Provenance and dedup stay name-based + +`Candidate.recognition_rule` continues to record the **grammar name** +(`grammar_name`), and `_dedup_candidates` continues to collapse on +`(value, recognition_rule, validation_rule)`. `semantics` is routing metadata; +the grammar name remains the audit identity of the recognition. Provenance +output is byte-identical to today for every existing input, excluding +digit-glued dates — the date grammars' lookaround bounds were tightened +post-ADR (see the Migration note). + +### 5. What a grammar-only addition now means + +Adding a grammar whose meaning is already shipped becomes a single-file +change: the grammar file itself. It declares `semantics = ""` +and is validated by the existing rule automatically — no rule edit, no +`target_*` change. This is the property that makes the grammar's +responsibility end at the `RecognitionMatch` (plus its one-word meaning +claim). A genuinely new meaning still requires a new rule: the canonical +value and provenance must come from an authoritative specification, and that +is rule territory by design. + +## Migration + +1. **Phase 1 — behavior-preserving rename.** Every shipped grammar declares + `semantics = ""`; every rule's `target_grammars` is renamed + `target_semantics` with the identical set. Routing keyed on semantics with + `semantics == name` is byte-identical to name routing; the full pre-PR gate + (ruff, pyright, import-linter, pytest) stays green with no test edits. + Community rule metadata (`target_grammars` → `target_semantics`) is a + breaking rename at 0.x, acceptable under the same policy as ADR-0002. + *Post-ADR correction:* during Migration #4 the four date grammars' + digit-lookaround bounds (`(?= 1.0. +- **New required metadata on every grammar.** `semantics` is a contributor + touch point in `HOW_TO_ADD_NEW_GRAMMAR.md`; a grammar without it fails at + class-definition time rather than at runtime. +- **Wrong `semantics` declarations are silent until a test catches them.** A + grammar claiming a shipped `semantics` id with divergent field mapping + would route to the wrong rule and produce a wrong-but-plausible canonical + value. The consistency guard is the mitigation; it is test-time, not + runtime. + +### Risks + +- **Same-semantics, different-meaning collisions.** Two grammars declaring the + same `semantics` with divergent field mapping silently mis-canonicalize. + Mitigation: the consistency-guard test per `semantics` id (Migration #3), + and a documented convention that `semantics` is a *semantic contract*: all + grammars claiming it must agree on notation field order and canonical form. +- **Coalescing drift during Phase 2.** Coalescing `target_semantics` sets must + never widen the set of meanings a rule validates. Mitigation: each + coalescing step runs the per-capability pipeline tests; the migration lands + one capability at a time. +- **Incomplete docs sweep.** 55 files referenced `target_grammars` at plan + time (pre-migration inventory; the plan's D10 re-count is authoritative — + superseded by the completed Migration #4 sweep). Mitigation: this ADR lands + before code; the sweep is enumerated in Migration #4. + +## Alternatives Considered + +1. **Keep `target_grammars` (status quo).** Rejected — a new grammar is + effective only when a rule file is edited, even when the grammar's meaning + is already validated. The slash-ISO one-line rule extension is the proof; + the coupling is name-based, not meaning-based, so it forces rule churn for + pure recognition additions. +2. **Additive `target_semantics` alongside `target_grammars` (OR routing).** + Rejected — two routing dimensions, ambiguous precedence, and a rule could + silently widen its authority by naming a semantics id without removing + stale grammar names. Replacement keeps a single, auditable affinity + declaration. +3. **Type-only routing.** Rules accept any notation of their declared type. + Rejected — the same `DateNotation` type carries different meanings per + grammar (US vs ISO field mapping); this is the F1 defect's failure mode, + not a fix for it. +4. **`same_semantics_as` pointer on the grammar.** A grammar references an + existing grammar whose meaning it shares. Rejected — requires equivalence + classes with canonical representatives, and validation of a reference + graph (acyclicity, consistency); a self-contained `semantics` id names the + meaning directly and avoids the representative problem. +5. **Infer meaning from the notation type.** Rejected — impossible in the + general case; meaning is not recoverable from `DateNotation(N1, N2, N3)` + without knowing the producing grammar's field mapping. `semantics` is the + minimal explicit claim that makes the inference sound. + +## References + +- ADR-0001 — Key Design Decisions #1 (Separate Recognition from Validation), + #2 (Notation as Internal Contract) — reaffirmed; the grammar purity + boundary this ADR extends +- ADR-0002 — breaking-change-at-0.x policy precedent +- `docs/superpowers/plans/2026-08-03-f1-grammar-rule-affinity.md` — the F1 + defect and the `target_grammars` fix this ADR generalizes to semantics +- `capability_homogeneity_audit.md` — F1 cartesian-product defect (#1), the + audit this ADR builds on +- `paxman/core/domain.py` — `Rule.__init_subclass__` (`target_grammars` + enforcement, replaced), `Grammar` (gains `semantics`), `GrammarRule` +- `paxman/engine/orchestrator.py` — `_validate_affinity`, + `_collect_candidates`, `_activated_rules` (routing sites, changed) +- `paxman/core/extensions.py` — `register_grammar` / `register_rule` + community seam (activation keyed on semantics) +- `HOW_TO_ADD_NEW_GRAMMAR.md` — Step 4 (rule-affinity requirement, replaced + by the semantics declaration) +- `README.md` — Community Extensions section (activation rule reworded) diff --git a/docs/superpowers/plans/2026-08-11-adr0003-semantic-affinity-routing.md b/docs/superpowers/plans/2026-08-11-adr0003-semantic-affinity-routing.md new file mode 100644 index 00000000..43fb39f8 --- /dev/null +++ b/docs/superpowers/plans/2026-08-11-adr0003-semantic-affinity-routing.md @@ -0,0 +1,772 @@ +# ADR-0003 Semantic Affinity Routing — Implementation Plan + +| **Title** | Route rules by meaning, not grammar name | +| **Date** | 2026-08-11 | +| **Status** | In progress — Tasks 1-2 landed, Tasks 3-10 pending | +| **Branch** | `refactor/semantic-affinity-routing` (commit per task) | +| **Authoritative spec** | `docs/adr/0003-semantic-affinity-routing.md` — where this plan and the ADR disagree on design, the ADR wins; verified file inventories (D10) supersede the ADR's pre-migration snapshot | +| **Supersedes** | `Rule.target_grammars` grammar-name affinity (F1 fix, PR #19) — the routing key becomes `semantics` | + +> **For agentic workers.** This plan is written to be executed by a worker +> agent one task at a time. Every task is TDD: **Step 1 RED** (write/adjust +> the failing test first), **Step 2 GREEN** (make it pass), then the scoped +> verify command and the commit. Do not skip steps, do not reorder tasks, do +> not "improve" the design — D-decisions are locked (§1). The full suite is +> only green after Task 4; the per-task verify commands are scoped so each +> task is independently green. Commit with the exact message given for each +> task. **Pure-mechanical tasks are exempt from a meaningful RED step** — +> Tasks 2, 4, and 9 say so explicitly and their instruction wins. Task 10 is +> a verify-only gate with **no commit**, so the exact-commit-message +> requirement does not apply to it. + +> **Progress — COMPLETE.** All ten tasks landed. +> +> | Task | Status | Commit | +> |------|--------|--------| +> | Task 1 — declare `semantics` on all shipped grammars | ✅ landed | `ebf21bb` | +> | Task 2 — rename `target_grammars` → `target_semantics` | ✅ landed | `f8ac1f6` | +> | Task 3 — enforce `Grammar.semantics` at class-definition time | ✅ landed | `0ea7727` | +> | Task 4 — engine routes on semantics | ✅ landed | `0f0f9d1` | +> | Task 5 — Phase 2: Date coalescing | ✅ landed | `3c592b8` | +> | Task 6 — Phase 2: Email coalescing | ✅ landed | `9317596` | +> | Task 7 — Phase 2: Phone coalescing | ✅ landed | `abae3e4` (`fd22152` prep fix) | +> | Task 8 — consistency guard | ✅ landed | `36315dc` | +> | Task 9 — docs sweep | ✅ landed | `f6f7790` | +> | Task 10 — final gate (no commit) | ✅ green | follow-up `125c06d` | +> +> Execution notes for the follow-up session that ran Tasks 5-10: +> +> - Task 5 also updated `tests/unit/test_grammar_shipped_grammars_declare_semantics_identity` +> (now `test_shipped_grammars_declare_semantics_identity`) via a module-level +> `_COALESCED_SEMANTICS` allowlist, so every commit stayed green — approved +> deviation; the allowlist mirrors D6 exactly and is itself locked by the +> guard's enumeration-completeness test. +> - M1/M2 (orchestrator docstring KeyError invariant + drop the duplicate +> sentence at ARCHITECTURE.md:201) and M3 (README fail-fast mechanism, from +> Task 5 execution) were folded into Task 9. Task 5's RED driver 2 failed +> with a silent-exclusion `assert False`, not the plan-predicted +> `ContractError` — README's claim is accurate and now documented precisely. +> - Task 8's multi-member group count is scoped per capability (Currency and +> Money share `code_recognition`/`symbol_recognition`/`word_recognition` +> across capabilities; routing is per-capability, so groups are enumerated +> per capability). +> - Task 10 gate: `ruff format --check .` flags 8 pre-existing historical +> docs (`docs/research/*`, `docs/superpowers/plans/*` — untouched on this +> branch, forbidden to edit, out of CI scope); the CI-authoritative gate +> (`ruff check paxman/ tests/ && ruff format --check paxman/ tests/`) is +> fully green. All ten plan tasks are done — do not re-execute them. + +--- + +## §1 Cross-Part Contract + +### Goal + +Implement ADR-0003: replace grammar-name affinity with **semantic affinity**. +`Grammar` gains a required `semantics: ClassVar[str]` enforced by a new +`Grammar.__init_subclass__`; `Rule.target_grammars` is **replaced** by +`Rule.target_semantics` (identical enforcement); the engine routes on +semantics at all three sites (`_validate_affinity`, `_collect_candidates`, +`_activated_rules`). Phase 1 is a byte-identical rename (`semantics == name` +for every shipped grammar); Phase 2 coalesces same-meaning grammars in Date, +Email, and Phone; a consistency-guard test locks same-semantics/same-field- +mapping; docs are swept. Provenance and candidate dedup stay name-based — +output is byte-identical to today for every existing input, excluding the +digit-glued date class (post-plan lookaround tightening; see Out of scope). + +### D-Decisions (locked — do not revisit without a new ADR) + +- **D1 — Phase 1 identity: `semantics == name` for all 26 shipped + grammars.** Routing keyed on semantics with `semantics == name` is + set-equal to name routing, so behavior is byte-identical (ADR Migration + #1). The identity is locked by a test (Task 1), not by convention. +- **D2 — The rename is atomic.** `target_grammars` → `target_semantics` + lands in ONE commit across `domain.py` (Rule ABC), `orchestrator.py`, the + 21 rule files (29 declarations), `extensions.py` (docstring), and the 12 + test files (47 hits). Any split-brain state is an import-time `TypeError` + from `Rule.__init_subclass__` — the sweep must be atomic (ADR-0002 plan + trap #3 precedent). +- **D3 — Enforcement mirrors `Rule`'s.** `Grammar.__init_subclass__` + requires `semantics` as a non-empty `str`, checked at class-definition + time. The ABC annotates `semantics: ClassVar[str]`; subclasses declare + bare `semantics = "..."` (matching the existing `name = "..."` style). + Inherited values satisfy the check (use `vars(cls).get(attribute, + getattr(cls, attribute))` exactly like `Rule` at domain.py:211) — so + `_CountingLongGrammar(_ProbeLongGrammar)` in + `tests/integration/test_recognition_seam.py` needs no edit. +- **D4 — Test doubles updated in the same commit as enforcement.** Adding + `Grammar.__init_subclass__` import-fails every test-defined `Grammar` + subclass lacking `semantics` (18 classes in 8 files). They declare + `semantics == ` (Phase-1 identity) in the same commit as the + enforcement lands (Task 3). +- **D5 — Engine routing.** `_validate_affinity` validates + `rule.target_semantics` against the composed semantics set + `{g.semantics for g in all_grammars}`; `_collect_candidates` routes via a + `semantics_by_name: dict[str, str] = {g.name: g.semantics ...}` map built + at composition time (`recognition.grammar.grammar_name` → semantics → + membership in `rule.target_semantics`); `_activated_rules` activates a + community rule when any extra-named grammar's semantics is in the rule's + `target_semantics`. Provenance (`recognition_rule`/`validation_rule`) and + `_dedup_candidates` stay name-based, unchanged (ADR §4). +- **D6 — Phase 2 coalescing scope.** Coalesce exactly three groups, one + capability per task, each verified by the per-capability pipeline tests: + Date `iso8601_recognition` + `slash_iso_recognition` → + `"iso8601_calendar_date"` (ADR's worked example), Email + `standard_recognition` + `obfuscated_recognition` → `"rfc5322_addr_spec"`, + Phone `e164_recognition` + `international_00_recognition` → + `"e164_international"`. Within Date, `us_recognition` → + `"us_calendar_date"` and `european_recognition` → + `"european_calendar_date"` (the ADR's own id vocabulary); the two Date + rules targeting `{"us_recognition","european_recognition"}` become + `{"us_calendar_date","european_calendar_date"}` — a two-element set + becomes a two-id set, **no widening**. +- **D7 — No-coalesce set (locked, widening is the failure mode).** Date + US/European stay separate (divergent field mapping — the F1 hazard the + ADR exists to prevent). Country `iso_3166_historical_ed2020.py` keeps + three distinct ids (one rule, three shapes). **ISBN is NOT coalesced**: + `isbn13_recognition` and `isbn10_recognition` keep identity ids because + the check-digit authorities differ (ISO 2108 mod-10 vs Users' Manual + mod-11); collapsing them would widen `iso_2108_ed2017.py`'s authority to + ISBN-10 input — exactly the "never widen" drift ADR risk #2 forbids. The + range rule (`isbn_range_message_ed2026.py`) keeps both ids. All other + capabilities (Country, Currency, IP, Money, URL) keep identity ids — + their 1:1 grammar↔rule mapping means the identity id already names the + meaning; renaming is cosmetic churn outside the ADR's migration scope. +- **D8 — Consistency guard is test-time, generic, and per-task-extended.** + A new `tests/unit/test_grammar_semantics_consistency.py` groups every + shipped grammar class by its declared `semantics` and, for each + multi-member group, asserts identical notation field mapping + + canonicalization expectations over shared probe rows. Modeled on the + `test_grammar_semantic_purity.py` precedent. Written at Task 5 with Date + probe rows (its first real subject), extended with Email/Phone rows in the + same commits as those coalescings — the new group's lookup failing before + coalescing is each task's RED step. Singleton groups pass by construction. +- **D9 — Docs sweep.** 6 in-scope files at repo root (32 references) + the + 2 nested AGENTS.md under `paxman/` are swept (ADR Migration #4). Historical + records are excluded (ADR-0002 precedent): `docs/superpowers/plans/*`, + `docs/report/*`, `docs/research/*`, `docs/adr/*`. The zero-grep proof must + exclude generated dirs (`htmlcov/`, `.hypothesis/`, `.pytest_cache/`, + `.venv/`) or it fails on a dirty local checkout. +- **D10 — Ground truth over ADR claims.** The ADR's "55 files reference + `target_grammars`" (L205) was a pre-migration estimate. Verified by + re-counting at the migration start commit: **55 files** — 44 sweep-relevant + (26 under `paxman/` — 24 `.py` + 2 nested AGENTS.md, 12 test files, 6 + repo-root doc files, `HOW_TO_ADD_NEW_GRAMMAR.md` included) plus 8 plan + 3 + research files excluded by D9. Where the ADR and this plan disagree on + counts/paths, this plan's verified inventory wins (the ADR's count is + superseded as stale). + +### Out of scope + +- No behavior change to recognition/validation/status semantics (Phase 1 is + byte-identical; Phase 2 coalesces declarations only). Post-plan correction: + the date grammars' digit-lookaround bounds were tightened so digit-glued + ids like `12026-01-15` no longer partially match — deliberate, so "no + behavior change" excludes that digit-glued class only. +- No rename of `GrammarRule.grammar_name`, `RecognizedRep`, `Candidate`, + or `_dedup_candidates` keys — provenance stays name-based (ADR §4). +- No edits to historical plans/research/ADR files (D9). +- No new runtime semantics introspection by the engine (ADR Migration #3 — + the consistency guard is test-time only). +- No semantic-id renaming outside the coalescing capabilities (D6/D7). + +--- + +## §2 Tasks + +### Task 1 — `feat(core): declare semantics on all shipped grammars` + +> ✅ **LANDED** — commit `ebf21bb` (2026-08-11). Do not re-execute; the +> progress banner in §2's header supersedes this task's steps. + +Phase 1 identity: every shipped grammar declares `semantics = ""`. No enforcement yet — these are inert attributes until Task 3. + +**Step 1 RED — new file `tests/unit/test_grammar_semantics_metadata.py`** +- Add `test_shipped_grammars_declare_semantics_identity`: for each of the + nine shipped capabilities, for each grammar returned by + `capability.get_grammars()`: assert `isinstance(grammar.semantics, str)`, + `grammar.semantics != ""`, and `grammar.semantics == grammar.name`. +- Mark with the `unit` marker. +- Run: `uv run pytest tests/unit/test_grammar_semantics_metadata.py -q` → + RED (`AttributeError: ... has no attribute 'semantics'`). + +**Step 2 GREEN — add `semantics = ""` to all 26 shipped grammar files** +(one bare class attribute each, placed after the `name` declaration, +mirroring the existing `name = "..."` style): + +| Capability | File | Grammar class | `semantics` value | +|------------|------|---------------|-------------------| +| Country | `grammar/alpha2_recognition.py` | `Alpha2Grammar` | `"alpha2_recognition"` | +| Country | `grammar/alpha3_recognition.py` | `Alpha3Grammar` | `"alpha3_recognition"` | +| Country | `grammar/numeric_recognition.py` | `NumericGrammar` | `"numeric_recognition"` | +| Country | `grammar/name_recognition.py` | `NameGrammar` | `"name_recognition"` | +| Currency | `grammar/code_recognition.py` | `CodeRecognition` | `"code_recognition"` | +| Currency | `grammar/symbol_recognition.py` | `SymbolRecognition` | `"symbol_recognition"` | +| Currency | `grammar/word_recognition.py` | `WordRecognition` | `"word_recognition"` | +| Date | `grammar/iso8601_recognition.py` | `ISO8601DateGrammar` | `"iso8601_recognition"` | +| Date | `grammar/us_recognition.py` | `USDateGrammar` | `"us_recognition"` | +| Date | `grammar/european_recognition.py` | `EuropeanDateGrammar` | `"european_recognition"` | +| Date | `grammar/slash_iso_recognition.py` | `SlashISODateGrammar` | `"slash_iso_recognition"` | +| Email | `grammar/standard_recognition.py` | `StandardEmailGrammar` | `"standard_recognition"` | +| Email | `grammar/obfuscated_recognition.py` | `ObfuscatedEmailGrammar` | `"obfuscated_recognition"` | +| Email | `grammar/localhost_recognition.py` | `LocalhostEmailGrammar` | `"localhost_recognition"` | +| IP | `grammar/ipv4_recognition.py` | `IPv4Grammar` | `"ipv4_recognition"` | +| IP | `grammar/ipv6_recognition.py` | `IPv6Grammar` | `"ipv6_recognition"` | +| ISBN | `grammar/isbn13_recognition.py` | `ISBN13RecognitionGrammar` | `"isbn13_recognition"` | +| ISBN | `grammar/isbn10_recognition.py` | `ISBN10RecognitionGrammar` | `"isbn10_recognition"` | +| Money | `grammar/code_recognition.py` | `CodeRecognition` | `"code_recognition"` | +| Money | `grammar/symbol_recognition.py` | `SymbolRecognition` | `"symbol_recognition"` | +| Money | `grammar/word_recognition.py` | `WordRecognition` | `"word_recognition"` | +| Phone | `grammar/e164_recognition.py` | `E164Grammar` | `"e164_recognition"` | +| Phone | `grammar/tel_uri_recognition.py` | `TelUriGrammar` | `"tel_uri_recognition"` | +| Phone | `grammar/international_00_recognition.py` | `International00Grammar` | `"international_00_recognition"` | +| Phone | `grammar/national_recognition.py` | `NationalGrammar` | `"national_recognition"` | +| URL | `grammar/absolute_uri_recognition.py` | `AbsoluteUriRecognition` | `"absolute_uri_recognition"` | + +All files are under `paxman/capabilities//`. + +**Verify** +```bash +uv run pytest tests/unit/test_grammar_semantics_metadata.py -q +uv run pytest -q +uv run ruff check paxman/capabilities/ tests/unit/test_grammar_semantics_metadata.py +``` + +**Commit** +``` +feat(core): declare semantics on all shipped grammars +``` + +--- + +### Task 2 — `refactor: rename target_grammars to target_semantics` + +> ✅ **LANDED** — commit `f8ac1f6` (2026-08-11). Do not re-execute; the +> progress banner in §2's header supersedes this task's steps. Zero +> `target_grammars` hits remain in `paxman/` or `tests/` (verified). + +The atomic rename (D2). Pure mechanical sweep — **no RED test**; the RED +state is the intermediate breakage demonstrated in Step 1 below, fixed by +Step 2 in the same commit. + +**Step 1 RED (demonstrate breakage — do not commit)** +- In `paxman/core/domain.py` only, rename the five `target_grammars` sites in + `Rule` (L191 annotation, L202 required tuple, L210 type-check loop, L218 + non-empty guard, L219 error message) to `target_semantics`. +- Run: `uv run pytest tests/unit/test_rule_metadata.py -q` → RED + (`TypeError: must define Rule metadata`) and + `uv run pyright paxman/capabilities/` → RED. This proves the sweep below + is mandatory and atomic. + +**Step 2 GREEN — sweep every remaining site in one commit** + +Source (`paxman/`): +| File | Sites | +|------|-------| +| `paxman/core/domain.py` | already renamed in Step 1 (5 sites) | +| `paxman/engine/orchestrator.py` | L287 (`_validate_affinity` read), L312 (`_collect_candidates` route), L399 (`_activated_rules` intersection), plus docstrings L302, L346, L389 | +| `paxman/core/extensions.py` | docstring L71 (`register_rule`) | + +Rule files — rename the attribute name in all 29 declarations (values +unchanged — Phase 1 identity): + +| File | Decl lines | Set value (unchanged) | +|------|-----------|-----------------------| +| `Country/rules/iso_3166_ed2024.py` | L61, L101, L141, L189 | `{"alpha2_recognition"}` / `{"alpha3_recognition"}` / `{"numeric_recognition"}` / `{"name_recognition"}` | +| `Country/rules/cldr_localized_ed2025.py` | L48 | `{"name_recognition"}` | +| `Country/rules/iso_3166_historical_ed2020.py` | L69-71 (multi-line) | `{"name_recognition","alpha2_recognition","numeric_recognition"}` | +| `Currency/rules/iso_4217_ed2015.py` | L45 | `{"code_recognition"}` | +| `Currency/rules/cldr_currencies_ed2025.py` | L125, L169 | `{"symbol_recognition"}` / `{"word_recognition"}` | +| `Date/rules/iso_8601_ed2019.py` | L36 | `{"iso8601_recognition","slash_iso_recognition"}` | +| `Date/rules/en_50160_ed2010.py` | L33 | `{"us_recognition","european_recognition"}` | +| `Date/rules/us_federal_rules_ed2023.py` | L33 | `{"us_recognition","european_recognition"}` | +| `Email/rules/rfc_5322_ed2008.py` | L36 | `{"standard_recognition","obfuscated_recognition"}` | +| `Email/rules/rfc_6761_ed2012.py` | L36 | `{"localhost_recognition"}` | +| `IP/rules/rfc_791_ed1981.py` | L33 | `{"ipv4_recognition"}` | +| `IP/rules/rfc_5952_ed2010.py` | L34 | `{"ipv6_recognition"}` | +| `ISBN/rules/iso_2108_ed2017.py` | L32, L55 | `{"isbn13_recognition"}` (both) | +| `ISBN/rules/isbn_users_manual_ed2012.py` | L34 | `{"isbn10_recognition"}` | +| `ISBN/rules/isbn_range_message_ed2026.py` | L37 | `{"isbn13_recognition","isbn10_recognition"}` | +| `Money/rules/iso_4217_ed2015.py` | L76 | `{"code_recognition"}` | +| `Money/rules/cldr_currencies_ed2025.py` | L136, L199 | `{"symbol_recognition"}` / `{"word_recognition"}` | +| `Phone/rules/rfc_3966_ed2004.py` | L34 | `{"tel_uri_recognition"}` | +| `Phone/rules/e164_ed2010.py` | L69, L111 | `{"e164_recognition","international_00_recognition"}` (both) | +| `Phone/rules/nanp_ed2024.py` | L93, L148 | `{"national_recognition"}` (both) | +| `URL/rules/whatwg_url_standard.py` | L43 | `{"absolute_uri_recognition"}` | + +All under `paxman/capabilities//`. + +Tests — rename the attribute everywhere (values unchanged). 12 files, 47 +hits: +| File | Sites | +|------|-------| +| `tests/unit/test_capability.py` | L38 `StubRule` | +| `tests/unit/test_extensions.py` | L85 `_DotDateRule`, L102 `_NamelessRule` | +| `tests/unit/test_rule_metadata.py` | `_RULE_METADATA_ATTRS` (L15-22, `"target_grammars"` at L20 → `"target_semantics"`), L97 conditional `if missing != "target_grammars":`, L118 class `_EmptyTargetGrammars` → `_EmptyTargetSemantics`, `match=` regexes embedding the attribute name (L85, L111 `"non-empty"`, L152 `"must be frozenset[str]"`) | +| `tests/integration/test_feature_gating.py` | L148 `_DanglingFeatureRule`, L214 `_DanglingGrammarRule`, docstring L58 | +| `tests/integration/test_recognition_seam.py` | L101 `_LongRule`, L126 `_ShortRule`, L403 `_CommunityRule` | +| `tests/integration/test_grammar_extensions.py` | L89 `DotDateRule`, L112 `SecondDateRule`, L135 `CommunityISO8601Rule`, L155 `DanglingDateRule`, docstring L149 | +| `tests/integration/test_format_value_seam.py` | L78 `_TokenRule`, L180 `_DualTokenRule` | +| `tests/integration/test_pipeline.py` | L167 `StubRule`, L192 `ExplodingRule`, L388 `_PhantomRule`, docstrings L374, L440 | +| `tests/capabilities/isbn/test_rules.py` | L182, L188, L194, L200-202 (`TestRuleConventions`) | +| `tests/capabilities/money/test_rules.py` | L145-147, L263-265, L378-380 + method names `test_target_grammars` → `test_target_semantics` | +| `tests/capabilities/currency/test_rules.py` | L100-102, L239-241, L309-311 + method names `test_target_grammars` → `test_target_semantics` | +| `tests/capabilities/url/test_rule.py` | L47 | + +Rule metadata values are grammar names today and remain grammar-name strings +in Phase 1 (D1/D2) — the value mapping is deferred to Phase 2 tasks. + +**Verify** +```bash +uv run pytest -q +uv run ruff check paxman/ tests/ +uv run pyright +``` +Also confirm the sweep is complete (zero hits inside `paxman/` and `tests/`): +```bash +grep -rn "target_grammars" paxman/ tests/ +``` +(No `|| echo "CLEAN"` fallback — zero matches prints nothing and exits 1; a +grep error exits ≥ 2 and stays visible.) + +**Commit** +``` +refactor: rename target_grammars to target_semantics +``` + +--- + +### Task 3 — `feat(core): enforce Grammar.semantics at class-definition time` + +The enforcement mirror (D3), landing together with the test-double sweep +(D4) — import-time failure otherwise. + +**Step 1 RED — extend `tests/unit/test_grammar_semantics_metadata.py`** +- Add `test_bare_grammar_subclass_raises_type_error`: a local + `class _BareGrammar(Grammar[Any])` with only `name` and `recognize()` + raises `TypeError` matching `"must define Grammar metadata"` (or the + exact message chosen in Step 2 — write the test to match it). +- Add `test_missing_semantics_raises`: parametrized — grammar with + everything but `semantics` → `TypeError` naming `semantics`. +- Add `test_empty_semantics_raises`: `semantics = ""` → `TypeError` + matching `"non-empty"`. +- Add `test_semantics_must_be_str`: `semantics = 42` / `semantics = + frozenset()` → `TypeError` matching `"semantics must be str"`. +- Add `test_inherited_semantics_satisfies_enforcement`: a subclass of a + compliant grammar (no own `semantics`) is accepted — locks the + `vars(cls).get` fallback (D3), covering `_CountingLongGrammar`. +- Run: `uv run pytest tests/unit/test_grammar_semantics_metadata.py -q` → + RED (no `__init_subclass__` yet). + +**Step 2 GREEN — one commit containing all three:** +1. `paxman/core/domain.py` — add to `Grammar` (L228-240): + - `semantics: ClassVar[str]` annotation (next to `name: str`). + - `__init_subclass__` mirroring `Rule`'s (L194-219): require `semantics` + via `hasattr`; type-check `vars(cls).get("semantics", + getattr(cls, "semantics"))` is `type(...) is str`; non-empty guard. + Error messages: `f"{cls.__name__} must define Grammar metadata: + semantics"`, `f"{cls.__name__}.semantics must be str"`, + `f"{cls.__name__}.semantics must be non-empty"`. + - `ClassVar` is already imported (`Rule` uses it at L191). +2. All 18 test-defined `Grammar` subclasses in 8 files get + `semantics = ""` (Phase-1 identity, D4): + `tests/unit/test_capability.py` L14 `StubGrammar`; `tests/unit/ + test_extensions.py` L42/51/60/69 (`_DotDateGrammar`, `_SecondGrammar`, + `_NamelessGrammar`, `_MixedCaseGrammar`); `tests/unit/test_discovery.py` + L61 `DotDateGrammar`; `tests/integration/test_feature_gating.py` L54 + `_NameRecognitionGrammar`; `tests/integration/test_grammar_extensions.py` + L35/54/73 (`DotDateGrammar`, `SecondDateGrammar`, `ClashingDateGrammar`); + `tests/integration/test_format_value_seam.py` L47/143 (`_TokenGrammar`, + `_DualTokenGrammar`); `tests/integration/test_pipeline.py` L127/136/357 + (`CrashGrammar`, `SimpleGrammar`, `_PhantomGrammar`); + `tests/integration/test_recognition_seam.py` L44/69/376 + (`_ProbeLongGrammar`, `_ProbeShortGrammar`, `_CommunityGrammar`). + Do NOT touch `_CountingLongGrammar` (L324 — inherits, D3) or the + lookalike test classes (`TestGrammarRule`, `TestGrammarDedup`, + `TestExtraGrammars` — not Grammar subclasses). +3. Ship the enforcement tests from Step 1. + +**Verify** +```bash +uv run pytest tests/unit/test_grammar_semantics_metadata.py -q +uv run pytest -q +uv run ruff check paxman/ tests/ +uv run pyright +``` + +**Commit** +``` +feat(core): enforce Grammar.semantics at class-definition time +``` + +--- + +### Task 4 — `refactor(engine): route on semantics, not grammar name` + +The three engine sites switch from name keys to semantics keys (D5). With +`semantics == name` (D1) the routing keys are set-equal, so this is +byte-identical — **no RED test is writable**; the full suite is the +regression net (a behavior change would surface as test failures). Do not +rename `GrammarRule.grammar_name`, `Candidate.recognition_rule`, or the +`_dedup_candidates` key — provenance stays name-based (ADR §4). + +**Step 1 GREEN — `paxman/engine/orchestrator.py`** +- `run_capability` (L54-89): after `all_grammars` is composed (L60-63), + build `semantics_by_name = {g.name: g.semantics for g in all_grammars}` + and pass it to `_validate_affinity`, `_collect_candidates`, and + `_activated_rules` (update signatures; `_activated_rules` currently takes + `(capability, contract)` at L385, `_collect_candidates` takes + `(capability, recognitions, rules)` at L295). +- `_validate_affinity` (L276-292): `known_grammars = {g.name ...}` → + `known_semantics = {g.semantics for g in all_grammars}`; iterate + `rule.target_semantics`; error message: "declares unknown semantics" + (keep the sorted-listing shape). +- `_collect_candidates` (L295-338): at L310-312, resolve the producing + grammar's semantics via the map — `semantics_by_name[grammar_name] not in + rule.target_semantics: continue`. Update the docstring (L302) to describe + semantic routing. +- `_activated_rules` (L385-400): replace `extra_grammars & + rule.target_grammars` with a semantics-keyed activation: + `extra_semantics = {semantics_by_name[n] for n in extra_grammars if n in + semantics_by_name}`; activate iff `extra_semantics & + rule.target_semantics`. Unknown extra names are silently skipped (existing + behavior preserved). +- Check no test asserts the OLD `_validate_affinity` error text verbatim + (grep `unknown grammar` in `tests/`); if one does, update it in this + commit to "unknown semantics". + +**Step 2 VERIFY (byte-identity proof)** +```bash +uv run pytest -q +uv run ruff check paxman/ tests/ +uv run pyright +``` +All green = routing keyed on semantics is behavior-identical (D1). + +**Commit** +``` +refactor(engine): route on semantics, not grammar name +``` + +--- + +### Task 5 — `refactor(capabilities): coalesce Date grammars to calendar-date semantics` + +First Phase 2 coalescing (ADR Migration #2, the ADR's own worked example). +Two RED drivers: the consistency-guard test's group lookup, and +`CommunityISO8601Rule`'s dangling semantics. + +**Step 1 RED** +- Create `tests/unit/test_grammar_semantics_consistency.py` (D8): a generic + guard that enumerates all shipped grammar classes, groups them by + `semantics`, and for each multi-member group runs shared probe rows + through each member's `recognize()` asserting identical notation fields + + canonicalization expectations. Seed it with the Date group's probe row: + `iso8601` input `"2026-01-15"` and `slash_iso` input `"2026/01/15"` both + produce `DateNotation(N1="2026", N2="01", N3="15")`, both canonicalize to + `"2026-01-15"` through the ISO rule. (Verify the exact `DateNotation` + field names from `paxman/capabilities/Date/notation.py` while writing.) +- In `tests/integration/test_grammar_extensions.py`, update + `CommunityISO8601Rule.target_semantics` (L135) from + `frozenset({"iso8601_recognition"})` to + `frozenset({"iso8601_calendar_date"})`. +- Run: `uv run pytest tests/unit/test_grammar_semantics_consistency.py + tests/integration/test_grammar_extensions.py -q` → RED: the guard test + cannot find a `"iso8601_calendar_date"` group (grammars still claim + identity ids) and `test_grammar_extensions.py` fails fast with + `ContractError` (dangling semantics — the coalesced id is not yet known). + +**Step 2 GREEN** +- Date grammars (4): `iso8601_recognition.py` and `slash_iso_recognition.py` + → `semantics = "iso8601_calendar_date"`; `us_recognition.py` → + `"us_calendar_date"`; `european_recognition.py` → `"european_calendar_date"`. +- Date rules (3): `iso_8601_ed2019.py` L36 → + `frozenset({"iso8601_calendar_date"})`; `en_50160_ed2010.py` L33 and + `us_federal_rules_ed2023.py` L33 → + `frozenset({"us_calendar_date", "european_calendar_date"})` (no widening, + D6). +- No other rule file changes (D7). +- **Plan deviation (approved 2026-08-11)**: `test_shipped_grammars_declare_semantics_identity` + (`tests/unit/test_grammar_semantics_metadata.py` L25-32) asserts `semantics == name` for + every shipped grammar; coalescing breaks it. Update it in this commit: keep the str / + non-empty assertions for all grammars, add a module-level `_COALESCED_SEMANTICS` frozenset + (seeded `{"iso8601_calendar_date", "us_calendar_date", "european_calendar_date"}`), and + assert `semantics == name or semantics in _COALESCED_SEMANTICS`. Tasks 6-7 extend the set + with the Email/Phone ids. Keeps every commit green — Task 8's `pytest tests/unit` verify + passes as written. + +**Verify** +```bash +uv run pytest tests/capabilities/date tests/integration/test_grammar_extensions.py \ + tests/unit/test_grammar_semantics_consistency.py -q +uv run pytest tests/integration -q +uv run pytest tests/unit/test_grammar_semantics_metadata.py -q +uv run ruff check paxman/capabilities/Date/ tests/ +uv run pyright +``` + +**Commit** +``` +refactor(capabilities): coalesce Date grammars to calendar-date semantics +``` + +--- + +### Task 6 — `refactor(capabilities): coalesce Email grammars to addr-spec semantics` + +Second coalescing (D6). The guard test is extended in the same commit — +its group lookup is the RED driver. + +**Step 1 RED — extend `tests/unit/test_grammar_semantics_consistency.py`** +- Add the `"rfc5322_addr_spec"` group's probe rows: `standard` input (e.g. + `"user@example.com"`) and `obfuscated` input (e.g. + `"user at example dot com"`) produce identical `EmailNotation` fields and + both canonicalize to `"user@example.com"` through the RFC 5322 rule. + (Verify exact `EmailNotation` fields from + `paxman/capabilities/Email/notation.py` while writing.) +- Run: `uv run pytest tests/unit/test_grammar_semantics_consistency.py -q` + → RED (no `"rfc5322_addr_spec"` group exists yet). + +**Step 2 GREEN** +- Email grammars (2): `standard_recognition.py` and + `obfuscated_recognition.py` → `semantics = "rfc5322_addr_spec"`. +- Email rule (1): `rfc_5322_ed2008.py` L36 → + `frozenset({"rfc5322_addr_spec"})`. +- Unchanged: `localhost_recognition.py` (identity), `rfc_6761_ed2012.py` L36 + `{"localhost_recognition"}`. + +**Verify** +```bash +uv run pytest tests/capabilities/email tests/unit/test_grammar_semantics_consistency.py -q +uv run pytest tests/integration -q +uv run ruff check paxman/capabilities/Email/ tests/ +uv run pyright +``` + +**Commit** +``` +refactor(capabilities): coalesce Email grammars to addr-spec semantics +``` + +--- + +### Task 7 — `refactor(capabilities): coalesce Phone grammars to E.164 semantics` + +Third coalescing (D6). Same pattern as Task 6. + +**Step 1 RED — extend `tests/unit/test_grammar_semantics_consistency.py`** +- Add the `"e164_international"` group's probe rows: `e164` input (e.g. + `"+15551234567"`) and `international_00` input (e.g. `"0015551234567"`) + produce identical `PhoneNotation` fields and both canonicalize to + `"+15551234567"` through the E.164 rule. (Verify exact `PhoneNotation` + fields from `paxman/capabilities/Phone/notation.py` while writing.) +- Run: `uv run pytest tests/unit/test_grammar_semantics_consistency.py -q` + → RED (no `"e164_international"` group yet). + +**Step 2 GREEN** +- Phone grammars (2): `e164_recognition.py` and + `international_00_recognition.py` → `semantics = "e164_international"`. +- Phone rules (2 classes, 1 file): `e164_ed2010.py` L69 and L111 → + `frozenset({"e164_international"})`. +- Unchanged: `tel_uri_recognition.py` + `rfc_3966_ed2004.py` L34 + (`{"tel_uri_recognition"}`), `national_recognition.py` + `nanp_ed2024.py` + L93/L148 (`{"national_recognition"}`). + +**Verify** +```bash +uv run pytest tests/capabilities/phone tests/unit/test_grammar_semantics_consistency.py -q +uv run pytest tests/integration -q +uv run ruff check paxman/capabilities/Phone/ tests/ +uv run pyright +``` + +**Commit** +``` +refactor(capabilities): coalesce Phone grammars to E.164 semantics +``` + +--- + +### Task 8 — `test: lock same-semantics field-mapping consistency` + +The guard's structural half: assert the no-coalesce and non-coalesced groups +stay singleton (D6/D7) and that every multi-member group is covered by probe +rows. This makes the guard a complete F1-style test-time guarantee (ADR +Migration #3). + +**Step 1 GREEN (test-only — the groups already exist after Tasks 5-7)** +- Extend `tests/unit/test_grammar_semantics_consistency.py`: + - A structural enumeration: every shipped grammar belongs to a semantics + group; every multi-member group MUST have probe rows defined in the + test's table (fails if a future coalescing adds a group without rows). + - Explicit singleton assertions for the no-coalesce set (D7): exactly one + grammar claims each of `"us_calendar_date"`, `"european_calendar_date"`, + `"name_recognition"`, `"alpha2_recognition"`, `"alpha3_recognition"`, + `"numeric_recognition"`, `"isbn13_recognition"`, `"isbn10_recognition"`. +- Run: `uv run pytest tests/unit/test_grammar_semantics_consistency.py -q` + → GREEN (all groups consistent after Tasks 5-7). If RED, a coalescing + drifted — investigate, do not weaken the test. + +**Verify** +```bash +uv run pytest tests/unit/test_grammar_semantics_consistency.py -q +uv run pytest tests/unit -q +``` + +**Commit** +``` +test: lock same-semantics field-mapping consistency +``` + +--- + +### Task 9 — `docs: sweep target_grammars and document semantic affinity` + +> **Follow-up items folded in here (M1-M2: Task 4 review; M3: Task 5 execution):** +> - **M1 (source docstring):** in `paxman/engine/orchestrator.py` `_collect_candidates`, add one sentence to the docstring stating the KeyError invariant for `semantics_by_name[grammar_name]`: recognitions are produced only by grammars in the composed `all_grammars` (`_recognize` filters against `supported_names`), the same list the map is built from. +> - **M2 (dead citation):** the `(ARCHITECTURE.md:201)` reference in that same docstring is stale (ARCHITECTURE.md has no routing/affinity content; L201 is "Quality Enforcement") — drop the dead line-number reference while updating the docstring. +> - **M3 (README fail-fast mechanism, verified against the engine 2026-08-11):** the "Rules of the seam" fail-fast bullet must state the real mechanism — `_activated_rules` resolves `extra_grammars` names via `semantics_by_name.get(n, n)`, so an unknown extra name keeps its own string and can activate a community rule targeting that (dangling) id, which `_validate_affinity` then rejects with `ContractError`. A rule that is NOT opted in is silently inert regardless of dangling targets. (Task 5's RED driver exercised the inert path, not the fail-fast path, hence the plan's original ContractError prediction did not match.) + +Docs sweep (ADR Migration #4, D9). **No RED step** — pure documentation. + +**Step 1 GREEN — rewrite the 6 in-scope repo-root files + 2 nested AGENTS.md** + +| File | References to update | +|------|----------------------| +| `README.md` | L458 (Community Extensions sample: `DotDateGrammar` gains `semantics = "dot_date_recognition"`, `DotDateRule` `target_grammars` → `target_semantics`), L485 + L488 ("Rules of the seam" bullets — opt-in and fail-fast now keyed on semantics) | +| `ARCHITECTURE.md` | L100 (composition guard), L174 (community opt-in), L176 (fail-fast `ContractError`) — reword to semantics vocabulary | +| `CONTEXT.md` | L140, L155 (six-metadata-attrs sentence → `target_semantics` + grammar `semantics` claim), L182, L207 | +| `HOW_TO_ADD_NEW_CAPABILITY.md` | L292 (Step 5 directive), L441, L449, L453 (rule template `target_semantics: ClassVar[frozenset[str]]`), L572 (orthogonality note), L1035, L1056, L1069 (checklist) | +| `HOW_TO_ADD_NEW_GRAMMAR.md` | L21, L23, L33, L174, L178, L192 (extended example), L200, L202, L287 (validation table) — **Step 4 is rewritten per ADR**: a grammar whose meaning is already shipped declares a shipped `semantics` id and stops (no rule edit); a genuinely new meaning requires a new rule | +| `capability_homogeneity_audit.md` | L63, L65 (proposed orchestrator line → semantics keyed), L233, L296, L307 (F1 addenda) | +| `paxman/core/AGENTS.md` | L25 (Rule six-attr list → `target_semantics`) + add the Grammar `semantics` convention | +| `paxman/capabilities/AGENTS.md` | L43 (rule-file conventions → `target_semantics`) + grammar `semantics` requirement | + +Note: `HOW_TO_ADD_NEW_GRAMMAR.md` and `HOW_TO_ADD_NEW_CAPABILITY.md` live at +the **repo root**, not under `docs/`. + +Do NOT touch (historical records, ADR-0002 precedent): `docs/superpowers/ +plans/*`, `docs/report/*`, `docs/research/*`, `docs/adr/*`. + +**Verify** (zero hits outside the excluded paths — the historical records and +generated dirs are excluded; `paxman/` and `tests/` are searched, because the +sweep covers the nested AGENTS.md there): +```bash +grep -rnE 'target_grammars' . \ + --exclude-dir=.git --exclude-dir=plans --exclude-dir=report \ + --exclude-dir=research --exclude-dir=adr \ + --exclude-dir=htmlcov --exclude-dir=.hypothesis \ + --exclude-dir=.pytest_cache --exclude-dir=.venv +``` +No `|| echo "CLEAN"` fallback: zero matches is the expected result (grep +prints nothing, exits 1); any grep error (exit ≥ 2) stays visible instead of +being reported as "CLEAN". + +**Commit** +``` +docs: sweep target_grammars and document semantic affinity +``` + +--- + +### Task 10 — Final gate (no commit) + +**Verify — full pre-PR gate** (authoritative per `.github/workflows/ci.yml`; +ruff lint and format are CI-scoped to `paxman/ tests/`): +```bash +uv run ruff check paxman/ tests/ && uv run ruff format --check paxman/ tests/ \ + && uv run pyright && uv run import-linter lint && uv run pytest +``` +Coverage gate (one include pattern per package — the brace shorthand +`paxman/{core,capabilities,engine,api}/*` is not expanded by the installed +coverage version and reports "No data to report"): +```bash +uv run coverage report --include="paxman/core/*" --fail-under=95 +uv run coverage report --include="paxman/capabilities/*" --fail-under=95 +uv run coverage report --include="paxman/engine/*" --fail-under=95 +uv run coverage report --include="paxman/api/*" --fail-under=95 +``` +Zero-grep proof (Task 9 Verify) shows no matches outside the excluded +paths; `target_grammars` appears nowhere in `paxman/` or `tests/`; `semantics` +is present on all 26 shipped grammars (Task 1 test still green). + +If any gate fails, fix it in a follow-up commit — never by weakening a test, +never by restoring `target_grammars`, never by editing the excluded +historical docs. + +--- + +## §3 Traps + +1. **Ordering is load-bearing.** Task 2 must include `domain.py` AND all 21 + rule files AND the 12 test files in one commit — any split is an + import-time `TypeError` (D2). Task 3 must include `domain.py` AND the 18 + test doubles in one commit — enforcement import-fails every grammar + subclass lacking `semantics` (D4). Task 5 must include the + `CommunityISO8601Rule` fixture update in the same commit as the Date + coalescing, or `tests/integration/test_grammar_extensions.py` fails fast + with `ContractError` (dangling `"iso8601_calendar_date"`). +2. **Never widen a rule's meaning set.** `target_semantics` coalescing is + set-collapse only: `{"iso8601_recognition","slash_iso_recognition"}` → + `{"iso8601_calendar_date"}` is fine; `{"us_recognition", + "european_recognition"}` → `{"us_calendar_date","european_calendar_date"}` + keeps two ids. ISBN is deliberately NOT coalesced (D7) — collapsing + isbn13/isbn10 would widen `iso_2108_ed2017.py` to ISBN-10 input. If a + coalescing looks like it needs a set to grow, it is wrong. +3. **Error-message coupling.** `tests/unit/test_rule_metadata.py` matches + `TypeError` text that embeds the attribute name (`match=missing`, L111 + `"non-empty"`, L152 `"must be frozenset[str]"`) and `domain.py` L219 bakes + `target_grammars` into the message — the Task 2 sweep must rename source + and test in the same commit. Same coupling applies to the new + `Grammar.__init_subclass__` messages (Task 3) — write the tests to the + exact messages you ship. +4. **ADR's "55 files" is stale.** Verified ground truth is 44 files + (D10). Use this plan's tables; the ADR's counts are for reference only. +5. **Generated artifacts trip the zero-grep proof.** `htmlcov/` (11 files), + `.hypothesis/`, `.pytest_cache/` contain stale `target_grammars` matches + and are gitignored — the Task 9/10 proof command excludes them (D9). +6. **`HOW_TO_ADD_*` guides are at the repo root**, not under `docs/` — path + mistakes in the sweep are silent no-ops. +7. **The three engine routing sites have no direct unit coverage** — + `_validate_affinity`, `_collect_candidates`, `_activated_rules` are + covered through integration fixtures (recognition seam, feature gating, + grammar extensions, pipeline). Do not "add coverage" by weakening those + fixtures; Task 4's byte-identity proof IS the integration suite. +8. **Provenance stays name-based.** Do not "helpfully" rename + `GrammarRule.grammar_name`, `Candidate.recognition_rule`, or the + `_dedup_candidates` key in Task 4 — ADR §4 pins them; changing them is + an output change outside this ADR. +9. **Property and e2e suites have zero `target_grammars` hits** — no edits + needed, but they instantiate shipped grammars, so Task 3's enforcement + applies to them at import. They are exercised at the Task 10 gate. +10. **`_CountingLongGrammar` needs no edit** (D3 — inheritance satisfies + the `vars(cls).get` fallback), and the lookalike classes + (`TestGrammarRule`, `TestGrammarDedup`, `TestExtraGrammars`) are test + classes, not Grammar subclasses — do not touch them. + +--- + +## §4 Definition of Done + +- [x] All 26 shipped grammars declare `semantics` (identity in Phase 1; + coalesced ids for Date/Email/Phone after Phase 2), enforced by + `Grammar.__init_subclass__` at class-definition time with tests. +- [x] Zero `target_grammars` anywhere in `paxman/` or `tests/`; the Task 9 + zero-grep proof shows no matches outside the excluded historical paths. +- [x] Engine routes on semantics at all three sites; provenance and + candidate dedup remain name-based (`GrammarRule.grammar_name`, + `Candidate.recognition_rule` unchanged). +- [x] Phase 2 coalescing landed for Date/Email/Phone exactly as D6/D7 + scope; no rule's `target_semantics` set grew. +- [x] `tests/unit/test_grammar_semantics_consistency.py` covers every + multi-member semantics group with probe rows and locks the singleton + no-coalesce set. +- [x] Docs swept (Task 9 files); README's community example shows + `semantics` on the grammar and `target_semantics` on the rule. +- [x] Full pre-PR gate green: `ruff check . && ruff format --check . && + pyright && import-linter lint && pytest` and 95% coverage per package. + (Gate as written is green under CI scope; `ruff format --check .` + additionally flags 8 pre-existing historical docs — see progress + table note.) + diff --git a/paxman/capabilities/AGENTS.md b/paxman/capabilities/AGENTS.md index c2e7cdef..1cd221b4 100644 --- a/paxman/capabilities/AGENTS.md +++ b/paxman/capabilities/AGENTS.md @@ -39,8 +39,8 @@ Every capability must conform to the same structural surface. `CapabilityContrac - **Notation** — `@dataclass(frozen=True, slots=True)`; one `str` field per component; the sole type parameter of the capability's `Grammar[NotationT]` / `Rule[NotationT]`. - **Contract** — `@dataclass(frozen=True)` extending `CapabilityContract`, NO `slots=True` (incompatible with the base `super()` pattern). Sets `DEFAULT_OUTPUT_FORMAT` / `OFFERED_OUTPUT_FORMATS` class vars; `capability_name` via `field(init=False)`; inherits `output_format` (always optional; base `__post_init__` resolves it and validates offered alternatives); `active_grammars` is optional — only feature-gated capabilities (Email, IP, ISBN) override it, and the base `None` default runs every shipped grammar. -- **Grammar** — one file = one recognizer; `name` = `{format}_recognition` (snake_case, unique per capability); emits span-bearing `RecognitionMatch` (half-open `[start, end)`, `raw_text`); syntax-only — extraction and sanitization, never validation, dedup, ordering, or token→canonical mapping. Two sanctioned strategies: Regex (shape) and Lexicon (key-only tables, kept in `grammar/data/` apart from recognition logic); see HOW_TO for the extended set. -- **Rule** — one file = one publication (module-level `PUBLICATION` provenance constant); class = one spec section; declares `name` (`Section {X.Y.Z}-{description}`), `strategy`, `provenance`, `citation`, `target_grammars` (non-empty), `requires_features` — all six enforced at import time. Rule classes sharing one publication live in the same file; authority-backed lookup tables live in `rules/data/`, separated from rule logic. +- **Grammar** — one file = one recognizer; `name` = `{format}_recognition` (snake_case, unique per capability); `semantics` = non-empty string declaring the meaning the grammar's notations carry (identity id unless the format shares another grammar's meaning, in which case declare that coalesced id — shared meaning ⇒ shared id); emits span-bearing `RecognitionMatch` (half-open `[start, end)`, `raw_text`); syntax-only — extraction and sanitization, never validation, dedup, ordering, or token→canonical mapping. Two sanctioned strategies: Regex (shape) and Lexicon (key-only tables, kept in `grammar/data/` apart from recognition logic); see HOW_TO for the extended set. +- **Rule** — one file = one publication (module-level `PUBLICATION` provenance constant); class = one spec section; declares `name` (`Section {X.Y.Z}-{description}`), `strategy`, `provenance`, `citation`, `target_semantics` (non-empty), `requires_features` — all six enforced at import time. Rule classes sharing one publication live in the same file; authority-backed lookup tables live in `rules/data/`, separated from rule logic. - **Feature gating — two loci, two statuses** — input-shape features toggle grammars via `active_grammars` (disabled grammar → `MISSING`), implemented only by the gated capabilities (Email, IP, ISBN) — other contracts inherit the `None` default, which runs every shipped grammar; authority features gate rules via `requires_features` (dropped rule → `INVALID`). Never gate inside `matches()`; never cast to read `include_*` flags. `typing.cast` is only for validity-affecting parameters. - **Presentation-only invariant** — rules never reference `output_format` (CI-scanned); `normalize()` always returns the default canonical form; `format_value()` is the only presentation seam, overridden only when `OFFERED_OUTPUT_FORMATS` is non-empty; formatting adds no provenance; offered formats must preserve the capability's ambiguity contract. - **`create_contract()`** — static, keyword-only; fixed common block first (`excluded_rules`, `pinned_rules`, `year`, `output_format`), capability-specific params after. diff --git a/paxman/capabilities/Country/grammar/alpha2_recognition.py b/paxman/capabilities/Country/grammar/alpha2_recognition.py index c1316472..5a45b1fb 100644 --- a/paxman/capabilities/Country/grammar/alpha2_recognition.py +++ b/paxman/capabilities/Country/grammar/alpha2_recognition.py @@ -18,6 +18,7 @@ class Alpha2Grammar(Grammar[CountryNotation]): """ name = "alpha2_recognition" + semantics = "alpha2_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CountryNotation]]: """Extract alpha-2 patterns from text. diff --git a/paxman/capabilities/Country/grammar/alpha3_recognition.py b/paxman/capabilities/Country/grammar/alpha3_recognition.py index 2805520d..06aa3b63 100644 --- a/paxman/capabilities/Country/grammar/alpha3_recognition.py +++ b/paxman/capabilities/Country/grammar/alpha3_recognition.py @@ -18,6 +18,7 @@ class Alpha3Grammar(Grammar[CountryNotation]): """ name = "alpha3_recognition" + semantics = "alpha3_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CountryNotation]]: """Extract alpha-3 patterns from text. diff --git a/paxman/capabilities/Country/grammar/name_recognition.py b/paxman/capabilities/Country/grammar/name_recognition.py index ee979006..62640cb0 100644 --- a/paxman/capabilities/Country/grammar/name_recognition.py +++ b/paxman/capabilities/Country/grammar/name_recognition.py @@ -49,6 +49,7 @@ class NameGrammar(Grammar[CountryNotation]): """ name = "name_recognition" + semantics = "name_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CountryNotation]]: """Extract a country name representation from text. diff --git a/paxman/capabilities/Country/grammar/numeric_recognition.py b/paxman/capabilities/Country/grammar/numeric_recognition.py index 1d299bcf..bdd16ff1 100644 --- a/paxman/capabilities/Country/grammar/numeric_recognition.py +++ b/paxman/capabilities/Country/grammar/numeric_recognition.py @@ -18,6 +18,7 @@ class NumericGrammar(Grammar[CountryNotation]): """ name = "numeric_recognition" + semantics = "numeric_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CountryNotation]]: """Extract numeric patterns from text. diff --git a/paxman/capabilities/Country/rules/cldr_localized_ed2025.py b/paxman/capabilities/Country/rules/cldr_localized_ed2025.py index 2bfcf303..5b1b977e 100644 --- a/paxman/capabilities/Country/rules/cldr_localized_ed2025.py +++ b/paxman/capabilities/Country/rules/cldr_localized_ed2025.py @@ -45,7 +45,7 @@ class SectionLocalizedNames(Rule[CountryNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "CLDR v45 localized country names" - target_grammars = frozenset({"name_recognition"}) + target_semantics = frozenset({"name_recognition"}) requires_features = frozenset({"include_localized"}) def matches(self, notation: CountryNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Country/rules/iso_3166_ed2024.py b/paxman/capabilities/Country/rules/iso_3166_ed2024.py index be382a9a..4edd79c5 100644 --- a/paxman/capabilities/Country/rules/iso_3166_ed2024.py +++ b/paxman/capabilities/Country/rules/iso_3166_ed2024.py @@ -58,7 +58,7 @@ class SectionAlpha2Codes(Rule[CountryNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "ISO 3166-1 alpha-2 codes" - target_grammars = frozenset({"alpha2_recognition"}) + target_semantics = frozenset({"alpha2_recognition"}) requires_features = frozenset() def matches(self, notation: CountryNotation, contract: Contract) -> bool: @@ -98,7 +98,7 @@ class SectionAlpha3Codes(Rule[CountryNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "ISO 3166-1 alpha-3 codes" - target_grammars = frozenset({"alpha3_recognition"}) + target_semantics = frozenset({"alpha3_recognition"}) requires_features = frozenset() def matches(self, notation: CountryNotation, contract: Contract) -> bool: @@ -138,7 +138,7 @@ class SectionNumericCodes(Rule[CountryNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "ISO 3166-1 numeric (M49) codes" - target_grammars = frozenset({"numeric_recognition"}) + target_semantics = frozenset({"numeric_recognition"}) requires_features = frozenset() def _normalize_key(self, value: str) -> str: @@ -186,7 +186,7 @@ class SectionNames(Rule[CountryNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "ISO 3166-1 official English short names" - target_grammars = frozenset({"name_recognition"}) + target_semantics = frozenset({"name_recognition"}) requires_features = frozenset() def matches(self, notation: CountryNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Country/rules/iso_3166_historical_ed2020.py b/paxman/capabilities/Country/rules/iso_3166_historical_ed2020.py index e6c929da..2a7e4b72 100644 --- a/paxman/capabilities/Country/rules/iso_3166_historical_ed2020.py +++ b/paxman/capabilities/Country/rules/iso_3166_historical_ed2020.py @@ -66,7 +66,7 @@ class SectionHistoricalNames(Rule[CountryNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "ISO 3166-3:2020 (formerly used names)" - target_grammars = frozenset( + target_semantics = frozenset( {"name_recognition", "alpha2_recognition", "numeric_recognition"} ) requires_features = frozenset({"include_historical"}) diff --git a/paxman/capabilities/Currency/grammar/code_recognition.py b/paxman/capabilities/Currency/grammar/code_recognition.py index d6fd2fba..9c581fa3 100644 --- a/paxman/capabilities/Currency/grammar/code_recognition.py +++ b/paxman/capabilities/Currency/grammar/code_recognition.py @@ -33,6 +33,7 @@ class CodeRecognition(Grammar[CurrencyNotation]): """ name = "code_recognition" + semantics = "code_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CurrencyNotation]]: """Extract standalone 3-letter code tokens from text. diff --git a/paxman/capabilities/Currency/grammar/symbol_recognition.py b/paxman/capabilities/Currency/grammar/symbol_recognition.py index 2d9f4a63..7ef6aa7d 100644 --- a/paxman/capabilities/Currency/grammar/symbol_recognition.py +++ b/paxman/capabilities/Currency/grammar/symbol_recognition.py @@ -44,6 +44,7 @@ class SymbolRecognition(Grammar[CurrencyNotation]): """ name = "symbol_recognition" + semantics = "symbol_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CurrencyNotation]]: """Extract standalone symbol tokens from text. diff --git a/paxman/capabilities/Currency/grammar/word_recognition.py b/paxman/capabilities/Currency/grammar/word_recognition.py index c1330c2a..54d40978 100644 --- a/paxman/capabilities/Currency/grammar/word_recognition.py +++ b/paxman/capabilities/Currency/grammar/word_recognition.py @@ -38,6 +38,7 @@ class WordRecognition(Grammar[CurrencyNotation]): """ name = "word_recognition" + semantics = "word_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CurrencyNotation]]: """Extract standalone display-name word tokens from text. diff --git a/paxman/capabilities/Currency/rules/cldr_currencies_ed2025.py b/paxman/capabilities/Currency/rules/cldr_currencies_ed2025.py index 2f7c70ab..9d74a770 100644 --- a/paxman/capabilities/Currency/rules/cldr_currencies_ed2025.py +++ b/paxman/capabilities/Currency/rules/cldr_currencies_ed2025.py @@ -122,7 +122,7 @@ class SectionSymbols(Rule[CurrencyNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "CLDR v47 currency symbols" - target_grammars = frozenset({"symbol_recognition"}) + target_semantics = frozenset({"symbol_recognition"}) requires_features = frozenset() def matches(self, notation: CurrencyNotation, contract: Contract) -> bool: @@ -166,7 +166,7 @@ class SectionNames(Rule[CurrencyNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "CLDR v47 currency display names" - target_grammars = frozenset({"word_recognition"}) + target_semantics = frozenset({"word_recognition"}) requires_features = frozenset() def matches(self, notation: CurrencyNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Currency/rules/iso_4217_ed2015.py b/paxman/capabilities/Currency/rules/iso_4217_ed2015.py index dfbda55d..17edfac4 100644 --- a/paxman/capabilities/Currency/rules/iso_4217_ed2015.py +++ b/paxman/capabilities/Currency/rules/iso_4217_ed2015.py @@ -42,7 +42,7 @@ class SectionCode(Rule[CurrencyNotation]): "ISO 4217:2015 alpha-3 currency codes, as amended by the ISO 4217 " "Maintenance Agency amendment series (SIX List One, 2026-01-01)" ) - target_grammars = frozenset({"code_recognition"}) + target_semantics = frozenset({"code_recognition"}) requires_features = frozenset() def matches(self, notation: CurrencyNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Date/grammar/european_recognition.py b/paxman/capabilities/Date/grammar/european_recognition.py index 6e03c372..b38f3186 100644 --- a/paxman/capabilities/Date/grammar/european_recognition.py +++ b/paxman/capabilities/Date/grammar/european_recognition.py @@ -10,17 +10,22 @@ from paxman.capabilities.Date.notation import DateNotation from paxman.core.domain import Grammar, RecognitionMatch -_EUROPEAN_DATE_PATTERN_4DIGIT = re.compile(r"(\d{1,2})/(\d{1,2})/(\d{4})") +_EUROPEAN_DATE_PATTERN_4DIGIT = re.compile(r"(? list[RecognitionMatch[DateNotation]]: """Extract European date patterns from text. diff --git a/paxman/capabilities/Date/grammar/iso8601_recognition.py b/paxman/capabilities/Date/grammar/iso8601_recognition.py index 55897c23..d683c05a 100644 --- a/paxman/capabilities/Date/grammar/iso8601_recognition.py +++ b/paxman/capabilities/Date/grammar/iso8601_recognition.py @@ -10,16 +10,21 @@ from paxman.capabilities.Date.notation import DateNotation from paxman.core.domain import Grammar, RecognitionMatch -_ISO8601_PATTERN = re.compile(r"(\d{4})-(\d{2})-(\d{2})") +_ISO8601_PATTERN = re.compile(r"(? list[RecognitionMatch[DateNotation]]: """Extract ISO 8601 date patterns from text.""" diff --git a/paxman/capabilities/Date/grammar/slash_iso_recognition.py b/paxman/capabilities/Date/grammar/slash_iso_recognition.py index 3c9e407b..5746ead1 100644 --- a/paxman/capabilities/Date/grammar/slash_iso_recognition.py +++ b/paxman/capabilities/Date/grammar/slash_iso_recognition.py @@ -10,7 +10,7 @@ from paxman.capabilities.Date.notation import DateNotation from paxman.core.domain import Grammar, RecognitionMatch -_SLASH_ISO_PATTERN = re.compile(r"(\d{4})/(\d{1,2})/(\d{1,2})") +_SLASH_ISO_PATTERN = re.compile(r"(? list[RecognitionMatch[DateNotation]]: """Extract YYYY/MM/DD date patterns from text.""" diff --git a/paxman/capabilities/Date/grammar/us_recognition.py b/paxman/capabilities/Date/grammar/us_recognition.py index 88d42409..22965f07 100644 --- a/paxman/capabilities/Date/grammar/us_recognition.py +++ b/paxman/capabilities/Date/grammar/us_recognition.py @@ -10,17 +10,22 @@ from paxman.capabilities.Date.notation import DateNotation from paxman.core.domain import Grammar, RecognitionMatch -_US_DATE_PATTERN_4DIGIT = re.compile(r"(\d{1,2})/(\d{1,2})/(\d{4})") +_US_DATE_PATTERN_4DIGIT = re.compile(r"(? list[RecognitionMatch[DateNotation]]: """Extract US date patterns from text. diff --git a/paxman/capabilities/Date/rules/en_50160_ed2010.py b/paxman/capabilities/Date/rules/en_50160_ed2010.py index c2e1d453..5d910ec2 100644 --- a/paxman/capabilities/Date/rules/en_50160_ed2010.py +++ b/paxman/capabilities/Date/rules/en_50160_ed2010.py @@ -30,7 +30,7 @@ class Section4DateFormat(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 4 (date format)" - target_grammars = frozenset({"us_recognition", "european_recognition"}) + target_semantics = frozenset({"us_calendar_date", "european_calendar_date"}) requires_features = frozenset() def _interpret_two_digit_year(self, year_str: str, contract: Contract) -> int: diff --git a/paxman/capabilities/Date/rules/iso_8601_ed2019.py b/paxman/capabilities/Date/rules/iso_8601_ed2019.py index 7f49a6b1..16f6746c 100644 --- a/paxman/capabilities/Date/rules/iso_8601_ed2019.py +++ b/paxman/capabilities/Date/rules/iso_8601_ed2019.py @@ -33,7 +33,7 @@ class Section431CalendarDate(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 4.3.1 (calendar date)" - target_grammars = frozenset({"iso8601_recognition", "slash_iso_recognition"}) + target_semantics = frozenset({"iso8601_calendar_date"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Date/rules/us_federal_rules_ed2023.py b/paxman/capabilities/Date/rules/us_federal_rules_ed2023.py index e0f221da..4abde9fb 100644 --- a/paxman/capabilities/Date/rules/us_federal_rules_ed2023.py +++ b/paxman/capabilities/Date/rules/us_federal_rules_ed2023.py @@ -30,7 +30,7 @@ class Section1DateFormat(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 1 (date format)" - target_grammars = frozenset({"us_recognition", "european_recognition"}) + target_semantics = frozenset({"us_calendar_date", "european_calendar_date"}) requires_features = frozenset() def _interpret_two_digit_year(self, year_str: str, contract: Contract) -> int: diff --git a/paxman/capabilities/Email/grammar/localhost_recognition.py b/paxman/capabilities/Email/grammar/localhost_recognition.py index 41e9573c..a9dfca0b 100644 --- a/paxman/capabilities/Email/grammar/localhost_recognition.py +++ b/paxman/capabilities/Email/grammar/localhost_recognition.py @@ -17,6 +17,7 @@ class LocalhostEmailGrammar(Grammar[EmailNotation]): """Localhost email recognition: user@localhost.""" name = "localhost_recognition" + semantics = "localhost_recognition" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: matches: list[RecognitionMatch[EmailNotation]] = [] diff --git a/paxman/capabilities/Email/grammar/obfuscated_recognition.py b/paxman/capabilities/Email/grammar/obfuscated_recognition.py index 641df8e5..e219b0db 100644 --- a/paxman/capabilities/Email/grammar/obfuscated_recognition.py +++ b/paxman/capabilities/Email/grammar/obfuscated_recognition.py @@ -21,6 +21,7 @@ class ObfuscatedEmailGrammar(Grammar[EmailNotation]): """Obfuscated email: 'user at domain dot tld' or 'user at domain.tld'.""" name = "obfuscated_recognition" + semantics = "rfc5322_addr_spec" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: """Extract obfuscated email patterns from text. diff --git a/paxman/capabilities/Email/grammar/standard_recognition.py b/paxman/capabilities/Email/grammar/standard_recognition.py index 92124bcd..9505a8cb 100644 --- a/paxman/capabilities/Email/grammar/standard_recognition.py +++ b/paxman/capabilities/Email/grammar/standard_recognition.py @@ -16,6 +16,7 @@ class StandardEmailGrammar(Grammar[EmailNotation]): """Standard email recognition: user@domain.tld.""" name = "standard_recognition" + semantics = "rfc5322_addr_spec" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: matches: list[RecognitionMatch[EmailNotation]] = [] diff --git a/paxman/capabilities/Email/rules/rfc_5322_ed2008.py b/paxman/capabilities/Email/rules/rfc_5322_ed2008.py index f0ac331f..9742e214 100644 --- a/paxman/capabilities/Email/rules/rfc_5322_ed2008.py +++ b/paxman/capabilities/Email/rules/rfc_5322_ed2008.py @@ -33,7 +33,7 @@ class Section341AddrSpec(Rule[EmailNotation]): strategy = RuleStrategy.REGEX provenance = PUBLICATION citation = "Section 3.4.1 (addr-spec)" - target_grammars = frozenset({"standard_recognition", "obfuscated_recognition"}) + target_semantics = frozenset({"rfc5322_addr_spec"}) requires_features = frozenset() def matches(self, notation: EmailNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Email/rules/rfc_6761_ed2012.py b/paxman/capabilities/Email/rules/rfc_6761_ed2012.py index 50eaa594..dc55ed8f 100644 --- a/paxman/capabilities/Email/rules/rfc_6761_ed2012.py +++ b/paxman/capabilities/Email/rules/rfc_6761_ed2012.py @@ -33,7 +33,7 @@ class Section63localhost(Rule[EmailNotation]): strategy = RuleStrategy.REGEX provenance = PUBLICATION citation = "Section 6.3 (localhost)" - target_grammars = frozenset({"localhost_recognition"}) + target_semantics = frozenset({"localhost_recognition"}) requires_features = frozenset() def matches(self, notation: EmailNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/IP/grammar/ipv4_recognition.py b/paxman/capabilities/IP/grammar/ipv4_recognition.py index 60d3eb68..ccec36dd 100644 --- a/paxman/capabilities/IP/grammar/ipv4_recognition.py +++ b/paxman/capabilities/IP/grammar/ipv4_recognition.py @@ -14,6 +14,7 @@ class IPv4Grammar(Grammar[IPNotation]): """IPv4 recognition: dotted-decimal format (e.g., 192.168.1.1).""" name = "ipv4_recognition" + semantics = "ipv4_recognition" def recognize(self, text: str) -> list[RecognitionMatch[IPNotation]]: """Extract IPv4 dotted-decimal patterns from text.""" diff --git a/paxman/capabilities/IP/grammar/ipv6_recognition.py b/paxman/capabilities/IP/grammar/ipv6_recognition.py index 107b4123..9774db17 100644 --- a/paxman/capabilities/IP/grammar/ipv6_recognition.py +++ b/paxman/capabilities/IP/grammar/ipv6_recognition.py @@ -49,6 +49,7 @@ class IPv6Grammar(Grammar[IPNotation]): """ name = "ipv6_recognition" + semantics = "ipv6_recognition" def recognize(self, text: str) -> list[RecognitionMatch[IPNotation]]: """Extract IPv6 address patterns from text. diff --git a/paxman/capabilities/IP/rules/rfc_5952_ed2010.py b/paxman/capabilities/IP/rules/rfc_5952_ed2010.py index ed90da84..ec08b72d 100644 --- a/paxman/capabilities/IP/rules/rfc_5952_ed2010.py +++ b/paxman/capabilities/IP/rules/rfc_5952_ed2010.py @@ -31,7 +31,7 @@ class Section4IPv6TextRepresentation(Rule[IPNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 4 (IPv6 text representation)" - target_grammars = frozenset({"ipv6_recognition"}) + target_semantics = frozenset({"ipv6_recognition"}) requires_features = frozenset() def matches(self, notation: IPNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/IP/rules/rfc_791_ed1981.py b/paxman/capabilities/IP/rules/rfc_791_ed1981.py index 899730c9..7bf6412b 100644 --- a/paxman/capabilities/IP/rules/rfc_791_ed1981.py +++ b/paxman/capabilities/IP/rules/rfc_791_ed1981.py @@ -30,7 +30,7 @@ class Section3Dot2IPv4Address(Rule[IPNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 3.2 (internet addressing)" - target_grammars = frozenset({"ipv4_recognition"}) + target_semantics = frozenset({"ipv4_recognition"}) requires_features = frozenset() @staticmethod diff --git a/paxman/capabilities/ISBN/grammar/isbn10_recognition.py b/paxman/capabilities/ISBN/grammar/isbn10_recognition.py index 98f6d3f0..4c70a837 100644 --- a/paxman/capabilities/ISBN/grammar/isbn10_recognition.py +++ b/paxman/capabilities/ISBN/grammar/isbn10_recognition.py @@ -18,6 +18,7 @@ class ISBN10RecognitionGrammar(Grammar[ISBNNotation]): """ISBN-10 recognition: 10-digit ISBN with optional label and separators.""" name = "isbn10_recognition" + semantics = "isbn10_recognition" def recognize(self, text: str) -> list[RecognitionMatch[ISBNNotation]]: matches: list[RecognitionMatch[ISBNNotation]] = [] diff --git a/paxman/capabilities/ISBN/grammar/isbn13_recognition.py b/paxman/capabilities/ISBN/grammar/isbn13_recognition.py index fe99a566..75b2ff6a 100644 --- a/paxman/capabilities/ISBN/grammar/isbn13_recognition.py +++ b/paxman/capabilities/ISBN/grammar/isbn13_recognition.py @@ -17,6 +17,7 @@ class ISBN13RecognitionGrammar(Grammar[ISBNNotation]): """ISBN-13 recognition: 13-digit ISBN with optional label and separators.""" name = "isbn13_recognition" + semantics = "isbn13_recognition" def recognize(self, text: str) -> list[RecognitionMatch[ISBNNotation]]: matches: list[RecognitionMatch[ISBNNotation]] = [] diff --git a/paxman/capabilities/ISBN/rules/isbn_range_message_ed2026.py b/paxman/capabilities/ISBN/rules/isbn_range_message_ed2026.py index 6fb60e58..00e0c919 100644 --- a/paxman/capabilities/ISBN/rules/isbn_range_message_ed2026.py +++ b/paxman/capabilities/ISBN/rules/isbn_range_message_ed2026.py @@ -34,7 +34,7 @@ class Section4RegistrantRange(Rule[ISBNNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "Section 4 (registrant range)" - target_grammars = frozenset({"isbn13_recognition", "isbn10_recognition"}) + target_semantics = frozenset({"isbn13_recognition", "isbn10_recognition"}) requires_features = frozenset({"include_range_validation"}) def matches(self, notation: ISBNNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/ISBN/rules/isbn_users_manual_ed2012.py b/paxman/capabilities/ISBN/rules/isbn_users_manual_ed2012.py index efa164bb..68326ae6 100644 --- a/paxman/capabilities/ISBN/rules/isbn_users_manual_ed2012.py +++ b/paxman/capabilities/ISBN/rules/isbn_users_manual_ed2012.py @@ -31,7 +31,7 @@ class Section6Isbn10CheckDigit(Rule[ISBNNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 6 (ISBN-10 check digit)" - target_grammars = frozenset({"isbn10_recognition"}) + target_semantics = frozenset({"isbn10_recognition"}) requires_features = frozenset() def matches(self, notation: ISBNNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/ISBN/rules/iso_2108_ed2017.py b/paxman/capabilities/ISBN/rules/iso_2108_ed2017.py index ad04e56d..457b5890 100644 --- a/paxman/capabilities/ISBN/rules/iso_2108_ed2017.py +++ b/paxman/capabilities/ISBN/rules/iso_2108_ed2017.py @@ -29,7 +29,7 @@ class Section53Isbn13CheckDigit(Rule[ISBNNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 5.3 (ISBN-13 check digit)" - target_grammars = frozenset({"isbn13_recognition"}) + target_semantics = frozenset({"isbn13_recognition"}) requires_features = frozenset() def matches(self, notation: ISBNNotation, contract: Contract) -> bool: @@ -52,7 +52,7 @@ class Section42Gs1Prefix(Rule[ISBNNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "Section 4.2 (GS1 prefix)" - target_grammars = frozenset({"isbn13_recognition"}) + target_semantics = frozenset({"isbn13_recognition"}) requires_features = frozenset() def matches(self, notation: ISBNNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Money/grammar/code_recognition.py b/paxman/capabilities/Money/grammar/code_recognition.py index a586b708..dce92260 100644 --- a/paxman/capabilities/Money/grammar/code_recognition.py +++ b/paxman/capabilities/Money/grammar/code_recognition.py @@ -38,6 +38,7 @@ class CodeRecognition(Grammar[MoneyNotation]): """ name = "code_recognition" + semantics = "code_recognition" def recognize(self, text: str) -> list[RecognitionMatch[MoneyNotation]]: """Extract code+amount tokens from text. diff --git a/paxman/capabilities/Money/grammar/symbol_recognition.py b/paxman/capabilities/Money/grammar/symbol_recognition.py index b908627b..b6c6693b 100644 --- a/paxman/capabilities/Money/grammar/symbol_recognition.py +++ b/paxman/capabilities/Money/grammar/symbol_recognition.py @@ -53,6 +53,7 @@ class SymbolRecognition(Grammar[MoneyNotation]): """ name = "symbol_recognition" + semantics = "symbol_recognition" def recognize(self, text: str) -> list[RecognitionMatch[MoneyNotation]]: """Extract symbol+amount tokens from text. diff --git a/paxman/capabilities/Money/grammar/word_recognition.py b/paxman/capabilities/Money/grammar/word_recognition.py index 632688c6..89fa5d81 100644 --- a/paxman/capabilities/Money/grammar/word_recognition.py +++ b/paxman/capabilities/Money/grammar/word_recognition.py @@ -45,6 +45,7 @@ class WordRecognition(Grammar[MoneyNotation]): """ name = "word_recognition" + semantics = "word_recognition" def recognize(self, text: str) -> list[RecognitionMatch[MoneyNotation]]: """Extract word+amount tokens from text. diff --git a/paxman/capabilities/Money/rules/cldr_currencies_ed2025.py b/paxman/capabilities/Money/rules/cldr_currencies_ed2025.py index 469756f9..74fdde34 100644 --- a/paxman/capabilities/Money/rules/cldr_currencies_ed2025.py +++ b/paxman/capabilities/Money/rules/cldr_currencies_ed2025.py @@ -133,7 +133,7 @@ class SectionSymbols(Rule[MoneyNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "CLDR v47 currency symbols" - target_grammars = frozenset({"symbol_recognition"}) + target_semantics = frozenset({"symbol_recognition"}) requires_features = frozenset() def matches(self, notation: MoneyNotation, contract: Contract) -> bool: @@ -196,7 +196,7 @@ class SectionNames(Rule[MoneyNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "CLDR v47 currency display names" - target_grammars = frozenset({"word_recognition"}) + target_semantics = frozenset({"word_recognition"}) requires_features = frozenset() def matches(self, notation: MoneyNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Money/rules/iso_4217_ed2015.py b/paxman/capabilities/Money/rules/iso_4217_ed2015.py index 7001de61..1ccec0ba 100644 --- a/paxman/capabilities/Money/rules/iso_4217_ed2015.py +++ b/paxman/capabilities/Money/rules/iso_4217_ed2015.py @@ -73,7 +73,7 @@ class SectionCode(Rule[MoneyNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "ISO 4217 currency codes" - target_grammars = frozenset({"code_recognition"}) + target_semantics = frozenset({"code_recognition"}) requires_features = frozenset() def matches(self, notation: MoneyNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Phone/grammar/e164_recognition.py b/paxman/capabilities/Phone/grammar/e164_recognition.py index 19364257..bcf4f153 100644 --- a/paxman/capabilities/Phone/grammar/e164_recognition.py +++ b/paxman/capabilities/Phone/grammar/e164_recognition.py @@ -57,6 +57,7 @@ class E164Grammar(Grammar[PhoneNotation]): """ name = "e164_recognition" + semantics = "e164_international" def recognize(self, text: str) -> list[RecognitionMatch[PhoneNotation]]: """Extract e164 patterns from text. diff --git a/paxman/capabilities/Phone/grammar/international_00_recognition.py b/paxman/capabilities/Phone/grammar/international_00_recognition.py index 5de15491..d66f909a 100644 --- a/paxman/capabilities/Phone/grammar/international_00_recognition.py +++ b/paxman/capabilities/Phone/grammar/international_00_recognition.py @@ -37,6 +37,7 @@ class International00Grammar(Grammar[PhoneNotation]): """ name = "international_00_recognition" + semantics = "e164_international" def recognize(self, text: str) -> list[RecognitionMatch[PhoneNotation]]: """Extract 00-prefixed international patterns from text. diff --git a/paxman/capabilities/Phone/grammar/national_recognition.py b/paxman/capabilities/Phone/grammar/national_recognition.py index 09c94015..1f41b412 100644 --- a/paxman/capabilities/Phone/grammar/national_recognition.py +++ b/paxman/capabilities/Phone/grammar/national_recognition.py @@ -47,6 +47,7 @@ class NationalGrammar(Grammar[PhoneNotation]): """ name = "national_recognition" + semantics = "national_recognition" def recognize(self, text: str) -> list[RecognitionMatch[PhoneNotation]]: """Extract national patterns from text. diff --git a/paxman/capabilities/Phone/grammar/tel_uri_recognition.py b/paxman/capabilities/Phone/grammar/tel_uri_recognition.py index d013712e..b8ede303 100644 --- a/paxman/capabilities/Phone/grammar/tel_uri_recognition.py +++ b/paxman/capabilities/Phone/grammar/tel_uri_recognition.py @@ -26,6 +26,7 @@ class TelUriGrammar(Grammar[PhoneNotation]): """ name = "tel_uri_recognition" + semantics = "tel_uri_recognition" def recognize(self, text: str) -> list[RecognitionMatch[PhoneNotation]]: """Extract tel: URI patterns from text. diff --git a/paxman/capabilities/Phone/rules/e164_ed2010.py b/paxman/capabilities/Phone/rules/e164_ed2010.py index 5d9c1f29..2c36ec89 100644 --- a/paxman/capabilities/Phone/rules/e164_ed2010.py +++ b/paxman/capabilities/Phone/rules/e164_ed2010.py @@ -66,7 +66,7 @@ class Section6_1InternationalNumber(Rule[PhoneNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 6.1 (number structure)" - target_grammars = frozenset({"e164_recognition", "international_00_recognition"}) + target_semantics = frozenset({"e164_international"}) requires_features = frozenset() def matches(self, notation: PhoneNotation, contract: Contract) -> bool: @@ -108,7 +108,7 @@ class Section6_2CountryCode(Rule[PhoneNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "Annex A (table of assigned country codes)" - target_grammars = frozenset({"e164_recognition", "international_00_recognition"}) + target_semantics = frozenset({"e164_international"}) requires_features = frozenset() def matches(self, notation: PhoneNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Phone/rules/nanp_ed2024.py b/paxman/capabilities/Phone/rules/nanp_ed2024.py index cfd19957..0549f198 100644 --- a/paxman/capabilities/Phone/rules/nanp_ed2024.py +++ b/paxman/capabilities/Phone/rules/nanp_ed2024.py @@ -90,7 +90,7 @@ class Section1_1NANPStructure(Rule[PhoneNotation]): strategy = RuleStrategy.REGEX provenance = PUBLICATION citation = "NANP numbering plan structure (NPA NXX-XXXX)" - target_grammars = frozenset({"national_recognition"}) + target_semantics = frozenset({"national_recognition"}) requires_features = frozenset() def matches(self, notation: PhoneNotation, contract: Contract) -> bool: @@ -145,7 +145,7 @@ class Section1_2ServiceNPA(Rule[PhoneNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "NANPA service NPA assignment table" - target_grammars = frozenset({"national_recognition"}) + target_semantics = frozenset({"national_recognition"}) requires_features = frozenset() def matches(self, notation: PhoneNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Phone/rules/rfc_3966_ed2004.py b/paxman/capabilities/Phone/rules/rfc_3966_ed2004.py index cf082dcb..baacbef8 100644 --- a/paxman/capabilities/Phone/rules/rfc_3966_ed2004.py +++ b/paxman/capabilities/Phone/rules/rfc_3966_ed2004.py @@ -31,7 +31,7 @@ class Section3TelUri(Rule[PhoneNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 3 (tel URI) / Section 3.1 (global numbers)" - target_grammars = frozenset({"tel_uri_recognition"}) + target_semantics = frozenset({"tel_uri_recognition"}) requires_features = frozenset() def matches(self, notation: PhoneNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/URL/grammar/absolute_uri_recognition.py b/paxman/capabilities/URL/grammar/absolute_uri_recognition.py index 1f1b8961..528f75e7 100644 --- a/paxman/capabilities/URL/grammar/absolute_uri_recognition.py +++ b/paxman/capabilities/URL/grammar/absolute_uri_recognition.py @@ -32,6 +32,7 @@ class AbsoluteUriRecognition(Grammar[URLNotation]): """Absolute-URI recognition: extracts scheme-anchored URI spans.""" name = "absolute_uri_recognition" + semantics = "absolute_uri_recognition" def recognize(self, text: str) -> list[RecognitionMatch[URLNotation]]: """Extract absolute-URI spans from text. diff --git a/paxman/capabilities/URL/rules/whatwg_url_standard.py b/paxman/capabilities/URL/rules/whatwg_url_standard.py index 190c12f0..9959ea8e 100644 --- a/paxman/capabilities/URL/rules/whatwg_url_standard.py +++ b/paxman/capabilities/URL/rules/whatwg_url_standard.py @@ -40,7 +40,7 @@ class WhatwgUrlStandard(Rule[URLNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 4.4 (basic URL parser); RFC 3986 §3.1 / RFC 3987 §2 grammar" - target_grammars = frozenset({"absolute_uri_recognition"}) + target_semantics = frozenset({"absolute_uri_recognition"}) requires_features = frozenset() def matches(self, notation: URLNotation, contract: Contract) -> bool: diff --git a/paxman/core/AGENTS.md b/paxman/core/AGENTS.md index 87239351..ae74a0e4 100644 --- a/paxman/core/AGENTS.md +++ b/paxman/core/AGENTS.md @@ -22,8 +22,8 @@ Core owns the domain vocabulary (pipeline value objects + `Rule`/`Grammar` ABCs) ## CONVENTIONS - **Layer discipline:** `paxman.core` must never import from `paxman.api`, `paxman.engine`, or `paxman.capabilities`. If a new core type needs something from outside, it does not belong here. - **Value objects** (`domain.py`): `@dataclass(frozen=True, slots=True)`. Spans are half-open with `len(raw_text) == end - start`, enforced in `__post_init__`. `GrammarRule` enforces lowercase names; `RecognizedRep.__hash__` handles unhashable list notations; `Candidate._provenance` is `init=False` and tuple-ized in `__init__`. -- **`Rule` subclasses** must declare `name`, `strategy`, `provenance`, `citation`, `target_grammars` (non-empty `frozenset[str]`), `requires_features` (`frozenset[str]`) as class attrs. `Rule.__init_subclass__` raises `TypeError` at import time for missing/mistyped metadata — keep it a hard import-time failure. `matches()`/`normalize()` never raise. -- **`Grammar` subclasses** must declare `name`; `recognize()` returns span-bearing `RecognitionMatch` only, never bare notation. +- **`Rule` subclasses** must declare `name`, `strategy`, `provenance`, `citation`, `target_semantics` (non-empty `frozenset[str]`), `requires_features` (`frozenset[str]`) as class attrs. `Rule.__init_subclass__` raises `TypeError` at import time for missing/mistyped metadata — keep it a hard import-time failure. `matches()`/`normalize()` never raise. +- **`Grammar` subclasses** must declare `name` and `semantics` — a non-empty string, the meaning id the grammar's notations carry (identity id by default; same-meaning grammars share one coalesced id). `Grammar.__init_subclass__` raises `TypeError` at import time for a missing, non-string, or empty `semantics`. `recognize()` returns span-bearing `RecognitionMatch` only, never bare notation. - **Contracts:** subclass `CapabilityContract`, never `Contract` directly. Set `DEFAULT_OUTPUT_FORMAT`/`OFFERED_OUTPUT_FORMATS`, set `capability_name` via `field(default=..., init=False)`. `active_grammars` is optional: it returns `None` by default (the engine then runs every shipped `get_grammars()` entry); only feature-gated capabilities (Email, IP, ISBN) override it. - **`output_format`** is always optional: `None`/`"default"`/`DEFAULT_OUTPUT_FORMAT` resolve to the default, offered formats resolve to themselves, anything else raises `ContractError`. Resolved once in `CapabilityContract.__post_init__`; subclasses with their own `__post_init__` call `super().__post_init__()` first. Note: `resolve_output_format` is imported lazily there to break the `capability_contract` ↔ `contract` import cycle. - **`pinned_rules` wins over `excluded_rules`** (non-`None` pins; empty tuple pins to nothing); `year` filtering still applies after pinning. diff --git a/paxman/core/domain.py b/paxman/core/domain.py index 9f18a6a8..d644074f 100644 --- a/paxman/core/domain.py +++ b/paxman/core/domain.py @@ -188,7 +188,7 @@ class Rule(ABC, Generic[NotationT]): strategy: RuleStrategy provenance: Provenance citation: str - target_grammars: ClassVar[frozenset[str]] + target_semantics: ClassVar[frozenset[str]] requires_features: ClassVar[frozenset[str]] def __init_subclass__(cls, **kwargs: object) -> None: @@ -199,7 +199,7 @@ def __init_subclass__(cls, **kwargs: object) -> None: "strategy", "provenance", "citation", - "target_grammars", + "target_semantics", "requires_features", ) missing = [attr for attr in required if not hasattr(cls, attr)] @@ -207,7 +207,7 @@ def __init_subclass__(cls, **kwargs: object) -> None: raise TypeError( f"{cls.__name__} must define Rule metadata: {', '.join(missing)}" ) - for attribute in ("target_grammars", "requires_features"): + for attribute in ("target_semantics", "requires_features"): value: Any = vars(cls).get(attribute, getattr(cls, attribute)) if type(value) is not frozenset: raise TypeError(f"{cls.__name__}.{attribute} must be frozenset[str]") @@ -215,8 +215,8 @@ def __init_subclass__(cls, **kwargs: object) -> None: vars(cls).get(attribute, getattr(cls, attribute)) ): raise TypeError(f"{cls.__name__}.{attribute} must be frozenset[str]") - if not cls.target_grammars: - raise TypeError(f"{cls.__name__}.target_grammars must be non-empty") + if not cls.target_semantics: + raise TypeError(f"{cls.__name__}.target_semantics must be non-empty") @abstractmethod def matches(self, notation: NotationT, contract: Contract) -> bool: ... @@ -229,6 +229,18 @@ class Grammar(ABC, Generic[NotationT]): """Base class for recognition grammars.""" name: str + semantics: ClassVar[str] + + def __init_subclass__(cls, **kwargs: object) -> None: + """Enforce Grammar metadata at class-definition time.""" + super().__init_subclass__(**kwargs) + if not hasattr(cls, "semantics"): + raise TypeError(f"{cls.__name__} must define Grammar metadata: semantics") + semantics: Any = vars(cls).get("semantics", cls.semantics) + if type(semantics) is not str: + raise TypeError(f"{cls.__name__}.semantics must be str") + if not cls.semantics: + raise TypeError(f"{cls.__name__}.semantics must be non-empty") @abstractmethod def recognize(self, text: str) -> list[RecognitionMatch[NotationT]]: diff --git a/paxman/core/extensions.py b/paxman/core/extensions.py index 330d10de..39d68dfd 100644 --- a/paxman/core/extensions.py +++ b/paxman/core/extensions.py @@ -68,7 +68,7 @@ def register_rule(capability_name: str, rule: Any) -> None: The ``rule`` parameter is ``Any`` because this is a runtime validation entry point: untyped callers may pass non-classes or non-Rule classes and the isinstance guard below provides the safety net. Rule metadata - (``target_grammars``, ``requires_features``, ...) is enforced by + (``target_semantics``, ``requires_features``, ...) is enforced by ``Rule.__init_subclass__`` at class-definition time; this function validates the class type and name uniqueness only. diff --git a/paxman/engine/orchestrator.py b/paxman/engine/orchestrator.py index ffa81383..000c7ba0 100644 --- a/paxman/engine/orchestrator.py +++ b/paxman/engine/orchestrator.py @@ -62,9 +62,13 @@ def run_capability(text: str, contract: Contract) -> ExecutionResult: *get_extended_grammars(capability.name), ] _assert_unique_names("grammar", all_grammars) - all_rules = [*capability.get_rules(), *_activated_rules(capability, contract)] + semantics_by_name = {g.name: g.semantics for g in all_grammars} + all_rules = [ + *capability.get_rules(), + *_activated_rules(capability, contract, semantics_by_name), + ] _assert_unique_names("rule", all_rules) - _validate_affinity(all_grammars, all_rules) + _validate_affinity(semantics_by_name, all_rules) recognitions = _recognize( text, all_grammars, @@ -74,7 +78,7 @@ def run_capability(text: str, contract: Contract) -> ExecutionResult: had_recognitions = len(recognitions) > 0 rules = _filter_rules(all_rules, contract) - candidates = _collect_candidates(capability, recognitions, rules) + candidates = _collect_candidates(capability, recognitions, rules, semantics_by_name) status = _determine_status(candidates, had_recognitions) canonical_value = _extract_canonical_value(candidates, status) @@ -93,8 +97,9 @@ def _assert_unique_names(kind: str, items: Sequence[Grammar[Any] | Rule[Any]]) - """Fail fast when a composed grammar or rule name is duplicated. Shipped names must never be shadowed or duplicated by community - extensions: a duplicate would make grammar-name routing and provenance - attribution ambiguous, so reject it at composition time (D4). + extensions: a duplicate would make provenance attribution ambiguous + (routing is semantic, but names remain the audit identity), so reject it + at composition time (D4). """ names = [item.name for item in items] duplicates = sorted({name for name in names if names.count(name) > 1}) @@ -274,21 +279,21 @@ def _filter_rules(all_rules: list[Rule[Any]], contract: Contract) -> list[Rule[A def _validate_affinity( - all_grammars: Sequence[Grammar[Any]], rules: list[Rule[Any]] + semantics_by_name: dict[str, str], rules: list[Rule[Any]] ) -> None: - """Ensure every rule's declared grammars exist in the composition. + """Ensure every rule's declared semantics exist in the composition. The composition covers shipped and community grammars alike; a dangling - grammar name would silently exclude a rule from ever running, so fail + semantics would silently exclude a rule from ever running, so fail fast at pipeline start rather than producing a wrong (e.g. INVALID) result. """ - known_grammars = {g.name for g in all_grammars} + known_semantics = set(semantics_by_name.values()) for rule in rules: - unknown = [g for g in rule.target_grammars if g not in known_grammars] + unknown = [s for s in rule.target_semantics if s not in known_semantics] if unknown: raise ContractError( - f"Rule {rule.name!r} declares unknown grammar(s) " - f"{sorted(unknown)}; available: {sorted(known_grammars)}" + f"Rule {rule.name!r} declares unknown semantics " + f"{sorted(unknown)}; available: {sorted(known_semantics)}" ) @@ -296,20 +301,25 @@ def _collect_candidates( capability: Capability[Any], recognitions: list[RecognizedRep[Any]], rules: list[Rule[Any]], + semantics_by_name: dict[str, str], ) -> list[Candidate]: """Match recognitions against rules and collect candidates. - Routes each recognition only to rules whose ``target_grammars`` includes the - producing grammar's name (ARCHITECTURE.md:201), formats each validated - value through the capability's ``format_value()`` seam, then dedups - identical candidate tuples so the candidate multiset is stable regardless - of routing. + Routes each recognition only to rules whose ``target_semantics`` includes + the producing grammar's semantics, formats each validated value through + the capability's ``format_value()`` seam, then dedups identical candidate + tuples so the candidate multiset is stable regardless of routing. + + The ``semantics_by_name[grammar_name]`` lookup cannot KeyError: + recognitions are produced only by grammars in the composed ``all_grammars`` + (``_recognize`` filters against ``supported_names``), the same list the map + is built from. """ candidates: list[Candidate] = [] for recognition in recognitions: grammar_name = recognition.grammar.grammar_name for rule in rules: - if grammar_name not in rule.target_grammars: + if semantics_by_name[grammar_name] not in rule.target_semantics: continue try: if rule.matches(recognition.notation, recognition.contract): @@ -343,7 +353,7 @@ def _dedup_candidates(candidates: list[Candidate]) -> list[Candidate]: Provenance is deterministic per (rule, grammar) pair, so collapsing on this key preserves all information while keeping the candidate multiset stable - under any future over-declaration of ``target_grammars``. + under any future over-declaration of ``target_semantics``. """ seen: set[tuple[str, str, str]] = set() deduped: list[Candidate] = [] @@ -383,18 +393,23 @@ def _extract_canonical_value( def _activated_rules( - capability: Capability[Any], contract: Contract + capability: Capability[Any], + contract: Contract, + semantics_by_name: dict[str, str], ) -> list[Rule[Any]]: """Community rules opt in like grammars: a rule runs only when the - contract names one of its ``target_grammars`` in ``extra_grammars``. + contract's ``extra_grammars`` resolve to one of its ``target_semantics``. - An un-opted community rule — even one targeting a shipped grammar — - never affects results, keeping extension behavior deterministic per - contract. + An unknown extra name keeps its own string as the semantics key, so a + rule declaring it (a dangling target) still fails fast in affinity + validation instead of being silently excluded. An un-opted community + rule — even one targeting a shipped grammar — never affects results, + keeping extension behavior deterministic per contract. """ extra_grammars = set(getattr(contract, "extra_grammars", ())) + extra_semantics = {semantics_by_name.get(n, n) for n in extra_grammars} return [ rule for rule in get_extended_rules(capability.name) - if extra_grammars & rule.target_grammars + if extra_semantics & rule.target_semantics ] diff --git a/tests/capabilities/currency/test_rules.py b/tests/capabilities/currency/test_rules.py index fb037bae..34ca7049 100644 --- a/tests/capabilities/currency/test_rules.py +++ b/tests/capabilities/currency/test_rules.py @@ -97,9 +97,9 @@ def test_strategy(self) -> None: """Verify the rule strategy enum.""" assert self.rule.strategy == RuleStrategy.LOOKUP_TABLE - def test_target_grammars(self) -> None: + def test_target_semantics(self) -> None: """The code rule targets only the code grammar.""" - assert self.rule.target_grammars == frozenset({"code_recognition"}) + assert self.rule.target_semantics == frozenset({"code_recognition"}) def test_requires_features_empty(self) -> None: """The ISO rule never gates on contract features (always runs).""" @@ -236,9 +236,9 @@ def test_strategy(self) -> None: """Verify the rule strategy enum.""" assert self.rule.strategy == RuleStrategy.LOOKUP_TABLE - def test_target_grammars(self) -> None: + def test_target_semantics(self) -> None: """The symbol rule targets only the symbol grammar.""" - assert self.rule.target_grammars == frozenset({"symbol_recognition"}) + assert self.rule.target_semantics == frozenset({"symbol_recognition"}) def test_requires_features_empty(self) -> None: """Never gate on default_currency: a shared bare symbol yields @@ -306,9 +306,9 @@ def test_strategy(self) -> None: """Verify the rule strategy enum.""" assert self.rule.strategy == RuleStrategy.LOOKUP_TABLE - def test_target_grammars(self) -> None: + def test_target_semantics(self) -> None: """The word rule targets only the word grammar.""" - assert self.rule.target_grammars == frozenset({"word_recognition"}) + assert self.rule.target_semantics == frozenset({"word_recognition"}) def test_requires_features_empty(self) -> None: """The CLDR name rule never gates on contract features.""" diff --git a/tests/capabilities/date/test_grammar.py b/tests/capabilities/date/test_grammar.py index 0dae85ce..fdb3ec00 100644 --- a/tests/capabilities/date/test_grammar.py +++ b/tests/capabilities/date/test_grammar.py @@ -41,6 +41,17 @@ def test_returns_empty_for_no_match(self) -> None: result = grammar.recognize("No dates here") assert result == [] + def test_does_not_match_embedded_in_digits(self) -> None: + """A date glued to surrounding digits is not recognized. + + The digit lookarounds prevent partial matches inside longer digit + runs (e.g. IDs). + """ + grammar = ISO8601DateGrammar() + assert grammar.recognize("12026-07-26") == [] + assert grammar.recognize("2026-07-261") == [] + assert grammar.recognize("12026-07-261") == [] + def test_grammar_name(self) -> None: grammar = ISO8601DateGrammar() assert grammar.name == "iso8601_recognition" @@ -88,6 +99,18 @@ def test_grammar_name(self) -> None: grammar = USDateGrammar() assert grammar.name == "us_recognition" + def test_does_not_match_embedded_in_digits(self) -> None: + """A date glued to surrounding digits is not recognized. + + Both year-length variants carry digit lookarounds, preventing + partial matches inside longer digit runs (e.g. IDs). + """ + grammar = USDateGrammar() + assert grammar.recognize("1207/26/2026") == [] + assert grammar.recognize("07/26/20261") == [] + assert grammar.recognize("1207/26/26") == [] + assert grammar.recognize("07/26/261") == [] + def test_emits_spans(self) -> None: result = self.grammar.recognize("x 07/26/2026 y") assert len(result) == 1 @@ -126,6 +149,18 @@ def test_grammar_name(self) -> None: grammar = EuropeanDateGrammar() assert grammar.name == "european_recognition" + def test_does_not_match_embedded_in_digits(self) -> None: + """A date glued to surrounding digits is not recognized. + + Both year-length variants carry digit lookarounds, preventing + partial matches inside longer digit runs (e.g. IDs). + """ + grammar = EuropeanDateGrammar() + assert grammar.recognize("1226/07/2026") == [] + assert grammar.recognize("26/07/20261") == [] + assert grammar.recognize("1226/07/26") == [] + assert grammar.recognize("26/07/261") == [] + def test_emits_spans(self) -> None: result = self.grammar.recognize("x 26/07/2026 y") assert len(result) == 1 @@ -176,6 +211,17 @@ def test_does_not_match_us_or_european_order(self) -> None: assert grammar.recognize("07/26/2026") == [] assert grammar.recognize("26/07/2026") == [] + def test_does_not_match_embedded_in_digits(self) -> None: + """A date run glued to surrounding digits is not recognized. + + The digit lookarounds prevent partial matches inside longer digit + runs (e.g. IDs), mirroring the 2-digit US/European patterns. + """ + grammar = SlashISODateGrammar() + assert grammar.recognize("12026/07/26") == [] + assert grammar.recognize("2026/07/261") == [] + assert grammar.recognize("12026/07/261") == [] + def test_grammar_name(self) -> None: grammar = SlashISODateGrammar() assert grammar.name == "slash_iso_recognition" diff --git a/tests/capabilities/isbn/test_rules.py b/tests/capabilities/isbn/test_rules.py index 85723c7b..8fc9c1b4 100644 --- a/tests/capabilities/isbn/test_rules.py +++ b/tests/capabilities/isbn/test_rules.py @@ -179,25 +179,25 @@ def test_rule_conventions(self) -> None: assert self.isbn13.name == "Section 5.3-isbn13-check-digit" assert self.isbn13.strategy == RuleStrategy.PARSER assert self.isbn13.citation == "Section 5.3 (ISBN-13 check digit)" - assert self.isbn13.target_grammars == frozenset({"isbn13_recognition"}) + assert self.isbn13.target_semantics == frozenset({"isbn13_recognition"}) assert self.isbn13.requires_features == frozenset() assert self.gs1.name == "Section 4.2-gs1-prefix" assert self.gs1.strategy == RuleStrategy.LOOKUP_TABLE assert self.gs1.citation == "Section 4.2 (GS1 prefix)" - assert self.gs1.target_grammars == frozenset({"isbn13_recognition"}) + assert self.gs1.target_semantics == frozenset({"isbn13_recognition"}) assert self.gs1.requires_features == frozenset() assert self.isbn10.name == "Section 6-isbn10-check-digit" assert self.isbn10.strategy == RuleStrategy.PARSER assert self.isbn10.citation == "Section 6 (ISBN-10 check digit)" - assert self.isbn10.target_grammars == frozenset({"isbn10_recognition"}) + assert self.isbn10.target_semantics == frozenset({"isbn10_recognition"}) assert self.isbn10.requires_features == frozenset() assert self.range.name == "Section 4-registrant-range" assert self.range.strategy == RuleStrategy.LOOKUP_TABLE assert self.range.citation == "Section 4 (registrant range)" - assert self.range.target_grammars == frozenset( + assert self.range.target_semantics == frozenset( {"isbn13_recognition", "isbn10_recognition"} ) diff --git a/tests/capabilities/money/test_rules.py b/tests/capabilities/money/test_rules.py index 666c27e5..3a439c71 100644 --- a/tests/capabilities/money/test_rules.py +++ b/tests/capabilities/money/test_rules.py @@ -142,9 +142,9 @@ def test_strategy(self) -> None: """Verify the rule strategy enum.""" assert self.rule.strategy == RuleStrategy.LOOKUP_TABLE - def test_target_grammars(self) -> None: + def test_target_semantics(self) -> None: """The code rule targets only the code grammar.""" - assert self.rule.target_grammars == frozenset({"code_recognition"}) + assert self.rule.target_semantics == frozenset({"code_recognition"}) def test_requires_features_empty(self) -> None: """The ISO rule never gates on contract features (always runs).""" @@ -260,9 +260,9 @@ def test_strategy(self) -> None: """Verify the rule strategy enum.""" assert self.rule.strategy == RuleStrategy.LOOKUP_TABLE - def test_target_grammars(self) -> None: + def test_target_semantics(self) -> None: """The symbol rule targets only the symbol grammar.""" - assert self.rule.target_grammars == frozenset({"symbol_recognition"}) + assert self.rule.target_semantics == frozenset({"symbol_recognition"}) def test_requires_features_empty(self) -> None: """Never gate on dollar_sign_currency: bare $ yields INVALID, not MISSING.""" @@ -375,9 +375,9 @@ def test_strategy(self) -> None: """Verify the rule strategy enum.""" assert self.rule.strategy == RuleStrategy.LOOKUP_TABLE - def test_target_grammars(self) -> None: + def test_target_semantics(self) -> None: """The word rule targets only the word grammar.""" - assert self.rule.target_grammars == frozenset({"word_recognition"}) + assert self.rule.target_semantics == frozenset({"word_recognition"}) def test_requires_features_empty(self) -> None: """The CLDR name rule never gates on contract features.""" diff --git a/tests/capabilities/url/test_rule.py b/tests/capabilities/url/test_rule.py index f2a5d9b6..c1dddb5b 100644 --- a/tests/capabilities/url/test_rule.py +++ b/tests/capabilities/url/test_rule.py @@ -44,7 +44,7 @@ def test_rule_metadata(self) -> None: """Rule metadata matches §1 (homogeneity contract, Money style).""" assert self.rule.name == "WHATWG URL Standard" assert self.rule.strategy == RuleStrategy.PARSER - assert self.rule.target_grammars == frozenset({"absolute_uri_recognition"}) + assert self.rule.target_semantics == frozenset({"absolute_uri_recognition"}) assert self.rule.requires_features == frozenset() assert ( self.rule.citation diff --git a/tests/integration/test_feature_gating.py b/tests/integration/test_feature_gating.py index 4a8d8b35..b3267948 100644 --- a/tests/integration/test_feature_gating.py +++ b/tests/integration/test_feature_gating.py @@ -55,12 +55,13 @@ class _NameRecognitionGrammar(Grammar[CountryNotation]): """Grammar emitting the CLDR localized key "Estados Unidos". Uses the real ``name_recognition`` grammar name so - ``SectionLocalizedNames``' declared ``target_grammars`` resolves, without + ``SectionLocalizedNames``' declared ``target_semantics`` resolves, without relying on the real ``NameGrammar``'s English/historical/Chinese lookup tables (localized recognition remediation is F3 scope, not F2). """ name = "name_recognition" + semantics = "name_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CountryNotation]]: return [ @@ -145,7 +146,7 @@ class _DanglingFeatureRule(Rule[CountryNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"name_recognition"}) + target_semantics = frozenset({"name_recognition"}) requires_features = frozenset({"not_a_contract_field"}) def matches(self, notation: CountryNotation, contract: object) -> bool: @@ -211,7 +212,7 @@ class _DanglingGrammarRule(Rule[CountryNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"missing_grammar"}) + target_semantics = frozenset({"missing_grammar"}) requires_features = frozenset() def matches(self, notation: CountryNotation, contract: object) -> bool: diff --git a/tests/integration/test_format_value_seam.py b/tests/integration/test_format_value_seam.py index c862bc06..96c72010 100644 --- a/tests/integration/test_format_value_seam.py +++ b/tests/integration/test_format_value_seam.py @@ -48,6 +48,7 @@ class _TokenGrammar(Grammar[_TokenNotation]): """Grammar that recognizes a single fixed token.""" name = "token_grammar" + semantics = "token_grammar" def recognize(self, text: str) -> list[RecognitionMatch[_TokenNotation]]: return [ @@ -75,7 +76,7 @@ class _TokenRule(Rule[_TokenNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"token_grammar"}) + target_semantics = frozenset({"token_grammar"}) requires_features = frozenset() def matches(self, notation: _TokenNotation, contract: Contract) -> bool: @@ -144,6 +145,7 @@ class _DualTokenGrammar(Grammar[_TokenNotation]): """Grammar that recognizes two distinct tokens.""" name = "dual_token_grammar" + semantics = "dual_token_grammar" def recognize(self, text: str) -> list[RecognitionMatch[_TokenNotation]]: return [ @@ -177,7 +179,7 @@ class _DualTokenRule(Rule[_TokenNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"dual_token_grammar"}) + target_semantics = frozenset({"dual_token_grammar"}) requires_features = frozenset() def matches(self, notation: _TokenNotation, contract: Contract) -> bool: diff --git a/tests/integration/test_grammar_extensions.py b/tests/integration/test_grammar_extensions.py index 4a62873b..2e3bd35b 100644 --- a/tests/integration/test_grammar_extensions.py +++ b/tests/integration/test_grammar_extensions.py @@ -36,6 +36,7 @@ class DotDateGrammar(Grammar[DateNotation]): """Community test double: recognizes YYYY.MM.DD (dot separator).""" name = "dot_date_recognition" + semantics = "dot_date_recognition" _PATTERN = re.compile(r"\b(\d{4})\.(\d{2})\.(\d{2})\b") def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: @@ -55,6 +56,7 @@ class SecondDateGrammar(Grammar[DateNotation]): """Second community test double — same shape, different normalization.""" name = "second_recognition" + semantics = "second_recognition" _PATTERN = re.compile(r"\b(\d{4})\.(\d{2})\.(\d{2})\b") def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: @@ -74,6 +76,7 @@ class ClashingDateGrammar(Grammar[DateNotation]): """Community grammar whose name collides with a shipped Date grammar.""" name = "iso8601_recognition" + semantics = "iso8601_recognition" def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: return [] @@ -86,7 +89,7 @@ class DotDateRule(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "community test double" - target_grammars = frozenset({"dot_date_recognition"}) + target_semantics = frozenset({"dot_date_recognition"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: @@ -109,7 +112,7 @@ class SecondDateRule(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "community test double" - target_grammars = frozenset({"second_recognition"}) + target_semantics = frozenset({"second_recognition"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: @@ -132,7 +135,7 @@ class CommunityISO8601Rule(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "community test double" - target_grammars = frozenset({"iso8601_recognition"}) + target_semantics = frozenset({"iso8601_calendar_date"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: @@ -146,13 +149,14 @@ def normalize(self, notation: DateNotation, contract: Contract) -> str: class DanglingDateRule(Rule[DateNotation]): - """Community rule whose target_grammars references a missing grammar.""" + """Community rule whose target_semantics references a semantics id that + no grammar claims.""" name = "dangling_date_rule" strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "community test double" - target_grammars = frozenset({"no_such_grammar"}) + target_semantics = frozenset({"no_such_grammar"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: @@ -235,12 +239,12 @@ def test_community_grammar_collision_with_shipped_raises(self) -> None: @pytest.mark.integration def test_community_rule_dangling_target_raises(self) -> None: - """An opted-in community rule naming a missing grammar fails fast.""" + """An opted-in community rule naming a missing semantics fails fast.""" register_rule("date", DanglingDateRule) contract = DateContract( extra_grammars=("dot_date_recognition", "no_such_grammar") ) - with pytest.raises(ContractError, match="unknown grammar"): + with pytest.raises(ContractError, match="unknown semantics"): run_capability("2024.01.01", contract) @pytest.mark.integration @@ -308,3 +312,27 @@ def test_community_rule_on_shipped_grammar_requires_activation(self) -> None: c.validation_rule == "community_iso8601_rule" for c in activated_result.candidates ) + + @pytest.mark.integration + def test_rule_opt_in_via_raw_semantics_id_without_grammar(self) -> None: + """A raw semantics id in ``extra_grammars`` activates rules targeting it + without opting in any community grammar. + + Locks the README-documented raw-name fallback + (``semantics_by_name.get(n, n)``): ``iso8601_calendar_date`` is a + known semantics id but not a grammar name, so the community rule + targeting it fires on shipped ISO recognitions while no community + grammar is activated (fail-fast ``ContractError`` applies only to ids + no grammar claims). + """ + register_rule("date", CommunityISO8601Rule) + + result = run_capability( + "2026-01-15", DateContract(extra_grammars=("iso8601_calendar_date",)) + ) + assert any( + c.validation_rule == "community_iso8601_rule" for c in result.candidates + ) + assert all( + c.recognition_rule == "iso8601_recognition" for c in result.candidates + ) diff --git a/tests/integration/test_pipeline.py b/tests/integration/test_pipeline.py index cf8815cb..cf723610 100644 --- a/tests/integration/test_pipeline.py +++ b/tests/integration/test_pipeline.py @@ -128,6 +128,7 @@ class CrashGrammar(Grammar[EmailNotation]): """Grammar whose recognize() always raises.""" name = "crash_grammar" + semantics = "crash_grammar" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: raise RuntimeError("grammar crashed") @@ -137,6 +138,7 @@ class SimpleGrammar(Grammar[EmailNotation]): """Grammar that returns a fixed notation.""" name = "simple_grammar" + semantics = "simple_grammar" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: return [ @@ -164,7 +166,7 @@ class StubRule(Rule[EmailNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"crash_grammar"}) + target_semantics = frozenset({"crash_grammar"}) requires_features = frozenset() def matches(self, notation: EmailNotation, contract: object) -> bool: @@ -189,7 +191,7 @@ class ExplodingRule(Rule[EmailNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"simple_grammar"}) + target_semantics = frozenset({"simple_grammar"}) requires_features = frozenset() def matches(self, notation: EmailNotation, contract: object) -> bool: @@ -358,6 +360,7 @@ class _PhantomGrammar(Grammar[EmailNotation]): """Grammar referenced by a rule that does not exist in the capability.""" name = "phantom_grammar" + semantics = "phantom_grammar" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: return [ @@ -371,7 +374,7 @@ def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: class _PhantomRule(Rule[EmailNotation]): - """Rule whose target_grammars names a non-existent grammar.""" + """Rule whose target_semantics names a non-existent grammar.""" name = "phantom_rule" strategy = RuleStrategy.REGEX @@ -385,7 +388,7 @@ class _PhantomRule(Rule[EmailNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"does_not_exist"}) + target_semantics = frozenset({"does_not_exist"}) requires_features = frozenset() def matches(self, notation: EmailNotation, contract: object) -> bool: @@ -437,7 +440,7 @@ def output_format(self) -> str | None: class TestGrammarRuleAffinity: - """F1: grammar→rule affinity declared via Rule.target_grammars.""" + """F1: grammar→rule affinity declared via Rule.target_semantics.""" @pytest.mark.integration @pytest.mark.parametrize("output_format", [None, "ISO", "US"]) diff --git a/tests/integration/test_recognition_seam.py b/tests/integration/test_recognition_seam.py index 045984df..14b76edd 100644 --- a/tests/integration/test_recognition_seam.py +++ b/tests/integration/test_recognition_seam.py @@ -49,6 +49,7 @@ class _ProbeLongGrammar(Grammar[_ProbeNotation]): """ name = "probe_long" + semantics = "probe_long" _patterns = (re.compile(r"AAAA"), re.compile(r"AA")) def recognize(self, text: str) -> list[RecognitionMatch[_ProbeNotation]]: @@ -70,6 +71,7 @@ class _ProbeShortGrammar(Grammar[_ProbeNotation]): """Recognizes 'AA' only.""" name = "probe_short" + semantics = "probe_short" def recognize(self, text: str) -> list[RecognitionMatch[_ProbeNotation]]: return [ @@ -98,7 +100,7 @@ class _LongRule(Rule[_ProbeNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"probe_long"}) + target_semantics = frozenset({"probe_long"}) requires_features = frozenset() def matches(self, notation: _ProbeNotation, contract: Contract) -> bool: @@ -123,7 +125,7 @@ class _ShortRule(Rule[_ProbeNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"probe_short"}) + target_semantics = frozenset({"probe_short"}) requires_features = frozenset() def matches(self, notation: _ProbeNotation, contract: Contract) -> bool: @@ -375,6 +377,7 @@ def test_fallback_never_activates_community_grammars(self) -> None: class _CommunityGrammar(Grammar[_ProbeNotation]): name = "probe_community" + semantics = "probe_community" def recognize(self, text: str) -> list[RecognitionMatch[_ProbeNotation]]: calls.append(self.name) @@ -400,7 +403,7 @@ class _CommunityRule(Rule[_ProbeNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"probe_community"}) + target_semantics = frozenset({"probe_community"}) requires_features = frozenset() def matches(self, notation: _ProbeNotation, contract: Contract) -> bool: diff --git a/tests/unit/test_capability.py b/tests/unit/test_capability.py index 08483154..2f4ddcb6 100644 --- a/tests/unit/test_capability.py +++ b/tests/unit/test_capability.py @@ -15,6 +15,7 @@ class StubGrammar(Grammar): """Minimal concrete grammar for testing Capability.""" name: str = "stub_grammar" + semantics = "stub_grammar" def recognize(self, text: str) -> list[Notation]: return [] @@ -35,7 +36,7 @@ class StubRule(Rule): publication_year=2024, ) citation: str = "test citation" - target_grammars = frozenset({"stub_grammar"}) + target_semantics = frozenset({"stub_grammar"}) requires_features = frozenset() def matches(self, notation: Notation, contract: Contract) -> bool: diff --git a/tests/unit/test_discovery.py b/tests/unit/test_discovery.py index f54e7ad5..c316cadc 100644 --- a/tests/unit/test_discovery.py +++ b/tests/unit/test_discovery.py @@ -62,6 +62,7 @@ class DotDateGrammar(Grammar): """Minimal community grammar for extension delegation tests.""" name = "dot_date_recognition" + semantics = "dot_date_recognition" def recognize(self, text: str) -> list[RecognitionMatch]: return [] diff --git a/tests/unit/test_extensions.py b/tests/unit/test_extensions.py index d5d61aab..618e31ba 100644 --- a/tests/unit/test_extensions.py +++ b/tests/unit/test_extensions.py @@ -43,6 +43,7 @@ class _DotDateGrammar(Grammar[Any]): """Minimal community grammar test double.""" name = "dot_date_recognition" + semantics = "dot_date_recognition" def recognize(self, text: str) -> list[RecognitionMatch[Any]]: return [] @@ -52,6 +53,7 @@ class _SecondGrammar(Grammar[Any]): """Second grammar — tests registration order.""" name = "second_recognition" + semantics = "second_recognition" def recognize(self, text: str) -> list[RecognitionMatch[Any]]: return [] @@ -61,6 +63,7 @@ class _NamelessGrammar(Grammar[Any]): """Grammar with an empty name — tests name validation.""" name = "" + semantics = "nameless_grammar" def recognize(self, text: str) -> list[RecognitionMatch[Any]]: return [] @@ -70,6 +73,7 @@ class _MixedCaseGrammar(Grammar[Any]): """Grammar with a mixed-case name — tests lowercase enforcement.""" name = "DotDateRecognition" + semantics = "DotDateRecognition" def recognize(self, text: str) -> list[RecognitionMatch[Any]]: return [] @@ -82,7 +86,7 @@ class _DotDateRule(Rule[Any]): strategy = RuleStrategy.PARSER provenance = _PROVENANCE citation = "community test double" - target_grammars = frozenset({"dot_date_recognition"}) + target_semantics = frozenset({"dot_date_recognition"}) requires_features = frozenset() def matches(self, notation: Any, contract: Any) -> bool: @@ -99,7 +103,7 @@ class _NamelessRule(Rule[Any]): strategy = RuleStrategy.PARSER provenance = _PROVENANCE citation = "community test double" - target_grammars = frozenset({"x"}) + target_semantics = frozenset({"x"}) requires_features = frozenset() def matches(self, notation: Any, contract: Any) -> bool: diff --git a/tests/unit/test_grammar_semantics_consistency.py b/tests/unit/test_grammar_semantics_consistency.py new file mode 100644 index 00000000..1c300138 --- /dev/null +++ b/tests/unit/test_grammar_semantics_consistency.py @@ -0,0 +1,306 @@ +"""D8 — same-semantics grammars must produce identical notation field mappings +and canonicalization; guards semantic affinity routing. + +The affinity-routing engine treats every grammar claiming the same +``semantics`` id as interchangeable: any member of a group may recognize an +input, and its notation is routed to the group's shared rules. A group whose +members map the same input to different notation fields (or whose shared rule +canonicalizes differently) would resolve the same text differently depending +on which member happened to recognize it — silent nondeterminism. This guard +enumerates all shipped grammar classes, groups them by ``semantics`` within +each capability, and pins every seeded group's members to a single canonical +mapping: each member must recognize at least one probe row into the group's +expected notation, and the group's shared rule must canonicalize every match +identically. Group members may recognize disjoint input sets (e.g. the dash +and slash ISO grammars), so agreement is asserted member-vs-table — no +cross-member comparison is claimed. +""" + +from __future__ import annotations + +from typing import Any, NamedTuple + +import pytest + +from paxman.capabilities import ( + IP, + ISBN, + URL, + Country, + Currency, + Date, + Email, + Money, + Phone, +) +from paxman.capabilities.Date.contract import DateContract +from paxman.capabilities.Date.notation import DateNotation +from paxman.capabilities.Date.rules.iso_8601_ed2019 import Section431CalendarDate +from paxman.capabilities.Email.contract import EmailContract +from paxman.capabilities.Email.notation import EmailNotation +from paxman.capabilities.Email.rules.rfc_5322_ed2008 import Section341AddrSpec +from paxman.capabilities.Phone.contract import PhoneContract +from paxman.capabilities.Phone.notation import PhoneNotation +from paxman.capabilities.Phone.rules.e164_ed2010 import Section6_1InternationalNumber +from paxman.core.capability_contract import CapabilityContract +from paxman.core.domain import Grammar, Rule + +_SHIPPED_CAPABILITIES = [Country, Currency, Date, Email, IP, ISBN, Money, Phone, URL] + +# D7 no-coalesce ids: groups that must NEVER grow a second member. The +# date formats and the six identity singletons are distinct enough that +# coalescing them would silently change what they resolve. +_NO_COALESCE_SEMANTICS = ( + "us_calendar_date", + "european_calendar_date", + "name_recognition", + "alpha2_recognition", + "alpha3_recognition", + "numeric_recognition", + "isbn13_recognition", + "isbn10_recognition", +) + +# Contract class used to drive each seeded group's shared rule ``normalize()``. +# The right contract per group keeps the guard honest if a rule ever starts +# reading contract fields. +_CONTRACTS: dict[str, type[CapabilityContract]] = { + "iso8601_calendar_date": DateContract, + "rfc5322_addr_spec": EmailContract, + "e164_international": PhoneContract, +} + + +class _ProbeRow(NamedTuple): + """One input run through every member of a semantics group. + + ``expected_notation`` is deliberately untyped: the probe table spans + capabilities, so each row's notation type is the group's own. + """ + + input: str + expected_notation: object + expected_canonical: str + + +# Probe rows keyed by semantics id. Each key must name a real group in the +# shipped grammar enumeration (test A); each member of a group must recognize +# the probe input into the identical notation, and the group's shared rule +# must canonicalize it identically (test B). +_PROBE_ROWS: dict[str, tuple[type[Rule[Any]], tuple[_ProbeRow, ...]]] = { + "iso8601_calendar_date": ( + Section431CalendarDate, + ( + _ProbeRow( + input="2026-01-15", + expected_notation=DateNotation(N1="2026", N2="01", N3="15"), + expected_canonical="2026-01-15", + ), + _ProbeRow( + input="2026/01/15", + expected_notation=DateNotation(N1="2026", N2="01", N3="15"), + expected_canonical="2026-01-15", + ), + ), + ), + "rfc5322_addr_spec": ( + Section341AddrSpec, + ( + _ProbeRow( + input="user@example.com", + expected_notation=EmailNotation( + local_part="user", domain_part="example.com" + ), + expected_canonical="user@example.com", + ), + _ProbeRow( + input="user at example dot com", + expected_notation=EmailNotation( + local_part="user", domain_part="example.com" + ), + expected_canonical="user@example.com", + ), + ), + ), + "e164_international": ( + Section6_1InternationalNumber, + ( + _ProbeRow( + input="+15551234567", + expected_notation=PhoneNotation(shape="e164", value="15551234567"), + expected_canonical="+15551234567", + ), + _ProbeRow( + input="0015551234567", + expected_notation=PhoneNotation(shape="e164", value="15551234567"), + expected_canonical="+15551234567", + ), + ), + ), +} + + +def _group_shipped_grammars_by_capability_semantics() -> dict[ + str, dict[str, list[type[Grammar[Any]]]] +]: + """Group shipped grammar classes per capability by their ``semantics`` id. + + The affinity-routing engine treats grammars as interchangeable only within + one capability, so groups are scoped per capability: a semantics id reused + across capabilities (Currency and Money both declaring ``code_recognition`` + etc.) yields separate per-capability groups that never co-route and must + not be probed as one unit. + """ + groups: dict[str, dict[str, list[type[Grammar[Any]]]]] = {} + for capability in _SHIPPED_CAPABILITIES: + per_capability: dict[str, list[type[Grammar[Any]]]] = {} + for grammar in capability().get_grammars(): + per_capability.setdefault(grammar.semantics, []).append(type(grammar)) + groups[capability.__name__] = per_capability + return groups + + +@pytest.mark.unit +def test_probe_keys_name_real_semantics_groups() -> None: + """Every probe-table key must be a real semantics group in the enumeration.""" + groups = _group_shipped_grammars_by_capability_semantics() + assert all( + any(key in per_capability for per_capability in groups.values()) + for key in _PROBE_ROWS + ) + + +@pytest.mark.unit +def test_same_semantics_grammars_agree_on_notation_and_canonical() -> None: + """Members of a seeded semantics group pin the group's canonical mapping. + + Every member must recognize at least one probe row into the group's + expected notation, and the group's shared rule must canonicalize every + match to the expected canonical value. Because members may recognize + disjoint input sets, agreement is asserted member-vs-table (against the + group's single expected mapping), not by comparing members on one input. + A member that recognizes none of the probes — or a probe that no member + recognizes — fails loudly instead of passing silently. + """ + groups = _group_shipped_grammars_by_capability_semantics() + for semantics, (rule_cls, probes) in _PROBE_ROWS.items(): + rule = rule_cls() + member_lists = [ + per_capability.get(semantics, ()) for per_capability in groups.values() + ] + assert sum(1 for members in member_lists if members) == 1, ( + f"semantics {semantics!r} spans multiple capabilities; probe rows " + "must stay scoped to one capability" + ) + members = [member for group_members in member_lists for member in group_members] + for member_cls in members: + member = member_cls() + matched_any = False + for probe in probes: + matches = member.recognize(probe.input) + if not matches: + continue + matched_any = True + assert all(m.notation == probe.expected_notation for m in matches), ( + f"{member_cls.__name__} mapped {probe.input!r} to " + f"{[m.notation for m in matches]}, expected " + f"{probe.expected_notation!r}" + ) + assert ( + rule.normalize(matches[0].notation, _CONTRACTS[semantics]()) + == probe.expected_canonical + ) + assert matched_any, ( + f"{member_cls.__name__} (semantics {semantics!r}) recognized none " + "of the probe rows" + ) + for probe in probes: + assert any(member_cls().recognize(probe.input) for member_cls in members), ( + f"probe {probe.input!r} matched by no member of {semantics!r}" + ) + + +@pytest.mark.unit +def test_every_shipped_grammar_belongs_to_one_semantics_group() -> None: + """No shipped grammar is dropped or duplicated by the semantics grouping. + + Every grammar enumerated via ``get_grammars()`` must land in exactly one + group with a non-empty semantics id; a dropped or double-counted grammar + would break the member-count equality. + """ + groups = _group_shipped_grammars_by_capability_semantics() + shipped_count = sum( + len(capability().get_grammars()) for capability in _SHIPPED_CAPABILITIES + ) + assert ( + sum( + len(members) + for per_capability in groups.values() + for members in per_capability.values() + ) + == shipped_count + ) + assert all( + semantics for per_capability in groups.values() for semantics in per_capability + ) + + +@pytest.mark.unit +def test_every_grammar_semantics_claimed_by_rule_target() -> None: + """Every shipped grammar's semantics is claimed by an in-capability rule. + + A grammar whose semantics no rule declares routes every recognition to + zero rules — input matching only it yields INVALID instead of resolving, + silently. Requiring in-capability rule-target coverage keeps + ``_collect_candidates()`` free of unroutable shipped grammars; a grammar + added without a claiming rule fails here at test time. + """ + for capability in _SHIPPED_CAPABILITIES: + instance = capability() + targets = {s for rule in instance.get_rules() for s in rule.target_semantics} + for grammar in instance.get_grammars(): + assert grammar.semantics in targets, ( + f"{capability.__name__} grammar {grammar.name!r} declares " + f"semantics {grammar.semantics!r} claimed by no shipped rule " + f"(rule targets: {sorted(targets)})" + ) + + +@pytest.mark.unit +def test_every_multi_member_semantics_group_has_probe_rows() -> None: + """A coalesced group must be seeded or the guard fails loudly. + + The affinity-routing engine only treats grammars as interchangeable within + one capability, so a multi-member group arises only from a coalescing + inside a capability. Cross-capability id reuse (Currency and Money both + declaring ``code_recognition`` etc.) is per-capability identity — those + grammars never co-route — and must not demand probe rows. A future + coalescing that adds a group without probe rows bypasses the same-notation + field-mapping guarantee and fails here. + """ + multi_member_ids: set[str] = set() + for capability in _SHIPPED_CAPABILITIES: + counts: dict[str, int] = {} + for grammar in capability().get_grammars(): + counts[grammar.semantics] = counts.get(grammar.semantics, 0) + 1 + multi_member_ids.update( + semantics for semantics, count in counts.items() if count > 1 + ) + assert multi_member_ids <= set(_PROBE_ROWS) + + +@pytest.mark.unit +def test_d7_no_coalesce_semantics_groups_stay_singleton() -> None: + """The D7-locked groups must never grow a second member. + + ``us_calendar_date``/``european_calendar_date`` are renamed singletons and + the other six are identity singletons; coalescing any of them would change + what the shared semantics resolves to. Each id must total exactly one + member across all capabilities. + """ + groups = _group_shipped_grammars_by_capability_semantics() + for semantics in _NO_COALESCE_SEMANTICS: + total = sum( + len(per_capability.get(semantics, ())) for per_capability in groups.values() + ) + assert total == 1, f"{semantics!r} must stay a singleton, found {total}" diff --git a/tests/unit/test_grammar_semantics_metadata.py b/tests/unit/test_grammar_semantics_metadata.py new file mode 100644 index 00000000..0f654093 --- /dev/null +++ b/tests/unit/test_grammar_semantics_metadata.py @@ -0,0 +1,140 @@ +"""Tests for grammar ``semantics`` metadata on shipped grammars.""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from paxman.capabilities import ( + IP, + ISBN, + URL, + Country, + Currency, + Date, + Email, + Money, + Phone, +) +from paxman.core.domain import Grammar, RecognitionMatch + +# Semantics ids that legitimately differ from a grammar's ``name`` (semantic +# affinity routing, ADR-0003): coalesced ids shared by several grammars +# (``iso8601_calendar_date``, ``rfc5322_addr_spec``, ``e164_international``) +# and renamed singletons (``us_calendar_date``, ``european_calendar_date``). +# A grammar in this set declares ``semantics`` differing from its ``name`` +# without failing the identity check. +_COALESCED_SEMANTICS: frozenset[str] = frozenset( + { + "iso8601_calendar_date", + "us_calendar_date", + "european_calendar_date", + "rfc5322_addr_spec", + "e164_international", + } +) + + +class TestGrammarSemanticsMetadata: + @pytest.mark.unit + def test_shipped_grammars_declare_semantics_identity(self) -> None: + """Every shipped grammar declares ``semantics``: identity with its name + for non-coalesced grammars, or one of the coalesced ids (an explicit + allowlist) for grammars sharing a semantic group.""" + capabilities = [Country, Currency, Date, Email, IP, ISBN, Money, Phone, URL] + for capability in capabilities: + for grammar in capability().get_grammars(): + assert isinstance(grammar.semantics, str) + assert grammar.semantics != "" + assert ( + grammar.semantics == grammar.name + or grammar.semantics in _COALESCED_SEMANTICS + ) + + @pytest.mark.unit + def test_renamed_singletons_pin_exact_semantics_ids(self) -> None: + """The renamed singleton grammars pin their exact ``semantics`` ids. + + The allowlist above would accept a cross-swap (e.g. ``us_recognition`` + declaring ``european_calendar_date``), and with dual-target date rules + such a swap is behaviorally inert today — but the moment any rule + targets a single id, a wrong declaration silently mis-canonicalizes + US/EU dates. Pin the name→id mapping explicitly (ADR-0003 + consistency-guard rationale). + """ + by_name = {grammar.name: grammar.semantics for grammar in Date().get_grammars()} + assert by_name["us_recognition"] == "us_calendar_date" + assert by_name["european_recognition"] == "european_calendar_date" + + +class TestGrammarSemanticsEnforcement: + @pytest.mark.unit + def test_bare_grammar_subclass_raises_type_error(self) -> None: + """A Grammar subclass missing ``semantics`` fails at class-definition time.""" + + with pytest.raises(TypeError, match="must define Grammar metadata"): + + class _BareGrammar(Grammar[Any]): + name = "bare_grammar" + + def recognize(self, text: str) -> list[RecognitionMatch[Any]]: + return [] + + @pytest.mark.unit + def test_missing_semantics_raises(self) -> None: + """The missing-``semantics`` error message names the attribute.""" + + with pytest.raises(TypeError, match="semantics"): + + class _MissingSemanticsGrammar(Grammar[Any]): + name = "missing_semantics_grammar" + + def recognize(self, text: str) -> list[RecognitionMatch[Any]]: + return [] + + @pytest.mark.unit + def test_empty_semantics_raises(self) -> None: + """An empty ``semantics`` string is valid type-wise but a bug: + the grammar would carry no semantics. The runtime guard must reject it.""" + with pytest.raises(TypeError, match="non-empty"): + + class _EmptySemanticsGrammar(Grammar[Any]): + name = "empty_semantics_grammar" + semantics = "" + + def recognize(self, text: str) -> list[RecognitionMatch[Any]]: + return [] + + @pytest.mark.unit + @pytest.mark.parametrize("value", [42, frozenset()]) + def test_semantics_must_be_str(self, value: object) -> None: + """A non-str ``semantics`` value fails during Grammar subclass creation.""" + namespace: dict[str, Any] = { + "name": "test_grammar", + "recognize": lambda self, text: [], + "semantics": value, + } + + with pytest.raises(TypeError, match="semantics must be str"): + type("_InvalidSemantics", (Grammar,), namespace) + + @pytest.mark.unit + def test_inherited_semantics_satisfies_enforcement(self) -> None: + """A subclass inheriting ``semantics`` needs no own declaration. + + Locks the ``vars(cls).get`` fallback: the resolved value is read from + the class namespace first, then from the MRO. + """ + + class _CompliantGrammar(Grammar[Any]): + name = "compliant_grammar" + semantics = "compliant_grammar" + + def recognize(self, text: str) -> list[RecognitionMatch[Any]]: + return [] + + class _InheritedSemanticsGrammar(_CompliantGrammar): + pass + + assert _InheritedSemanticsGrammar.semantics == "compliant_grammar" diff --git a/tests/unit/test_rule_metadata.py b/tests/unit/test_rule_metadata.py index 9feb8568..8d4dced4 100644 --- a/tests/unit/test_rule_metadata.py +++ b/tests/unit/test_rule_metadata.py @@ -17,7 +17,7 @@ "strategy", "provenance", "citation", - "target_grammars", + "target_semantics", "requires_features", ) @@ -93,8 +93,8 @@ class _IncompleteRule(Rule[str]): provenance = _TEST_PROVENANCE if missing != "citation": citation = "test citation" - if missing != "target_grammars": - target_grammars = frozenset({"test_grammar"}) + if missing != "target_semantics": + target_semantics = frozenset({"test_grammar"}) if missing != "requires_features": requires_features = frozenset() @@ -105,17 +105,17 @@ def normalize(self, notation: str, contract: Contract) -> str: return "" @pytest.mark.unit - def test_empty_target_grammars_raises(self) -> None: - """An empty target_grammars frozenset is valid type-wise but a bug: + def test_empty_target_semantics_raises(self) -> None: + """An empty target_semantics frozenset is valid type-wise but a bug: the rule would match nothing. The runtime guard must reject it.""" with pytest.raises(TypeError, match="non-empty"): - class _EmptyTargetGrammars(Rule[str]): + class _EmptyTargetSemantics(Rule[str]): name = "test_rule" strategy = RuleStrategy.REGEX provenance = _TEST_PROVENANCE citation = "test citation" - target_grammars = frozenset() + target_semantics = frozenset() requires_features = frozenset() def matches(self, notation: str, contract: Contract) -> bool: @@ -128,8 +128,8 @@ def normalize(self, notation: str, contract: Contract) -> str: @pytest.mark.parametrize( ("attribute", "value"), [ - ("target_grammars", "test_grammar"), - ("target_grammars", ["test_grammar"]), + ("target_semantics", "test_grammar"), + ("target_semantics", ["test_grammar"]), ("requires_features", frozenset({1})), ], ) @@ -142,7 +142,7 @@ def test_affinity_metadata_requires_frozenset_of_strings( "strategy": RuleStrategy.REGEX, "provenance": _TEST_PROVENANCE, "citation": "test citation", - "target_grammars": frozenset({"test_grammar"}), + "target_semantics": frozenset({"test_grammar"}), "requires_features": frozenset(), "matches": lambda self, notation, contract: True, "normalize": lambda self, notation, contract: "",