From ebf21bb0b61a892811fd689a3fdd175bae640376 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 16:56:36 +0800 Subject: [PATCH 01/22] feat(core): declare semantics on all shipped grammars --- .../Country/grammar/alpha2_recognition.py | 1 + .../Country/grammar/alpha3_recognition.py | 1 + .../Country/grammar/name_recognition.py | 1 + .../Country/grammar/numeric_recognition.py | 1 + .../Currency/grammar/code_recognition.py | 1 + .../Currency/grammar/symbol_recognition.py | 1 + .../Currency/grammar/word_recognition.py | 1 + .../Date/grammar/european_recognition.py | 1 + .../Date/grammar/iso8601_recognition.py | 1 + .../Date/grammar/slash_iso_recognition.py | 1 + .../Date/grammar/us_recognition.py | 1 + .../Email/grammar/localhost_recognition.py | 1 + .../Email/grammar/obfuscated_recognition.py | 1 + .../Email/grammar/standard_recognition.py | 1 + .../IP/grammar/ipv4_recognition.py | 1 + .../IP/grammar/ipv6_recognition.py | 1 + .../ISBN/grammar/isbn10_recognition.py | 1 + .../ISBN/grammar/isbn13_recognition.py | 1 + .../Money/grammar/code_recognition.py | 1 + .../Money/grammar/symbol_recognition.py | 1 + .../Money/grammar/word_recognition.py | 1 + .../Phone/grammar/e164_recognition.py | 1 + .../grammar/international_00_recognition.py | 1 + .../Phone/grammar/national_recognition.py | 1 + .../Phone/grammar/tel_uri_recognition.py | 1 + .../URL/grammar/absolute_uri_recognition.py | 1 + tests/unit/test_grammar_semantics_metadata.py | 29 +++++++++++++++++++ 27 files changed, 55 insertions(+) create mode 100644 tests/unit/test_grammar_semantics_metadata.py diff --git a/paxman/capabilities/Country/grammar/alpha2_recognition.py b/paxman/capabilities/Country/grammar/alpha2_recognition.py index c1316472..5a45b1fb 100644 --- a/paxman/capabilities/Country/grammar/alpha2_recognition.py +++ b/paxman/capabilities/Country/grammar/alpha2_recognition.py @@ -18,6 +18,7 @@ class Alpha2Grammar(Grammar[CountryNotation]): """ name = "alpha2_recognition" + semantics = "alpha2_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CountryNotation]]: """Extract alpha-2 patterns from text. diff --git a/paxman/capabilities/Country/grammar/alpha3_recognition.py b/paxman/capabilities/Country/grammar/alpha3_recognition.py index 2805520d..06aa3b63 100644 --- a/paxman/capabilities/Country/grammar/alpha3_recognition.py +++ b/paxman/capabilities/Country/grammar/alpha3_recognition.py @@ -18,6 +18,7 @@ class Alpha3Grammar(Grammar[CountryNotation]): """ name = "alpha3_recognition" + semantics = "alpha3_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CountryNotation]]: """Extract alpha-3 patterns from text. diff --git a/paxman/capabilities/Country/grammar/name_recognition.py b/paxman/capabilities/Country/grammar/name_recognition.py index ee979006..62640cb0 100644 --- a/paxman/capabilities/Country/grammar/name_recognition.py +++ b/paxman/capabilities/Country/grammar/name_recognition.py @@ -49,6 +49,7 @@ class NameGrammar(Grammar[CountryNotation]): """ name = "name_recognition" + semantics = "name_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CountryNotation]]: """Extract a country name representation from text. diff --git a/paxman/capabilities/Country/grammar/numeric_recognition.py b/paxman/capabilities/Country/grammar/numeric_recognition.py index 1d299bcf..bdd16ff1 100644 --- a/paxman/capabilities/Country/grammar/numeric_recognition.py +++ b/paxman/capabilities/Country/grammar/numeric_recognition.py @@ -18,6 +18,7 @@ class NumericGrammar(Grammar[CountryNotation]): """ name = "numeric_recognition" + semantics = "numeric_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CountryNotation]]: """Extract numeric patterns from text. diff --git a/paxman/capabilities/Currency/grammar/code_recognition.py b/paxman/capabilities/Currency/grammar/code_recognition.py index d6fd2fba..9c581fa3 100644 --- a/paxman/capabilities/Currency/grammar/code_recognition.py +++ b/paxman/capabilities/Currency/grammar/code_recognition.py @@ -33,6 +33,7 @@ class CodeRecognition(Grammar[CurrencyNotation]): """ name = "code_recognition" + semantics = "code_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CurrencyNotation]]: """Extract standalone 3-letter code tokens from text. diff --git a/paxman/capabilities/Currency/grammar/symbol_recognition.py b/paxman/capabilities/Currency/grammar/symbol_recognition.py index 2d9f4a63..7ef6aa7d 100644 --- a/paxman/capabilities/Currency/grammar/symbol_recognition.py +++ b/paxman/capabilities/Currency/grammar/symbol_recognition.py @@ -44,6 +44,7 @@ class SymbolRecognition(Grammar[CurrencyNotation]): """ name = "symbol_recognition" + semantics = "symbol_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CurrencyNotation]]: """Extract standalone symbol tokens from text. diff --git a/paxman/capabilities/Currency/grammar/word_recognition.py b/paxman/capabilities/Currency/grammar/word_recognition.py index c1330c2a..54d40978 100644 --- a/paxman/capabilities/Currency/grammar/word_recognition.py +++ b/paxman/capabilities/Currency/grammar/word_recognition.py @@ -38,6 +38,7 @@ class WordRecognition(Grammar[CurrencyNotation]): """ name = "word_recognition" + semantics = "word_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CurrencyNotation]]: """Extract standalone display-name word tokens from text. diff --git a/paxman/capabilities/Date/grammar/european_recognition.py b/paxman/capabilities/Date/grammar/european_recognition.py index 6e03c372..210f7a46 100644 --- a/paxman/capabilities/Date/grammar/european_recognition.py +++ b/paxman/capabilities/Date/grammar/european_recognition.py @@ -21,6 +21,7 @@ class EuropeanDateGrammar(Grammar[DateNotation]): """ name = "european_recognition" + semantics = "european_recognition" def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: """Extract European date patterns from text. diff --git a/paxman/capabilities/Date/grammar/iso8601_recognition.py b/paxman/capabilities/Date/grammar/iso8601_recognition.py index 55897c23..9ebe1d67 100644 --- a/paxman/capabilities/Date/grammar/iso8601_recognition.py +++ b/paxman/capabilities/Date/grammar/iso8601_recognition.py @@ -20,6 +20,7 @@ class ISO8601DateGrammar(Grammar[DateNotation]): """ name = "iso8601_recognition" + semantics = "iso8601_recognition" def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: """Extract ISO 8601 date patterns from text.""" diff --git a/paxman/capabilities/Date/grammar/slash_iso_recognition.py b/paxman/capabilities/Date/grammar/slash_iso_recognition.py index 3c9e407b..17be5453 100644 --- a/paxman/capabilities/Date/grammar/slash_iso_recognition.py +++ b/paxman/capabilities/Date/grammar/slash_iso_recognition.py @@ -26,6 +26,7 @@ class SlashISODateGrammar(Grammar[DateNotation]): """ name = "slash_iso_recognition" + semantics = "slash_iso_recognition" def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: """Extract YYYY/MM/DD date patterns from text.""" diff --git a/paxman/capabilities/Date/grammar/us_recognition.py b/paxman/capabilities/Date/grammar/us_recognition.py index 88d42409..7fbb1253 100644 --- a/paxman/capabilities/Date/grammar/us_recognition.py +++ b/paxman/capabilities/Date/grammar/us_recognition.py @@ -21,6 +21,7 @@ class USDateGrammar(Grammar[DateNotation]): """ name = "us_recognition" + semantics = "us_recognition" def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: """Extract US date patterns from text. diff --git a/paxman/capabilities/Email/grammar/localhost_recognition.py b/paxman/capabilities/Email/grammar/localhost_recognition.py index 41e9573c..a9dfca0b 100644 --- a/paxman/capabilities/Email/grammar/localhost_recognition.py +++ b/paxman/capabilities/Email/grammar/localhost_recognition.py @@ -17,6 +17,7 @@ class LocalhostEmailGrammar(Grammar[EmailNotation]): """Localhost email recognition: user@localhost.""" name = "localhost_recognition" + semantics = "localhost_recognition" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: matches: list[RecognitionMatch[EmailNotation]] = [] diff --git a/paxman/capabilities/Email/grammar/obfuscated_recognition.py b/paxman/capabilities/Email/grammar/obfuscated_recognition.py index 641df8e5..30f53064 100644 --- a/paxman/capabilities/Email/grammar/obfuscated_recognition.py +++ b/paxman/capabilities/Email/grammar/obfuscated_recognition.py @@ -21,6 +21,7 @@ class ObfuscatedEmailGrammar(Grammar[EmailNotation]): """Obfuscated email: 'user at domain dot tld' or 'user at domain.tld'.""" name = "obfuscated_recognition" + semantics = "obfuscated_recognition" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: """Extract obfuscated email patterns from text. diff --git a/paxman/capabilities/Email/grammar/standard_recognition.py b/paxman/capabilities/Email/grammar/standard_recognition.py index 92124bcd..cbb156f3 100644 --- a/paxman/capabilities/Email/grammar/standard_recognition.py +++ b/paxman/capabilities/Email/grammar/standard_recognition.py @@ -16,6 +16,7 @@ class StandardEmailGrammar(Grammar[EmailNotation]): """Standard email recognition: user@domain.tld.""" name = "standard_recognition" + semantics = "standard_recognition" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: matches: list[RecognitionMatch[EmailNotation]] = [] diff --git a/paxman/capabilities/IP/grammar/ipv4_recognition.py b/paxman/capabilities/IP/grammar/ipv4_recognition.py index 60d3eb68..ccec36dd 100644 --- a/paxman/capabilities/IP/grammar/ipv4_recognition.py +++ b/paxman/capabilities/IP/grammar/ipv4_recognition.py @@ -14,6 +14,7 @@ class IPv4Grammar(Grammar[IPNotation]): """IPv4 recognition: dotted-decimal format (e.g., 192.168.1.1).""" name = "ipv4_recognition" + semantics = "ipv4_recognition" def recognize(self, text: str) -> list[RecognitionMatch[IPNotation]]: """Extract IPv4 dotted-decimal patterns from text.""" diff --git a/paxman/capabilities/IP/grammar/ipv6_recognition.py b/paxman/capabilities/IP/grammar/ipv6_recognition.py index 107b4123..9774db17 100644 --- a/paxman/capabilities/IP/grammar/ipv6_recognition.py +++ b/paxman/capabilities/IP/grammar/ipv6_recognition.py @@ -49,6 +49,7 @@ class IPv6Grammar(Grammar[IPNotation]): """ name = "ipv6_recognition" + semantics = "ipv6_recognition" def recognize(self, text: str) -> list[RecognitionMatch[IPNotation]]: """Extract IPv6 address patterns from text. diff --git a/paxman/capabilities/ISBN/grammar/isbn10_recognition.py b/paxman/capabilities/ISBN/grammar/isbn10_recognition.py index 98f6d3f0..4c70a837 100644 --- a/paxman/capabilities/ISBN/grammar/isbn10_recognition.py +++ b/paxman/capabilities/ISBN/grammar/isbn10_recognition.py @@ -18,6 +18,7 @@ class ISBN10RecognitionGrammar(Grammar[ISBNNotation]): """ISBN-10 recognition: 10-digit ISBN with optional label and separators.""" name = "isbn10_recognition" + semantics = "isbn10_recognition" def recognize(self, text: str) -> list[RecognitionMatch[ISBNNotation]]: matches: list[RecognitionMatch[ISBNNotation]] = [] diff --git a/paxman/capabilities/ISBN/grammar/isbn13_recognition.py b/paxman/capabilities/ISBN/grammar/isbn13_recognition.py index fe99a566..75b2ff6a 100644 --- a/paxman/capabilities/ISBN/grammar/isbn13_recognition.py +++ b/paxman/capabilities/ISBN/grammar/isbn13_recognition.py @@ -17,6 +17,7 @@ class ISBN13RecognitionGrammar(Grammar[ISBNNotation]): """ISBN-13 recognition: 13-digit ISBN with optional label and separators.""" name = "isbn13_recognition" + semantics = "isbn13_recognition" def recognize(self, text: str) -> list[RecognitionMatch[ISBNNotation]]: matches: list[RecognitionMatch[ISBNNotation]] = [] diff --git a/paxman/capabilities/Money/grammar/code_recognition.py b/paxman/capabilities/Money/grammar/code_recognition.py index a586b708..dce92260 100644 --- a/paxman/capabilities/Money/grammar/code_recognition.py +++ b/paxman/capabilities/Money/grammar/code_recognition.py @@ -38,6 +38,7 @@ class CodeRecognition(Grammar[MoneyNotation]): """ name = "code_recognition" + semantics = "code_recognition" def recognize(self, text: str) -> list[RecognitionMatch[MoneyNotation]]: """Extract code+amount tokens from text. diff --git a/paxman/capabilities/Money/grammar/symbol_recognition.py b/paxman/capabilities/Money/grammar/symbol_recognition.py index b908627b..b6c6693b 100644 --- a/paxman/capabilities/Money/grammar/symbol_recognition.py +++ b/paxman/capabilities/Money/grammar/symbol_recognition.py @@ -53,6 +53,7 @@ class SymbolRecognition(Grammar[MoneyNotation]): """ name = "symbol_recognition" + semantics = "symbol_recognition" def recognize(self, text: str) -> list[RecognitionMatch[MoneyNotation]]: """Extract symbol+amount tokens from text. diff --git a/paxman/capabilities/Money/grammar/word_recognition.py b/paxman/capabilities/Money/grammar/word_recognition.py index 632688c6..89fa5d81 100644 --- a/paxman/capabilities/Money/grammar/word_recognition.py +++ b/paxman/capabilities/Money/grammar/word_recognition.py @@ -45,6 +45,7 @@ class WordRecognition(Grammar[MoneyNotation]): """ name = "word_recognition" + semantics = "word_recognition" def recognize(self, text: str) -> list[RecognitionMatch[MoneyNotation]]: """Extract word+amount tokens from text. diff --git a/paxman/capabilities/Phone/grammar/e164_recognition.py b/paxman/capabilities/Phone/grammar/e164_recognition.py index 19364257..f518d70c 100644 --- a/paxman/capabilities/Phone/grammar/e164_recognition.py +++ b/paxman/capabilities/Phone/grammar/e164_recognition.py @@ -57,6 +57,7 @@ class E164Grammar(Grammar[PhoneNotation]): """ name = "e164_recognition" + semantics = "e164_recognition" def recognize(self, text: str) -> list[RecognitionMatch[PhoneNotation]]: """Extract e164 patterns from text. diff --git a/paxman/capabilities/Phone/grammar/international_00_recognition.py b/paxman/capabilities/Phone/grammar/international_00_recognition.py index 5de15491..eb180fa9 100644 --- a/paxman/capabilities/Phone/grammar/international_00_recognition.py +++ b/paxman/capabilities/Phone/grammar/international_00_recognition.py @@ -37,6 +37,7 @@ class International00Grammar(Grammar[PhoneNotation]): """ name = "international_00_recognition" + semantics = "international_00_recognition" def recognize(self, text: str) -> list[RecognitionMatch[PhoneNotation]]: """Extract 00-prefixed international patterns from text. diff --git a/paxman/capabilities/Phone/grammar/national_recognition.py b/paxman/capabilities/Phone/grammar/national_recognition.py index 09c94015..1f41b412 100644 --- a/paxman/capabilities/Phone/grammar/national_recognition.py +++ b/paxman/capabilities/Phone/grammar/national_recognition.py @@ -47,6 +47,7 @@ class NationalGrammar(Grammar[PhoneNotation]): """ name = "national_recognition" + semantics = "national_recognition" def recognize(self, text: str) -> list[RecognitionMatch[PhoneNotation]]: """Extract national patterns from text. diff --git a/paxman/capabilities/Phone/grammar/tel_uri_recognition.py b/paxman/capabilities/Phone/grammar/tel_uri_recognition.py index d013712e..b8ede303 100644 --- a/paxman/capabilities/Phone/grammar/tel_uri_recognition.py +++ b/paxman/capabilities/Phone/grammar/tel_uri_recognition.py @@ -26,6 +26,7 @@ class TelUriGrammar(Grammar[PhoneNotation]): """ name = "tel_uri_recognition" + semantics = "tel_uri_recognition" def recognize(self, text: str) -> list[RecognitionMatch[PhoneNotation]]: """Extract tel: URI patterns from text. diff --git a/paxman/capabilities/URL/grammar/absolute_uri_recognition.py b/paxman/capabilities/URL/grammar/absolute_uri_recognition.py index 1f1b8961..528f75e7 100644 --- a/paxman/capabilities/URL/grammar/absolute_uri_recognition.py +++ b/paxman/capabilities/URL/grammar/absolute_uri_recognition.py @@ -32,6 +32,7 @@ class AbsoluteUriRecognition(Grammar[URLNotation]): """Absolute-URI recognition: extracts scheme-anchored URI spans.""" name = "absolute_uri_recognition" + semantics = "absolute_uri_recognition" def recognize(self, text: str) -> list[RecognitionMatch[URLNotation]]: """Extract absolute-URI spans from text. diff --git a/tests/unit/test_grammar_semantics_metadata.py b/tests/unit/test_grammar_semantics_metadata.py new file mode 100644 index 00000000..84c33f5c --- /dev/null +++ b/tests/unit/test_grammar_semantics_metadata.py @@ -0,0 +1,29 @@ +"""Tests for grammar ``semantics`` metadata on shipped grammars.""" + +from __future__ import annotations + +import pytest + +from paxman.capabilities import ( + IP, + ISBN, + URL, + Country, + Currency, + Date, + Email, + Money, + Phone, +) + + +class TestGrammarSemanticsMetadata: + @pytest.mark.unit + def test_shipped_grammars_declare_semantics_identity(self) -> None: + """Every shipped grammar declares ``semantics`` equal to its name.""" + capabilities = [Country, Currency, Date, Email, IP, ISBN, Money, Phone, URL] + for capability in capabilities: + for grammar in capability().get_grammars(): + assert isinstance(grammar.semantics, str) + assert grammar.semantics != "" + assert grammar.semantics == grammar.name From f8ac1f6e817d906fb9d7d0d5169cf68159aa48a0 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 17:13:26 +0800 Subject: [PATCH 02/22] refactor: rename target_grammars to target_semantics --- .../Country/rules/cldr_localized_ed2025.py | 2 +- .../Country/rules/iso_3166_ed2024.py | 8 ++++---- .../rules/iso_3166_historical_ed2020.py | 2 +- .../Currency/rules/cldr_currencies_ed2025.py | 4 ++-- .../Currency/rules/iso_4217_ed2015.py | 2 +- .../Date/rules/en_50160_ed2010.py | 2 +- .../Date/rules/iso_8601_ed2019.py | 2 +- .../Date/rules/us_federal_rules_ed2023.py | 2 +- .../Email/rules/rfc_5322_ed2008.py | 2 +- .../Email/rules/rfc_6761_ed2012.py | 2 +- .../capabilities/IP/rules/rfc_5952_ed2010.py | 2 +- .../capabilities/IP/rules/rfc_791_ed1981.py | 2 +- .../ISBN/rules/isbn_range_message_ed2026.py | 2 +- .../ISBN/rules/isbn_users_manual_ed2012.py | 2 +- .../ISBN/rules/iso_2108_ed2017.py | 4 ++-- .../Money/rules/cldr_currencies_ed2025.py | 4 ++-- .../Money/rules/iso_4217_ed2015.py | 2 +- .../capabilities/Phone/rules/e164_ed2010.py | 4 ++-- .../capabilities/Phone/rules/nanp_ed2024.py | 4 ++-- .../Phone/rules/rfc_3966_ed2004.py | 2 +- .../URL/rules/whatwg_url_standard.py | 2 +- paxman/core/domain.py | 10 +++++----- paxman/core/extensions.py | 2 +- paxman/engine/orchestrator.py | 12 +++++------ tests/capabilities/currency/test_rules.py | 12 +++++------ tests/capabilities/isbn/test_rules.py | 8 ++++---- tests/capabilities/money/test_rules.py | 12 +++++------ tests/capabilities/url/test_rule.py | 2 +- tests/integration/test_feature_gating.py | 6 +++--- tests/integration/test_format_value_seam.py | 4 ++-- tests/integration/test_grammar_extensions.py | 10 +++++----- tests/integration/test_pipeline.py | 10 +++++----- tests/integration/test_recognition_seam.py | 6 +++--- tests/unit/test_capability.py | 2 +- tests/unit/test_extensions.py | 4 ++-- tests/unit/test_rule_metadata.py | 20 +++++++++---------- 36 files changed, 89 insertions(+), 89 deletions(-) diff --git a/paxman/capabilities/Country/rules/cldr_localized_ed2025.py b/paxman/capabilities/Country/rules/cldr_localized_ed2025.py index 2bfcf303..5b1b977e 100644 --- a/paxman/capabilities/Country/rules/cldr_localized_ed2025.py +++ b/paxman/capabilities/Country/rules/cldr_localized_ed2025.py @@ -45,7 +45,7 @@ class SectionLocalizedNames(Rule[CountryNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "CLDR v45 localized country names" - target_grammars = frozenset({"name_recognition"}) + target_semantics = frozenset({"name_recognition"}) requires_features = frozenset({"include_localized"}) def matches(self, notation: CountryNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Country/rules/iso_3166_ed2024.py b/paxman/capabilities/Country/rules/iso_3166_ed2024.py index be382a9a..4edd79c5 100644 --- a/paxman/capabilities/Country/rules/iso_3166_ed2024.py +++ b/paxman/capabilities/Country/rules/iso_3166_ed2024.py @@ -58,7 +58,7 @@ class SectionAlpha2Codes(Rule[CountryNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "ISO 3166-1 alpha-2 codes" - target_grammars = frozenset({"alpha2_recognition"}) + target_semantics = frozenset({"alpha2_recognition"}) requires_features = frozenset() def matches(self, notation: CountryNotation, contract: Contract) -> bool: @@ -98,7 +98,7 @@ class SectionAlpha3Codes(Rule[CountryNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "ISO 3166-1 alpha-3 codes" - target_grammars = frozenset({"alpha3_recognition"}) + target_semantics = frozenset({"alpha3_recognition"}) requires_features = frozenset() def matches(self, notation: CountryNotation, contract: Contract) -> bool: @@ -138,7 +138,7 @@ class SectionNumericCodes(Rule[CountryNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "ISO 3166-1 numeric (M49) codes" - target_grammars = frozenset({"numeric_recognition"}) + target_semantics = frozenset({"numeric_recognition"}) requires_features = frozenset() def _normalize_key(self, value: str) -> str: @@ -186,7 +186,7 @@ class SectionNames(Rule[CountryNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "ISO 3166-1 official English short names" - target_grammars = frozenset({"name_recognition"}) + target_semantics = frozenset({"name_recognition"}) requires_features = frozenset() def matches(self, notation: CountryNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Country/rules/iso_3166_historical_ed2020.py b/paxman/capabilities/Country/rules/iso_3166_historical_ed2020.py index e6c929da..2a7e4b72 100644 --- a/paxman/capabilities/Country/rules/iso_3166_historical_ed2020.py +++ b/paxman/capabilities/Country/rules/iso_3166_historical_ed2020.py @@ -66,7 +66,7 @@ class SectionHistoricalNames(Rule[CountryNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "ISO 3166-3:2020 (formerly used names)" - target_grammars = frozenset( + target_semantics = frozenset( {"name_recognition", "alpha2_recognition", "numeric_recognition"} ) requires_features = frozenset({"include_historical"}) diff --git a/paxman/capabilities/Currency/rules/cldr_currencies_ed2025.py b/paxman/capabilities/Currency/rules/cldr_currencies_ed2025.py index 2f7c70ab..9d74a770 100644 --- a/paxman/capabilities/Currency/rules/cldr_currencies_ed2025.py +++ b/paxman/capabilities/Currency/rules/cldr_currencies_ed2025.py @@ -122,7 +122,7 @@ class SectionSymbols(Rule[CurrencyNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "CLDR v47 currency symbols" - target_grammars = frozenset({"symbol_recognition"}) + target_semantics = frozenset({"symbol_recognition"}) requires_features = frozenset() def matches(self, notation: CurrencyNotation, contract: Contract) -> bool: @@ -166,7 +166,7 @@ class SectionNames(Rule[CurrencyNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "CLDR v47 currency display names" - target_grammars = frozenset({"word_recognition"}) + target_semantics = frozenset({"word_recognition"}) requires_features = frozenset() def matches(self, notation: CurrencyNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Currency/rules/iso_4217_ed2015.py b/paxman/capabilities/Currency/rules/iso_4217_ed2015.py index dfbda55d..17edfac4 100644 --- a/paxman/capabilities/Currency/rules/iso_4217_ed2015.py +++ b/paxman/capabilities/Currency/rules/iso_4217_ed2015.py @@ -42,7 +42,7 @@ class SectionCode(Rule[CurrencyNotation]): "ISO 4217:2015 alpha-3 currency codes, as amended by the ISO 4217 " "Maintenance Agency amendment series (SIX List One, 2026-01-01)" ) - target_grammars = frozenset({"code_recognition"}) + target_semantics = frozenset({"code_recognition"}) requires_features = frozenset() def matches(self, notation: CurrencyNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Date/rules/en_50160_ed2010.py b/paxman/capabilities/Date/rules/en_50160_ed2010.py index c2e1d453..15cb29c4 100644 --- a/paxman/capabilities/Date/rules/en_50160_ed2010.py +++ b/paxman/capabilities/Date/rules/en_50160_ed2010.py @@ -30,7 +30,7 @@ class Section4DateFormat(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 4 (date format)" - target_grammars = frozenset({"us_recognition", "european_recognition"}) + target_semantics = frozenset({"us_recognition", "european_recognition"}) requires_features = frozenset() def _interpret_two_digit_year(self, year_str: str, contract: Contract) -> int: diff --git a/paxman/capabilities/Date/rules/iso_8601_ed2019.py b/paxman/capabilities/Date/rules/iso_8601_ed2019.py index 7f49a6b1..d15d32d6 100644 --- a/paxman/capabilities/Date/rules/iso_8601_ed2019.py +++ b/paxman/capabilities/Date/rules/iso_8601_ed2019.py @@ -33,7 +33,7 @@ class Section431CalendarDate(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 4.3.1 (calendar date)" - target_grammars = frozenset({"iso8601_recognition", "slash_iso_recognition"}) + target_semantics = frozenset({"iso8601_recognition", "slash_iso_recognition"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Date/rules/us_federal_rules_ed2023.py b/paxman/capabilities/Date/rules/us_federal_rules_ed2023.py index e0f221da..1d627ad8 100644 --- a/paxman/capabilities/Date/rules/us_federal_rules_ed2023.py +++ b/paxman/capabilities/Date/rules/us_federal_rules_ed2023.py @@ -30,7 +30,7 @@ class Section1DateFormat(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 1 (date format)" - target_grammars = frozenset({"us_recognition", "european_recognition"}) + target_semantics = frozenset({"us_recognition", "european_recognition"}) requires_features = frozenset() def _interpret_two_digit_year(self, year_str: str, contract: Contract) -> int: diff --git a/paxman/capabilities/Email/rules/rfc_5322_ed2008.py b/paxman/capabilities/Email/rules/rfc_5322_ed2008.py index f0ac331f..fc681110 100644 --- a/paxman/capabilities/Email/rules/rfc_5322_ed2008.py +++ b/paxman/capabilities/Email/rules/rfc_5322_ed2008.py @@ -33,7 +33,7 @@ class Section341AddrSpec(Rule[EmailNotation]): strategy = RuleStrategy.REGEX provenance = PUBLICATION citation = "Section 3.4.1 (addr-spec)" - target_grammars = frozenset({"standard_recognition", "obfuscated_recognition"}) + target_semantics = frozenset({"standard_recognition", "obfuscated_recognition"}) requires_features = frozenset() def matches(self, notation: EmailNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Email/rules/rfc_6761_ed2012.py b/paxman/capabilities/Email/rules/rfc_6761_ed2012.py index 50eaa594..dc55ed8f 100644 --- a/paxman/capabilities/Email/rules/rfc_6761_ed2012.py +++ b/paxman/capabilities/Email/rules/rfc_6761_ed2012.py @@ -33,7 +33,7 @@ class Section63localhost(Rule[EmailNotation]): strategy = RuleStrategy.REGEX provenance = PUBLICATION citation = "Section 6.3 (localhost)" - target_grammars = frozenset({"localhost_recognition"}) + target_semantics = frozenset({"localhost_recognition"}) requires_features = frozenset() def matches(self, notation: EmailNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/IP/rules/rfc_5952_ed2010.py b/paxman/capabilities/IP/rules/rfc_5952_ed2010.py index ed90da84..ec08b72d 100644 --- a/paxman/capabilities/IP/rules/rfc_5952_ed2010.py +++ b/paxman/capabilities/IP/rules/rfc_5952_ed2010.py @@ -31,7 +31,7 @@ class Section4IPv6TextRepresentation(Rule[IPNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 4 (IPv6 text representation)" - target_grammars = frozenset({"ipv6_recognition"}) + target_semantics = frozenset({"ipv6_recognition"}) requires_features = frozenset() def matches(self, notation: IPNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/IP/rules/rfc_791_ed1981.py b/paxman/capabilities/IP/rules/rfc_791_ed1981.py index 899730c9..7bf6412b 100644 --- a/paxman/capabilities/IP/rules/rfc_791_ed1981.py +++ b/paxman/capabilities/IP/rules/rfc_791_ed1981.py @@ -30,7 +30,7 @@ class Section3Dot2IPv4Address(Rule[IPNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 3.2 (internet addressing)" - target_grammars = frozenset({"ipv4_recognition"}) + target_semantics = frozenset({"ipv4_recognition"}) requires_features = frozenset() @staticmethod diff --git a/paxman/capabilities/ISBN/rules/isbn_range_message_ed2026.py b/paxman/capabilities/ISBN/rules/isbn_range_message_ed2026.py index 6fb60e58..00e0c919 100644 --- a/paxman/capabilities/ISBN/rules/isbn_range_message_ed2026.py +++ b/paxman/capabilities/ISBN/rules/isbn_range_message_ed2026.py @@ -34,7 +34,7 @@ class Section4RegistrantRange(Rule[ISBNNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "Section 4 (registrant range)" - target_grammars = frozenset({"isbn13_recognition", "isbn10_recognition"}) + target_semantics = frozenset({"isbn13_recognition", "isbn10_recognition"}) requires_features = frozenset({"include_range_validation"}) def matches(self, notation: ISBNNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/ISBN/rules/isbn_users_manual_ed2012.py b/paxman/capabilities/ISBN/rules/isbn_users_manual_ed2012.py index efa164bb..68326ae6 100644 --- a/paxman/capabilities/ISBN/rules/isbn_users_manual_ed2012.py +++ b/paxman/capabilities/ISBN/rules/isbn_users_manual_ed2012.py @@ -31,7 +31,7 @@ class Section6Isbn10CheckDigit(Rule[ISBNNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 6 (ISBN-10 check digit)" - target_grammars = frozenset({"isbn10_recognition"}) + target_semantics = frozenset({"isbn10_recognition"}) requires_features = frozenset() def matches(self, notation: ISBNNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/ISBN/rules/iso_2108_ed2017.py b/paxman/capabilities/ISBN/rules/iso_2108_ed2017.py index ad04e56d..457b5890 100644 --- a/paxman/capabilities/ISBN/rules/iso_2108_ed2017.py +++ b/paxman/capabilities/ISBN/rules/iso_2108_ed2017.py @@ -29,7 +29,7 @@ class Section53Isbn13CheckDigit(Rule[ISBNNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 5.3 (ISBN-13 check digit)" - target_grammars = frozenset({"isbn13_recognition"}) + target_semantics = frozenset({"isbn13_recognition"}) requires_features = frozenset() def matches(self, notation: ISBNNotation, contract: Contract) -> bool: @@ -52,7 +52,7 @@ class Section42Gs1Prefix(Rule[ISBNNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "Section 4.2 (GS1 prefix)" - target_grammars = frozenset({"isbn13_recognition"}) + target_semantics = frozenset({"isbn13_recognition"}) requires_features = frozenset() def matches(self, notation: ISBNNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Money/rules/cldr_currencies_ed2025.py b/paxman/capabilities/Money/rules/cldr_currencies_ed2025.py index 469756f9..74fdde34 100644 --- a/paxman/capabilities/Money/rules/cldr_currencies_ed2025.py +++ b/paxman/capabilities/Money/rules/cldr_currencies_ed2025.py @@ -133,7 +133,7 @@ class SectionSymbols(Rule[MoneyNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "CLDR v47 currency symbols" - target_grammars = frozenset({"symbol_recognition"}) + target_semantics = frozenset({"symbol_recognition"}) requires_features = frozenset() def matches(self, notation: MoneyNotation, contract: Contract) -> bool: @@ -196,7 +196,7 @@ class SectionNames(Rule[MoneyNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "CLDR v47 currency display names" - target_grammars = frozenset({"word_recognition"}) + target_semantics = frozenset({"word_recognition"}) requires_features = frozenset() def matches(self, notation: MoneyNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Money/rules/iso_4217_ed2015.py b/paxman/capabilities/Money/rules/iso_4217_ed2015.py index 7001de61..1ccec0ba 100644 --- a/paxman/capabilities/Money/rules/iso_4217_ed2015.py +++ b/paxman/capabilities/Money/rules/iso_4217_ed2015.py @@ -73,7 +73,7 @@ class SectionCode(Rule[MoneyNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "ISO 4217 currency codes" - target_grammars = frozenset({"code_recognition"}) + target_semantics = frozenset({"code_recognition"}) requires_features = frozenset() def matches(self, notation: MoneyNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Phone/rules/e164_ed2010.py b/paxman/capabilities/Phone/rules/e164_ed2010.py index 5d9c1f29..48334612 100644 --- a/paxman/capabilities/Phone/rules/e164_ed2010.py +++ b/paxman/capabilities/Phone/rules/e164_ed2010.py @@ -66,7 +66,7 @@ class Section6_1InternationalNumber(Rule[PhoneNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 6.1 (number structure)" - target_grammars = frozenset({"e164_recognition", "international_00_recognition"}) + target_semantics = frozenset({"e164_recognition", "international_00_recognition"}) requires_features = frozenset() def matches(self, notation: PhoneNotation, contract: Contract) -> bool: @@ -108,7 +108,7 @@ class Section6_2CountryCode(Rule[PhoneNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "Annex A (table of assigned country codes)" - target_grammars = frozenset({"e164_recognition", "international_00_recognition"}) + target_semantics = frozenset({"e164_recognition", "international_00_recognition"}) requires_features = frozenset() def matches(self, notation: PhoneNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Phone/rules/nanp_ed2024.py b/paxman/capabilities/Phone/rules/nanp_ed2024.py index cfd19957..0549f198 100644 --- a/paxman/capabilities/Phone/rules/nanp_ed2024.py +++ b/paxman/capabilities/Phone/rules/nanp_ed2024.py @@ -90,7 +90,7 @@ class Section1_1NANPStructure(Rule[PhoneNotation]): strategy = RuleStrategy.REGEX provenance = PUBLICATION citation = "NANP numbering plan structure (NPA NXX-XXXX)" - target_grammars = frozenset({"national_recognition"}) + target_semantics = frozenset({"national_recognition"}) requires_features = frozenset() def matches(self, notation: PhoneNotation, contract: Contract) -> bool: @@ -145,7 +145,7 @@ class Section1_2ServiceNPA(Rule[PhoneNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "NANPA service NPA assignment table" - target_grammars = frozenset({"national_recognition"}) + target_semantics = frozenset({"national_recognition"}) requires_features = frozenset() def matches(self, notation: PhoneNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Phone/rules/rfc_3966_ed2004.py b/paxman/capabilities/Phone/rules/rfc_3966_ed2004.py index cf082dcb..baacbef8 100644 --- a/paxman/capabilities/Phone/rules/rfc_3966_ed2004.py +++ b/paxman/capabilities/Phone/rules/rfc_3966_ed2004.py @@ -31,7 +31,7 @@ class Section3TelUri(Rule[PhoneNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 3 (tel URI) / Section 3.1 (global numbers)" - target_grammars = frozenset({"tel_uri_recognition"}) + target_semantics = frozenset({"tel_uri_recognition"}) requires_features = frozenset() def matches(self, notation: PhoneNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/URL/rules/whatwg_url_standard.py b/paxman/capabilities/URL/rules/whatwg_url_standard.py index 190c12f0..9959ea8e 100644 --- a/paxman/capabilities/URL/rules/whatwg_url_standard.py +++ b/paxman/capabilities/URL/rules/whatwg_url_standard.py @@ -40,7 +40,7 @@ class WhatwgUrlStandard(Rule[URLNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 4.4 (basic URL parser); RFC 3986 §3.1 / RFC 3987 §2 grammar" - target_grammars = frozenset({"absolute_uri_recognition"}) + target_semantics = frozenset({"absolute_uri_recognition"}) requires_features = frozenset() def matches(self, notation: URLNotation, contract: Contract) -> bool: diff --git a/paxman/core/domain.py b/paxman/core/domain.py index 9f18a6a8..cf4b0f47 100644 --- a/paxman/core/domain.py +++ b/paxman/core/domain.py @@ -188,7 +188,7 @@ class Rule(ABC, Generic[NotationT]): strategy: RuleStrategy provenance: Provenance citation: str - target_grammars: ClassVar[frozenset[str]] + target_semantics: ClassVar[frozenset[str]] requires_features: ClassVar[frozenset[str]] def __init_subclass__(cls, **kwargs: object) -> None: @@ -199,7 +199,7 @@ def __init_subclass__(cls, **kwargs: object) -> None: "strategy", "provenance", "citation", - "target_grammars", + "target_semantics", "requires_features", ) missing = [attr for attr in required if not hasattr(cls, attr)] @@ -207,7 +207,7 @@ def __init_subclass__(cls, **kwargs: object) -> None: raise TypeError( f"{cls.__name__} must define Rule metadata: {', '.join(missing)}" ) - for attribute in ("target_grammars", "requires_features"): + for attribute in ("target_semantics", "requires_features"): value: Any = vars(cls).get(attribute, getattr(cls, attribute)) if type(value) is not frozenset: raise TypeError(f"{cls.__name__}.{attribute} must be frozenset[str]") @@ -215,8 +215,8 @@ def __init_subclass__(cls, **kwargs: object) -> None: vars(cls).get(attribute, getattr(cls, attribute)) ): raise TypeError(f"{cls.__name__}.{attribute} must be frozenset[str]") - if not cls.target_grammars: - raise TypeError(f"{cls.__name__}.target_grammars must be non-empty") + if not cls.target_semantics: + raise TypeError(f"{cls.__name__}.target_semantics must be non-empty") @abstractmethod def matches(self, notation: NotationT, contract: Contract) -> bool: ... diff --git a/paxman/core/extensions.py b/paxman/core/extensions.py index 330d10de..39d68dfd 100644 --- a/paxman/core/extensions.py +++ b/paxman/core/extensions.py @@ -68,7 +68,7 @@ def register_rule(capability_name: str, rule: Any) -> None: The ``rule`` parameter is ``Any`` because this is a runtime validation entry point: untyped callers may pass non-classes or non-Rule classes and the isinstance guard below provides the safety net. Rule metadata - (``target_grammars``, ``requires_features``, ...) is enforced by + (``target_semantics``, ``requires_features``, ...) is enforced by ``Rule.__init_subclass__`` at class-definition time; this function validates the class type and name uniqueness only. diff --git a/paxman/engine/orchestrator.py b/paxman/engine/orchestrator.py index ffa81383..39622789 100644 --- a/paxman/engine/orchestrator.py +++ b/paxman/engine/orchestrator.py @@ -284,7 +284,7 @@ def _validate_affinity( """ known_grammars = {g.name for g in all_grammars} for rule in rules: - unknown = [g for g in rule.target_grammars if g not in known_grammars] + unknown = [g for g in rule.target_semantics if g not in known_grammars] if unknown: raise ContractError( f"Rule {rule.name!r} declares unknown grammar(s) " @@ -299,7 +299,7 @@ def _collect_candidates( ) -> list[Candidate]: """Match recognitions against rules and collect candidates. - Routes each recognition only to rules whose ``target_grammars`` includes the + Routes each recognition only to rules whose ``target_semantics`` includes the producing grammar's name (ARCHITECTURE.md:201), formats each validated value through the capability's ``format_value()`` seam, then dedups identical candidate tuples so the candidate multiset is stable regardless @@ -309,7 +309,7 @@ def _collect_candidates( for recognition in recognitions: grammar_name = recognition.grammar.grammar_name for rule in rules: - if grammar_name not in rule.target_grammars: + if grammar_name not in rule.target_semantics: continue try: if rule.matches(recognition.notation, recognition.contract): @@ -343,7 +343,7 @@ def _dedup_candidates(candidates: list[Candidate]) -> list[Candidate]: Provenance is deterministic per (rule, grammar) pair, so collapsing on this key preserves all information while keeping the candidate multiset stable - under any future over-declaration of ``target_grammars``. + under any future over-declaration of ``target_semantics``. """ seen: set[tuple[str, str, str]] = set() deduped: list[Candidate] = [] @@ -386,7 +386,7 @@ def _activated_rules( capability: Capability[Any], contract: Contract ) -> list[Rule[Any]]: """Community rules opt in like grammars: a rule runs only when the - contract names one of its ``target_grammars`` in ``extra_grammars``. + contract names one of its ``target_semantics`` in ``extra_grammars``. An un-opted community rule — even one targeting a shipped grammar — never affects results, keeping extension behavior deterministic per @@ -396,5 +396,5 @@ def _activated_rules( return [ rule for rule in get_extended_rules(capability.name) - if extra_grammars & rule.target_grammars + if extra_grammars & rule.target_semantics ] diff --git a/tests/capabilities/currency/test_rules.py b/tests/capabilities/currency/test_rules.py index fb037bae..34ca7049 100644 --- a/tests/capabilities/currency/test_rules.py +++ b/tests/capabilities/currency/test_rules.py @@ -97,9 +97,9 @@ def test_strategy(self) -> None: """Verify the rule strategy enum.""" assert self.rule.strategy == RuleStrategy.LOOKUP_TABLE - def test_target_grammars(self) -> None: + def test_target_semantics(self) -> None: """The code rule targets only the code grammar.""" - assert self.rule.target_grammars == frozenset({"code_recognition"}) + assert self.rule.target_semantics == frozenset({"code_recognition"}) def test_requires_features_empty(self) -> None: """The ISO rule never gates on contract features (always runs).""" @@ -236,9 +236,9 @@ def test_strategy(self) -> None: """Verify the rule strategy enum.""" assert self.rule.strategy == RuleStrategy.LOOKUP_TABLE - def test_target_grammars(self) -> None: + def test_target_semantics(self) -> None: """The symbol rule targets only the symbol grammar.""" - assert self.rule.target_grammars == frozenset({"symbol_recognition"}) + assert self.rule.target_semantics == frozenset({"symbol_recognition"}) def test_requires_features_empty(self) -> None: """Never gate on default_currency: a shared bare symbol yields @@ -306,9 +306,9 @@ def test_strategy(self) -> None: """Verify the rule strategy enum.""" assert self.rule.strategy == RuleStrategy.LOOKUP_TABLE - def test_target_grammars(self) -> None: + def test_target_semantics(self) -> None: """The word rule targets only the word grammar.""" - assert self.rule.target_grammars == frozenset({"word_recognition"}) + assert self.rule.target_semantics == frozenset({"word_recognition"}) def test_requires_features_empty(self) -> None: """The CLDR name rule never gates on contract features.""" diff --git a/tests/capabilities/isbn/test_rules.py b/tests/capabilities/isbn/test_rules.py index 85723c7b..8fc9c1b4 100644 --- a/tests/capabilities/isbn/test_rules.py +++ b/tests/capabilities/isbn/test_rules.py @@ -179,25 +179,25 @@ def test_rule_conventions(self) -> None: assert self.isbn13.name == "Section 5.3-isbn13-check-digit" assert self.isbn13.strategy == RuleStrategy.PARSER assert self.isbn13.citation == "Section 5.3 (ISBN-13 check digit)" - assert self.isbn13.target_grammars == frozenset({"isbn13_recognition"}) + assert self.isbn13.target_semantics == frozenset({"isbn13_recognition"}) assert self.isbn13.requires_features == frozenset() assert self.gs1.name == "Section 4.2-gs1-prefix" assert self.gs1.strategy == RuleStrategy.LOOKUP_TABLE assert self.gs1.citation == "Section 4.2 (GS1 prefix)" - assert self.gs1.target_grammars == frozenset({"isbn13_recognition"}) + assert self.gs1.target_semantics == frozenset({"isbn13_recognition"}) assert self.gs1.requires_features == frozenset() assert self.isbn10.name == "Section 6-isbn10-check-digit" assert self.isbn10.strategy == RuleStrategy.PARSER assert self.isbn10.citation == "Section 6 (ISBN-10 check digit)" - assert self.isbn10.target_grammars == frozenset({"isbn10_recognition"}) + assert self.isbn10.target_semantics == frozenset({"isbn10_recognition"}) assert self.isbn10.requires_features == frozenset() assert self.range.name == "Section 4-registrant-range" assert self.range.strategy == RuleStrategy.LOOKUP_TABLE assert self.range.citation == "Section 4 (registrant range)" - assert self.range.target_grammars == frozenset( + assert self.range.target_semantics == frozenset( {"isbn13_recognition", "isbn10_recognition"} ) diff --git a/tests/capabilities/money/test_rules.py b/tests/capabilities/money/test_rules.py index 666c27e5..3a439c71 100644 --- a/tests/capabilities/money/test_rules.py +++ b/tests/capabilities/money/test_rules.py @@ -142,9 +142,9 @@ def test_strategy(self) -> None: """Verify the rule strategy enum.""" assert self.rule.strategy == RuleStrategy.LOOKUP_TABLE - def test_target_grammars(self) -> None: + def test_target_semantics(self) -> None: """The code rule targets only the code grammar.""" - assert self.rule.target_grammars == frozenset({"code_recognition"}) + assert self.rule.target_semantics == frozenset({"code_recognition"}) def test_requires_features_empty(self) -> None: """The ISO rule never gates on contract features (always runs).""" @@ -260,9 +260,9 @@ def test_strategy(self) -> None: """Verify the rule strategy enum.""" assert self.rule.strategy == RuleStrategy.LOOKUP_TABLE - def test_target_grammars(self) -> None: + def test_target_semantics(self) -> None: """The symbol rule targets only the symbol grammar.""" - assert self.rule.target_grammars == frozenset({"symbol_recognition"}) + assert self.rule.target_semantics == frozenset({"symbol_recognition"}) def test_requires_features_empty(self) -> None: """Never gate on dollar_sign_currency: bare $ yields INVALID, not MISSING.""" @@ -375,9 +375,9 @@ def test_strategy(self) -> None: """Verify the rule strategy enum.""" assert self.rule.strategy == RuleStrategy.LOOKUP_TABLE - def test_target_grammars(self) -> None: + def test_target_semantics(self) -> None: """The word rule targets only the word grammar.""" - assert self.rule.target_grammars == frozenset({"word_recognition"}) + assert self.rule.target_semantics == frozenset({"word_recognition"}) def test_requires_features_empty(self) -> None: """The CLDR name rule never gates on contract features.""" diff --git a/tests/capabilities/url/test_rule.py b/tests/capabilities/url/test_rule.py index f2a5d9b6..c1dddb5b 100644 --- a/tests/capabilities/url/test_rule.py +++ b/tests/capabilities/url/test_rule.py @@ -44,7 +44,7 @@ def test_rule_metadata(self) -> None: """Rule metadata matches §1 (homogeneity contract, Money style).""" assert self.rule.name == "WHATWG URL Standard" assert self.rule.strategy == RuleStrategy.PARSER - assert self.rule.target_grammars == frozenset({"absolute_uri_recognition"}) + assert self.rule.target_semantics == frozenset({"absolute_uri_recognition"}) assert self.rule.requires_features == frozenset() assert ( self.rule.citation diff --git a/tests/integration/test_feature_gating.py b/tests/integration/test_feature_gating.py index 4a8d8b35..ff2fa690 100644 --- a/tests/integration/test_feature_gating.py +++ b/tests/integration/test_feature_gating.py @@ -55,7 +55,7 @@ class _NameRecognitionGrammar(Grammar[CountryNotation]): """Grammar emitting the CLDR localized key "Estados Unidos". Uses the real ``name_recognition`` grammar name so - ``SectionLocalizedNames``' declared ``target_grammars`` resolves, without + ``SectionLocalizedNames``' declared ``target_semantics`` resolves, without relying on the real ``NameGrammar``'s English/historical/Chinese lookup tables (localized recognition remediation is F3 scope, not F2). """ @@ -145,7 +145,7 @@ class _DanglingFeatureRule(Rule[CountryNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"name_recognition"}) + target_semantics = frozenset({"name_recognition"}) requires_features = frozenset({"not_a_contract_field"}) def matches(self, notation: CountryNotation, contract: object) -> bool: @@ -211,7 +211,7 @@ class _DanglingGrammarRule(Rule[CountryNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"missing_grammar"}) + target_semantics = frozenset({"missing_grammar"}) requires_features = frozenset() def matches(self, notation: CountryNotation, contract: object) -> bool: diff --git a/tests/integration/test_format_value_seam.py b/tests/integration/test_format_value_seam.py index c862bc06..de0d31bd 100644 --- a/tests/integration/test_format_value_seam.py +++ b/tests/integration/test_format_value_seam.py @@ -75,7 +75,7 @@ class _TokenRule(Rule[_TokenNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"token_grammar"}) + target_semantics = frozenset({"token_grammar"}) requires_features = frozenset() def matches(self, notation: _TokenNotation, contract: Contract) -> bool: @@ -177,7 +177,7 @@ class _DualTokenRule(Rule[_TokenNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"dual_token_grammar"}) + target_semantics = frozenset({"dual_token_grammar"}) requires_features = frozenset() def matches(self, notation: _TokenNotation, contract: Contract) -> bool: diff --git a/tests/integration/test_grammar_extensions.py b/tests/integration/test_grammar_extensions.py index 4a62873b..22fbc2f1 100644 --- a/tests/integration/test_grammar_extensions.py +++ b/tests/integration/test_grammar_extensions.py @@ -86,7 +86,7 @@ class DotDateRule(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "community test double" - target_grammars = frozenset({"dot_date_recognition"}) + target_semantics = frozenset({"dot_date_recognition"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: @@ -109,7 +109,7 @@ class SecondDateRule(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "community test double" - target_grammars = frozenset({"second_recognition"}) + target_semantics = frozenset({"second_recognition"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: @@ -132,7 +132,7 @@ class CommunityISO8601Rule(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "community test double" - target_grammars = frozenset({"iso8601_recognition"}) + target_semantics = frozenset({"iso8601_recognition"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: @@ -146,13 +146,13 @@ def normalize(self, notation: DateNotation, contract: Contract) -> str: class DanglingDateRule(Rule[DateNotation]): - """Community rule whose target_grammars references a missing grammar.""" + """Community rule whose target_semantics references a missing grammar.""" name = "dangling_date_rule" strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "community test double" - target_grammars = frozenset({"no_such_grammar"}) + target_semantics = frozenset({"no_such_grammar"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: diff --git a/tests/integration/test_pipeline.py b/tests/integration/test_pipeline.py index cf8815cb..bf3a6216 100644 --- a/tests/integration/test_pipeline.py +++ b/tests/integration/test_pipeline.py @@ -164,7 +164,7 @@ class StubRule(Rule[EmailNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"crash_grammar"}) + target_semantics = frozenset({"crash_grammar"}) requires_features = frozenset() def matches(self, notation: EmailNotation, contract: object) -> bool: @@ -189,7 +189,7 @@ class ExplodingRule(Rule[EmailNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"simple_grammar"}) + target_semantics = frozenset({"simple_grammar"}) requires_features = frozenset() def matches(self, notation: EmailNotation, contract: object) -> bool: @@ -371,7 +371,7 @@ def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: class _PhantomRule(Rule[EmailNotation]): - """Rule whose target_grammars names a non-existent grammar.""" + """Rule whose target_semantics names a non-existent grammar.""" name = "phantom_rule" strategy = RuleStrategy.REGEX @@ -385,7 +385,7 @@ class _PhantomRule(Rule[EmailNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"does_not_exist"}) + target_semantics = frozenset({"does_not_exist"}) requires_features = frozenset() def matches(self, notation: EmailNotation, contract: object) -> bool: @@ -437,7 +437,7 @@ def output_format(self) -> str | None: class TestGrammarRuleAffinity: - """F1: grammar→rule affinity declared via Rule.target_grammars.""" + """F1: grammar→rule affinity declared via Rule.target_semantics.""" @pytest.mark.integration @pytest.mark.parametrize("output_format", [None, "ISO", "US"]) diff --git a/tests/integration/test_recognition_seam.py b/tests/integration/test_recognition_seam.py index 045984df..e84c4849 100644 --- a/tests/integration/test_recognition_seam.py +++ b/tests/integration/test_recognition_seam.py @@ -98,7 +98,7 @@ class _LongRule(Rule[_ProbeNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"probe_long"}) + target_semantics = frozenset({"probe_long"}) requires_features = frozenset() def matches(self, notation: _ProbeNotation, contract: Contract) -> bool: @@ -123,7 +123,7 @@ class _ShortRule(Rule[_ProbeNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"probe_short"}) + target_semantics = frozenset({"probe_short"}) requires_features = frozenset() def matches(self, notation: _ProbeNotation, contract: Contract) -> bool: @@ -400,7 +400,7 @@ class _CommunityRule(Rule[_ProbeNotation]): publication_year=2024, ) citation = "test" - target_grammars = frozenset({"probe_community"}) + target_semantics = frozenset({"probe_community"}) requires_features = frozenset() def matches(self, notation: _ProbeNotation, contract: Contract) -> bool: diff --git a/tests/unit/test_capability.py b/tests/unit/test_capability.py index 08483154..27fb6390 100644 --- a/tests/unit/test_capability.py +++ b/tests/unit/test_capability.py @@ -35,7 +35,7 @@ class StubRule(Rule): publication_year=2024, ) citation: str = "test citation" - target_grammars = frozenset({"stub_grammar"}) + target_semantics = frozenset({"stub_grammar"}) requires_features = frozenset() def matches(self, notation: Notation, contract: Contract) -> bool: diff --git a/tests/unit/test_extensions.py b/tests/unit/test_extensions.py index d5d61aab..afbb18cb 100644 --- a/tests/unit/test_extensions.py +++ b/tests/unit/test_extensions.py @@ -82,7 +82,7 @@ class _DotDateRule(Rule[Any]): strategy = RuleStrategy.PARSER provenance = _PROVENANCE citation = "community test double" - target_grammars = frozenset({"dot_date_recognition"}) + target_semantics = frozenset({"dot_date_recognition"}) requires_features = frozenset() def matches(self, notation: Any, contract: Any) -> bool: @@ -99,7 +99,7 @@ class _NamelessRule(Rule[Any]): strategy = RuleStrategy.PARSER provenance = _PROVENANCE citation = "community test double" - target_grammars = frozenset({"x"}) + target_semantics = frozenset({"x"}) requires_features = frozenset() def matches(self, notation: Any, contract: Any) -> bool: diff --git a/tests/unit/test_rule_metadata.py b/tests/unit/test_rule_metadata.py index 9feb8568..8d4dced4 100644 --- a/tests/unit/test_rule_metadata.py +++ b/tests/unit/test_rule_metadata.py @@ -17,7 +17,7 @@ "strategy", "provenance", "citation", - "target_grammars", + "target_semantics", "requires_features", ) @@ -93,8 +93,8 @@ class _IncompleteRule(Rule[str]): provenance = _TEST_PROVENANCE if missing != "citation": citation = "test citation" - if missing != "target_grammars": - target_grammars = frozenset({"test_grammar"}) + if missing != "target_semantics": + target_semantics = frozenset({"test_grammar"}) if missing != "requires_features": requires_features = frozenset() @@ -105,17 +105,17 @@ def normalize(self, notation: str, contract: Contract) -> str: return "" @pytest.mark.unit - def test_empty_target_grammars_raises(self) -> None: - """An empty target_grammars frozenset is valid type-wise but a bug: + def test_empty_target_semantics_raises(self) -> None: + """An empty target_semantics frozenset is valid type-wise but a bug: the rule would match nothing. The runtime guard must reject it.""" with pytest.raises(TypeError, match="non-empty"): - class _EmptyTargetGrammars(Rule[str]): + class _EmptyTargetSemantics(Rule[str]): name = "test_rule" strategy = RuleStrategy.REGEX provenance = _TEST_PROVENANCE citation = "test citation" - target_grammars = frozenset() + target_semantics = frozenset() requires_features = frozenset() def matches(self, notation: str, contract: Contract) -> bool: @@ -128,8 +128,8 @@ def normalize(self, notation: str, contract: Contract) -> str: @pytest.mark.parametrize( ("attribute", "value"), [ - ("target_grammars", "test_grammar"), - ("target_grammars", ["test_grammar"]), + ("target_semantics", "test_grammar"), + ("target_semantics", ["test_grammar"]), ("requires_features", frozenset({1})), ], ) @@ -142,7 +142,7 @@ def test_affinity_metadata_requires_frozenset_of_strings( "strategy": RuleStrategy.REGEX, "provenance": _TEST_PROVENANCE, "citation": "test citation", - "target_grammars": frozenset({"test_grammar"}), + "target_semantics": frozenset({"test_grammar"}), "requires_features": frozenset(), "matches": lambda self, notation, contract: True, "normalize": lambda self, notation, contract: "", From 811c30f74574135ebde86a5c214d60ed120b24d8 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 19:27:55 +0800 Subject: [PATCH 03/22] docs(adr): add ADR-0003 semantic affinity routing and implementation plan --- docs/adr/0003-semantic-affinity-routing.md | 254 ++++++ ...08-11-adr0003-semantic-affinity-routing.md | 728 ++++++++++++++++++ 2 files changed, 982 insertions(+) create mode 100644 docs/adr/0003-semantic-affinity-routing.md create mode 100644 docs/superpowers/plans/2026-08-11-adr0003-semantic-affinity-routing.md diff --git a/docs/adr/0003-semantic-affinity-routing.md b/docs/adr/0003-semantic-affinity-routing.md new file mode 100644 index 00000000..d65afa72 --- /dev/null +++ b/docs/adr/0003-semantic-affinity-routing.md @@ -0,0 +1,254 @@ +# ADR-0003: Semantic Affinity — Route Rules by Meaning, Not Grammar Name + +## Status + +Accepted + +## Context + +Paxman separates Recognition from Validation: grammars produce span-bearing +`RecognitionMatch`es carrying a notation; rules validate that notation against +an authoritative specification and emit the canonical value with provenance. +The two layers are joined by **grammar-name affinity**: `Rule.target_grammars` +declares the grammar names whose output the rule understands, and the engine +routes a recognition only to rules that name its producing grammar. + +This affinity is enforced at three engine sites and one domain site: + +- `Rule.__init_subclass__` requires a non-empty `target_grammars` + `frozenset[str]` at class-definition time (`paxman/core/domain.py`). +- `_validate_affinity` fails fast with `ContractError` when a rule names a + grammar absent from the composed set (`paxman/engine/orchestrator.py`). +- `_collect_candidates` routes each recognition to rules via + `grammar_name not in rule.target_grammars`. +- `_activated_rules` opts a community rule in only when the contract names one + of its `target_grammars` in `extra_grammars`. + +The affinity exists to prevent the F1 defect (the cartesian product: every rule +validating every grammar's matches regardless of whether the rule's authority +covers that representation), which the homogeneity audit ranked as defect #1 +and which PR #19 fixed via `target_grammars` + the one-line orchestrator +filter. + +The name-based binding has a structural cost: **a new grammar is effective +only when a rule file is edited.** The shipped proof is the slash-ISO grammar +(`SlashISODateGrammar`, `YYYY/MM/DD`): it shares ISO 8601's position mapping +and canonical form, but activating it required extending the ISO rule's +`target_grammars` frozenset by hand. The grammar author's work was complete at +`RecognitionMatch`; the rule edit exists only to declare what the grammar +cannot declare today — *what its notation means*. + +This violates the responsibility boundary the grammar is otherwise held to. +Grammars are prohibited from importing rule-layer data and from assigning +canonical meaning; they are required to end at the span-bearing match. But the +one thing that would let a grammar stand alone — a statement of its own +semantics — has no home in the `Grammar` ABC, which carries only `name`. + +The same-notation-type hazard that motivates affinity is real and must +survive any redesign: `DateNotation(N1, N2, N3)` means `(year, month, day)` +from `iso8601_recognition` but `(month, day, year)` from `us_recognition`. +Affinity must remain *explicit*, it must remain fail-fast on dangling +declarations, and it must remain deterministic per contract. + +## Decision + +Replace grammar-name affinity with **semantic affinity**: a rule targets the +*meaning* of the notations it validates, and each grammar declares that +meaning. `Rule.target_grammars` is **replaced** by +`Rule.target_semantics`; `Grammar` gains a required `semantics` identifier. + +### 1. `Grammar.semantics` — the meaning claim + +`Grammar` (in `paxman/core/domain.py`) gains a required class attribute: + +```python +class Grammar(ABC, Generic[NotationT]): + """Base class for recognition grammars.""" + + name: str + semantics: ClassVar[str] +``` + +- `semantics` is a stable identifier for the *meaning* the grammar's notation + carries — what the notation says, not how it is written. +- It is enforced by a new `Grammar.__init_subclass__` mirroring `Rule`'s + metadata enforcement: non-empty `str`, required at class-definition time. +- It is a **claim**, not an interpretation: the grammar still never imports + rule-layer data, never validates, and never maps tokens to canonical values. + The semantic purity boundary is unchanged; `semantics` only names the + meaning the grammar already produces. +- Example: `iso8601_recognition` and `slash_iso_recognition` both declare + `semantics = "iso8601_calendar_date"`; `us_recognition` declares + `semantics = "us_calendar_date"`. + +### 2. `Rule.target_semantics` — replace `target_grammars` + +`Rule.target_grammars: ClassVar[frozenset[str]]` becomes +`Rule.target_semantics: ClassVar[frozenset[str]]` with identical enforcement +(non-empty `frozenset[str]`, checked by `Rule.__init_subclass__`). The ISO +rule's declaration collapses from two grammar names to one meaning: + +```python +# before +target_grammars = frozenset({"iso8601_recognition", "slash_iso_recognition"}) +# after +target_semantics = frozenset({"iso8601_calendar_date"}) +``` + +### 3. Engine routing — three one-line sites + +- `_validate_affinity`: validate `rule.target_semantics` against the composed + **semantics set** `{g.semantics for g in all_grammars}` instead of the + grammar-name set. Dangling semantics still fail fast with `ContractError`. +- `_collect_candidates`: route via + `recognition.grammar.semantics not in rule.target_semantics`. The engine + builds a `semantics_by_name` map at composition time so the producing + grammar's semantics is resolvable from the `RecognizedRep`'s + `GrammarRule.grammar_name`. +- `_activated_rules`: a community rule activates when the contract names, in + `extra_grammars`, any grammar whose `semantics` is in the rule's + `target_semantics` — the opt-in discipline is unchanged, keyed on meaning. + +### 4. Provenance and dedup stay name-based + +`Candidate.recognition_rule` continues to record the **grammar name** +(`grammar_name`), and `_dedup_candidates` continues to collapse on +`(value, recognition_rule, validation_rule)`. `semantics` is routing metadata; +the grammar name remains the audit identity of the recognition. Provenance +output is byte-identical to today for every existing input. + +### 5. What a grammar-only addition now means + +Adding a grammar whose meaning is already shipped becomes a single-file +change: the grammar file itself. It declares `semantics = ""` +and is validated by the existing rule automatically — no rule edit, no +`target_*` change. This is the property that makes the grammar's +responsibility end at the `RecognitionMatch` (plus its one-word meaning +claim). A genuinely new meaning still requires a new rule: the canonical +value and provenance must come from an authoritative specification, and that +is rule territory by design. + +## Migration + +1. **Phase 1 — behavior-preserving rename.** Every shipped grammar declares + `semantics = ""`; every rule's `target_grammars` is renamed + `target_semantics` with the identical set. Routing keyed on semantics with + `semantics == name` is byte-identical to name routing; the full pre-PR gate + (ruff, pyright, import-linter, pytest) stays green with no test edits. + Community rule metadata (`target_grammars` → `target_semantics`) is a + breaking rename at 0.x, acceptable under the same policy as ADR-0002. +2. **Phase 2 — coalescing.** Grammars that share meaning declare a common + `semantics` id (e.g. both Date ISO grammars → `"iso8601_calendar_date"`), + and the affected rules' `target_semantics` coalesce to the single id. Same + behavior, simpler declarations; each coalescing step is verified by the + per-capability pipeline tests. +3. **Consistency guard.** A new test asserts that every grammar declaring the + same `semantics` emits notations with identical field mapping and + canonicalization expectations — the F1-style protection against + same-semantics/different-meaning collisions. This is a test-time + guarantee, like the existing homogeneity tests; the engine does not + introspect notation semantics at runtime. +4. **Docs sweep.** `HOW_TO_ADD_NEW_GRAMMAR.md` (Step 4 becomes: declare a + shipped `semantics` and stop, or add a rule for a new meaning), + `HOW_TO_ADD_NEW_CAPABILITY.md`, `ARCHITECTURE.md`, `README.md` (Community + Extensions section), `CONTEXT.md`, the capability `AGENTS.md` conventions, + and `capability_homogeneity_audit.md`. Historical capability plans and + research docs are left as historical records (ADR-0002 precedent). + +## Consequences + +### Positive + +- **Grammar-only additions become effective for existing meanings.** The + slash-ISO case collapses from a two-file change (grammar + rule edit) to a + one-file change (grammar only). The grammar's responsibility now ends at + the `RecognitionMatch` plus its meaning claim. +- **Rules declare intent, not implementation.** `target_semantics = + {"us_calendar_date"}` says what the rule understands; recognition + implementations can be refined or added without rule churn. +- **The community seam widens.** A downstream user registers a grammar + (via `register_grammar` + `extra_grammars` opt-in) whose meaning is already + validated, and shipped rules validate it — no `register_rule` needed. + Recognition becomes the pluggable half of the seam. +- **F1 protection preserved.** Affinity stays explicit, deterministic, and + fail-fast on dangling declarations — the cartesian-product defect cannot + return via routing. +- **Provenance output unchanged.** `recognition_rule`/`validation_rule` + attribution and candidate dedup are untouched; existing consumers see no + behavioral difference in Phase 1. + +### Negative + +- **Breaking community API change.** Community rules must rename + `target_grammars` → `target_semantics`. Acceptable at 0.x; semver-major at + >= 1.0. +- **New required metadata on every grammar.** `semantics` is a contributor + touch point in `HOW_TO_ADD_NEW_GRAMMAR.md`; a grammar without it fails at + class-definition time rather than at runtime. +- **Wrong `semantics` declarations are silent until a test catches them.** A + grammar claiming a shipped `semantics` id with divergent field mapping + would route to the wrong rule and produce a wrong-but-plausible canonical + value. The consistency guard is the mitigation; it is test-time, not + runtime. + +### Risks + +- **Same-semantics, different-meaning collisions.** Two grammars declaring the + same `semantics` with divergent field mapping silently mis-canonicalize. + Mitigation: the consistency-guard test per `semantics` id (Migration #3), + and a documented convention that `semantics` is a *semantic contract*: all + grammars claiming it must agree on notation field order and canonical form. +- **Coalescing drift during Phase 2.** Coalescing `target_semantics` sets must + never widen the set of meanings a rule validates. Mitigation: each + coalescing step runs the per-capability pipeline tests; the migration lands + one capability at a time. +- **Incomplete docs sweep.** 55 files reference `target_grammars` (24 rule + files, 3 engine sites, 1 domain ABC, extension registries docstring, test + suites, and 6 documentation surfaces). Mitigation: this ADR lands before + code; the sweep is enumerated in Migration #4. + +## Alternatives Considered + +1. **Keep `target_grammars` (status quo).** Rejected — a new grammar is + effective only when a rule file is edited, even when the grammar's meaning + is already validated. The slash-ISO one-line rule extension is the proof; + the coupling is name-based, not meaning-based, so it forces rule churn for + pure recognition additions. +2. **Additive `target_semantics` alongside `target_grammars` (OR routing).** + Rejected — two routing dimensions, ambiguous precedence, and a rule could + silently widen its authority by naming a semantics id without removing + stale grammar names. Replacement keeps a single, auditable affinity + declaration. +3. **Type-only routing.** Rules accept any notation of their declared type. + Rejected — the same `DateNotation` type carries different meanings per + grammar (US vs ISO field mapping); this is the F1 defect's failure mode, + not a fix for it. +4. **`same_semantics_as` pointer on the grammar.** A grammar references an + existing grammar whose meaning it shares. Rejected — requires equivalence + classes with canonical representatives, and validation of a reference + graph (acyclicity, consistency); a self-contained `semantics` id names the + meaning directly and avoids the representative problem. +5. **Infer meaning from the notation type.** Rejected — impossible in the + general case; meaning is not recoverable from `DateNotation(N1, N2, N3)` + without knowing the producing grammar's field mapping. `semantics` is the + minimal explicit claim that makes the inference sound. + +## References + +- ADR-0001 — Key Design Decisions #1 (Separate Recognition from Validation), + #2 (Notation as Internal Contract) — reaffirmed; the grammar purity + boundary this ADR extends +- ADR-0002 — breaking-change-at-0.x policy precedent +- `docs/superpowers/plans/2026-08-03-f1-grammar-rule-affinity.md` — the F1 + defect and the `target_grammars` fix this ADR generalizes to semantics +- `capability_homogeneity_audit.md` — F1 cartesian-product defect (#1), the + audit this ADR builds on +- `paxman/core/domain.py` — `Rule.__init_subclass__` (`target_grammars` + enforcement, replaced), `Grammar` (gains `semantics`), `GrammarRule` +- `paxman/engine/orchestrator.py` — `_validate_affinity`, + `_collect_candidates`, `_activated_rules` (routing sites, changed) +- `paxman/core/extensions.py` — `register_grammar` / `register_rule` + community seam (activation keyed on semantics) +- `HOW_TO_ADD_NEW_GRAMMAR.md` — Step 4 (rule-affinity requirement, replaced + by the semantics declaration) +- `README.md` — Community Extensions section (activation rule reworded) diff --git a/docs/superpowers/plans/2026-08-11-adr0003-semantic-affinity-routing.md b/docs/superpowers/plans/2026-08-11-adr0003-semantic-affinity-routing.md new file mode 100644 index 00000000..a11a3ed3 --- /dev/null +++ b/docs/superpowers/plans/2026-08-11-adr0003-semantic-affinity-routing.md @@ -0,0 +1,728 @@ +# ADR-0003 Semantic Affinity Routing — Implementation Plan + +| **Title** | Route rules by meaning, not grammar name | +| **Date** | 2026-08-11 | +| **Status** | In progress — Tasks 1-2 landed, Tasks 3-10 pending | +| **Branch** | `refactor/semantic-affinity-routing` (commit per task) | +| **Authoritative spec** | `docs/adr/0003-semantic-affinity-routing.md` — where this plan and the ADR disagree, the ADR wins | +| **Supersedes** | `Rule.target_grammars` grammar-name affinity (F1 fix, PR #19) — the routing key becomes `semantics` | + +> **For agentic workers.** This plan is written to be executed by a worker +> agent one task at a time. Every task is TDD: **Step 1 RED** (write/adjust +> the failing test first), **Step 2 GREEN** (make it pass), then the scoped +> verify command and the commit. Do not skip steps, do not reorder tasks, do +> not "improve" the design — D-decisions are locked (§1). The full suite is +> only green after Task 4; the per-task verify commands are scoped so each +> task is independently green. Commit with the exact message given for each +> task. **Pure-mechanical tasks are exempt from a meaningful RED step** — +> Tasks 2, 4, and 9 say so explicitly and their instruction wins. Task 10 is +> a verify-only gate with **no commit**, so the exact-commit-message +> requirement does not apply to it. + +> **Progress — START HERE at Task 3.** +> +> | Task | Status | Commit | +> |------|--------|--------| +> | Task 1 — declare `semantics` on all shipped grammars | ✅ landed | `ebf21bb` | +> | Task 2 — rename `target_grammars` → `target_semantics` | ✅ landed | `f8ac1f6` | +> | Task 3 — enforce `Grammar.semantics` at class-definition time | ⬜ **pending** | — | +> | Task 4 — engine routes on semantics | ⬜ pending | — | +> | Tasks 5-7 — Phase 2 coalescing (Date, Email, Phone) | ⬜ pending | — | +> | Task 8 — consistency guard | ⬜ pending | — | +> | Task 9 — docs sweep | ⬜ pending | — | +> | Task 10 — final gate (no commit) | ⬜ pending | — | +> +> Tasks 1 and 2 are **done — do not re-execute them.** `Grammar` still lacks +> `semantics` and `__init_subclass__` (`paxman/core/domain.py:228-240`); the +> engine still routes on grammar names (`paxman/engine/orchestrator.py:287`, +> `:312`, `:399`) — those line numbers are current and verified. Task 3's +> inventory is verified: exactly 18 test-defined `Grammar` subclasses in 8 +> test files need `semantics = ""`, at the exact locations listed in +> Task 3. + +--- + +## §1 Cross-Part Contract + +### Goal + +Implement ADR-0003: replace grammar-name affinity with **semantic affinity**. +`Grammar` gains a required `semantics: ClassVar[str]` enforced by a new +`Grammar.__init_subclass__`; `Rule.target_grammars` is **replaced** by +`Rule.target_semantics` (identical enforcement); the engine routes on +semantics at all three sites (`_validate_affinity`, `_collect_candidates`, +`_activated_rules`). Phase 1 is a byte-identical rename (`semantics == name` +for every shipped grammar); Phase 2 coalesces same-meaning grammars in Date, +Email, and Phone; a consistency-guard test locks same-semantics/same-field- +mapping; docs are swept. Provenance and candidate dedup stay name-based — +output is byte-identical to today for every existing input. + +### D-Decisions (locked — do not revisit without a new ADR) + +- **D1 — Phase 1 identity: `semantics == name` for all 26 shipped + grammars.** Routing keyed on semantics with `semantics == name` is + set-equal to name routing, so behavior is byte-identical (ADR Migration + #1). The identity is locked by a test (Task 1), not by convention. +- **D2 — The rename is atomic.** `target_grammars` → `target_semantics` + lands in ONE commit across `domain.py` (Rule ABC), `orchestrator.py`, the + 21 rule files (29 declarations), `extensions.py` (docstring), and the 12 + test files (47 hits). Any split-brain state is an import-time `TypeError` + from `Rule.__init_subclass__` — the sweep must be atomic (ADR-0002 plan + trap #3 precedent). +- **D3 — Enforcement mirrors `Rule`'s.** `Grammar.__init_subclass__` + requires `semantics` as a non-empty `str`, checked at class-definition + time. The ABC annotates `semantics: ClassVar[str]`; subclasses declare + bare `semantics = "..."` (matching the existing `name = "..."` style). + Inherited values satisfy the check (use `vars(cls).get(attribute, + getattr(cls, attribute))` exactly like `Rule` at domain.py:211) — so + `_CountingLongGrammar(_ProbeLongGrammar)` in + `tests/integration/test_recognition_seam.py` needs no edit. +- **D4 — Test doubles updated in the same commit as enforcement.** Adding + `Grammar.__init_subclass__` import-fails every test-defined `Grammar` + subclass lacking `semantics` (18 classes in 8 files). They declare + `semantics == ` (Phase-1 identity) in the same commit as the + enforcement lands (Task 3). +- **D5 — Engine routing.** `_validate_affinity` validates + `rule.target_semantics` against the composed semantics set + `{g.semantics for g in all_grammars}`; `_collect_candidates` routes via a + `semantics_by_name: dict[str, str] = {g.name: g.semantics ...}` map built + at composition time (`recognition.grammar.grammar_name` → semantics → + membership in `rule.target_semantics`); `_activated_rules` activates a + community rule when any extra-named grammar's semantics is in the rule's + `target_semantics`. Provenance (`recognition_rule`/`validation_rule`) and + `_dedup_candidates` stay name-based, unchanged (ADR §4). +- **D6 — Phase 2 coalescing scope.** Coalesce exactly three groups, one + capability per task, each verified by the per-capability pipeline tests: + Date `iso8601_recognition` + `slash_iso_recognition` → + `"iso8601_calendar_date"` (ADR's worked example), Email + `standard_recognition` + `obfuscated_recognition` → `"rfc5322_addr_spec"`, + Phone `e164_recognition` + `international_00_recognition` → + `"e164_international"`. Within Date, `us_recognition` → + `"us_calendar_date"` and `european_recognition` → + `"european_calendar_date"` (the ADR's own id vocabulary); the two Date + rules targeting `{"us_recognition","european_recognition"}` become + `{"us_calendar_date","european_calendar_date"}` — a two-element set + becomes a two-id set, **no widening**. +- **D7 — No-coalesce set (locked, widening is the failure mode).** Date + US/European stay separate (divergent field mapping — the F1 hazard the + ADR exists to prevent). Country `iso_3166_historical_ed2020.py` keeps + three distinct ids (one rule, three shapes). **ISBN is NOT coalesced**: + `isbn13_recognition` and `isbn10_recognition` keep identity ids because + the check-digit authorities differ (ISO 2108 mod-10 vs Users' Manual + mod-11); collapsing them would widen `iso_2108_ed2017.py`'s authority to + ISBN-10 input — exactly the "never widen" drift ADR risk #2 forbids. The + range rule (`isbn_range_message_ed2026.py`) keeps both ids. All other + capabilities (Country, Currency, IP, Money, URL) keep identity ids — + their 1:1 grammar↔rule mapping means the identity id already names the + meaning; renaming is cosmetic churn outside the ADR's migration scope. +- **D8 — Consistency guard is test-time, generic, and per-task-extended.** + A new `tests/unit/test_grammar_semantics_consistency.py` groups every + shipped grammar class by its declared `semantics` and, for each + multi-member group, asserts identical notation field mapping + + canonicalization expectations over shared probe rows. Modeled on the + `test_grammar_semantic_purity.py` precedent. Written at Task 5 with Date + probe rows (its first real subject), extended with Email/Phone rows in the + same commits as those coalescings — the new group's lookup failing before + coalescing is each task's RED step. Singleton groups pass by construction. +- **D9 — Docs sweep.** 6 in-scope files at repo root (32 references) + the + 2 nested AGENTS.md under `paxman/` are swept (ADR Migration #4). Historical + records are excluded (ADR-0002 precedent): `docs/superpowers/plans/*`, + `docs/report/*`, `docs/research/*`, `docs/adr/*`. The zero-grep proof must + exclude generated dirs (`htmlcov/`, `.hypothesis/`, `.pytest_cache/`, + `.venv/`) or it fails on a dirty local checkout. +- **D10 — Ground truth over ADR claims.** The ADR's "55 files reference + `target_grammars`" (L205) is stale. Verified ground truth: **44 files** + (26 under `paxman/` — 24 `.py` + 2 nested AGENTS.md, 12 test files, 6 + repo-root doc files). Where the ADR and this plan disagree on + counts/paths, this plan's verified inventory wins. + +### Out of scope + +- No behavior change to recognition/validation/status semantics (Phase 1 is + byte-identical; Phase 2 coalesces declarations only). +- No rename of `GrammarRule.grammar_name`, `RecognizedRep`, `Candidate`, + or `_dedup_candidates` keys — provenance stays name-based (ADR §4). +- No edits to historical plans/research/ADR files (D9). +- No new runtime semantics introspection by the engine (ADR Migration #3 — + the consistency guard is test-time only). +- No semantic-id renaming outside the coalescing capabilities (D6/D7). + +--- + +## §2 Tasks + +### Task 1 — `feat(core): declare semantics on all shipped grammars` + +> ✅ **LANDED** — commit `ebf21bb` (2026-08-11). Do not re-execute; the +> progress banner in §2's header supersedes this task's steps. + +Phase 1 identity: every shipped grammar declares `semantics = ""`. No enforcement yet — these are inert attributes until Task 3. + +**Step 1 RED — new file `tests/unit/test_grammar_semantics_metadata.py`** +- Add `test_shipped_grammars_declare_semantics_identity`: for each of the + nine shipped capabilities, for each grammar returned by + `capability.get_grammars()`: assert `isinstance(grammar.semantics, str)`, + `grammar.semantics != ""`, and `grammar.semantics == grammar.name`. +- Mark with the `unit` marker. +- Run: `uv run pytest tests/unit/test_grammar_semantics_metadata.py -q` → + RED (`AttributeError: ... has no attribute 'semantics'`). + +**Step 2 GREEN — add `semantics = ""` to all 26 shipped grammar files** +(one bare class attribute each, placed after the `name` declaration, +mirroring the existing `name = "..."` style): + +| Capability | File | Grammar class | `semantics` value | +|------------|------|---------------|-------------------| +| Country | `grammar/alpha2_recognition.py` | `Alpha2Grammar` | `"alpha2_recognition"` | +| Country | `grammar/alpha3_recognition.py` | `Alpha3Grammar` | `"alpha3_recognition"` | +| Country | `grammar/numeric_recognition.py` | `NumericGrammar` | `"numeric_recognition"` | +| Country | `grammar/name_recognition.py` | `NameGrammar` | `"name_recognition"` | +| Currency | `grammar/code_recognition.py` | `CodeRecognition` | `"code_recognition"` | +| Currency | `grammar/symbol_recognition.py` | `SymbolRecognition` | `"symbol_recognition"` | +| Currency | `grammar/word_recognition.py` | `WordRecognition` | `"word_recognition"` | +| Date | `grammar/iso8601_recognition.py` | `ISO8601DateGrammar` | `"iso8601_recognition"` | +| Date | `grammar/us_recognition.py` | `USDateGrammar` | `"us_recognition"` | +| Date | `grammar/european_recognition.py` | `EuropeanDateGrammar` | `"european_recognition"` | +| Date | `grammar/slash_iso_recognition.py` | `SlashISODateGrammar` | `"slash_iso_recognition"` | +| Email | `grammar/standard_recognition.py` | `StandardEmailGrammar` | `"standard_recognition"` | +| Email | `grammar/obfuscated_recognition.py` | `ObfuscatedEmailGrammar` | `"obfuscated_recognition"` | +| Email | `grammar/localhost_recognition.py` | `LocalhostEmailGrammar` | `"localhost_recognition"` | +| IP | `grammar/ipv4_recognition.py` | `IPv4Grammar` | `"ipv4_recognition"` | +| IP | `grammar/ipv6_recognition.py` | `IPv6Grammar` | `"ipv6_recognition"` | +| ISBN | `grammar/isbn13_recognition.py` | `ISBN13RecognitionGrammar` | `"isbn13_recognition"` | +| ISBN | `grammar/isbn10_recognition.py` | `ISBN10RecognitionGrammar` | `"isbn10_recognition"` | +| Money | `grammar/code_recognition.py` | `CodeRecognition` | `"code_recognition"` | +| Money | `grammar/symbol_recognition.py` | `SymbolRecognition` | `"symbol_recognition"` | +| Money | `grammar/word_recognition.py` | `WordRecognition` | `"word_recognition"` | +| Phone | `grammar/e164_recognition.py` | `E164Grammar` | `"e164_recognition"` | +| Phone | `grammar/tel_uri_recognition.py` | `TelUriGrammar` | `"tel_uri_recognition"` | +| Phone | `grammar/international_00_recognition.py` | `International00Grammar` | `"international_00_recognition"` | +| Phone | `grammar/national_recognition.py` | `NationalGrammar` | `"national_recognition"` | +| URL | `grammar/absolute_uri_recognition.py` | `AbsoluteUriRecognition` | `"absolute_uri_recognition"` | + +All files are under `paxman/capabilities//`. + +**Verify** +```bash +uv run pytest tests/unit/test_grammar_semantics_metadata.py -q +uv run pytest -q +uv run ruff check paxman/capabilities/ tests/unit/test_grammar_semantics_metadata.py +``` + +**Commit** +``` +feat(core): declare semantics on all shipped grammars +``` + +--- + +### Task 2 — `refactor: rename target_grammars to target_semantics` + +> ✅ **LANDED** — commit `f8ac1f6` (2026-08-11). Do not re-execute; the +> progress banner in §2's header supersedes this task's steps. Zero +> `target_grammars` hits remain in `paxman/` or `tests/` (verified). + +The atomic rename (D2). Pure mechanical sweep — **no RED test**; the RED +state is the intermediate breakage demonstrated in Step 1 below, fixed by +Step 2 in the same commit. + +**Step 1 RED (demonstrate breakage — do not commit)** +- In `paxman/core/domain.py` only, rename the five `target_grammars` sites in + `Rule` (L191 annotation, L202 required tuple, L210 type-check loop, L218 + non-empty guard, L219 error message) to `target_semantics`. +- Run: `uv run pytest tests/unit/test_rule_metadata.py -q` → RED + (`TypeError: must define Rule metadata`) and + `uv run pyright paxman/capabilities/` → RED. This proves the sweep below + is mandatory and atomic. + +**Step 2 GREEN — sweep every remaining site in one commit** + +Source (`paxman/`): +| File | Sites | +|------|-------| +| `paxman/core/domain.py` | already renamed in Step 1 (5 sites) | +| `paxman/engine/orchestrator.py` | L287 (`_validate_affinity` read), L312 (`_collect_candidates` route), L399 (`_activated_rules` intersection), plus docstrings L302, L346, L389 | +| `paxman/core/extensions.py` | docstring L71 (`register_rule`) | + +Rule files — rename the attribute name in all 29 declarations (values +unchanged — Phase 1 identity): + +| File | Decl lines | Set value (unchanged) | +|------|-----------|-----------------------| +| `Country/rules/iso_3166_ed2024.py` | L61, L101, L141, L189 | `{"alpha2_recognition"}` / `{"alpha3_recognition"}` / `{"numeric_recognition"}` / `{"name_recognition"}` | +| `Country/rules/cldr_localized_ed2025.py` | L48 | `{"name_recognition"}` | +| `Country/rules/iso_3166_historical_ed2020.py` | L69-71 (multi-line) | `{"name_recognition","alpha2_recognition","numeric_recognition"}` | +| `Currency/rules/iso_4217_ed2015.py` | L45 | `{"code_recognition"}` | +| `Currency/rules/cldr_currencies_ed2025.py` | L125, L169 | `{"symbol_recognition"}` / `{"word_recognition"}` | +| `Date/rules/iso_8601_ed2019.py` | L36 | `{"iso8601_recognition","slash_iso_recognition"}` | +| `Date/rules/en_50160_ed2010.py` | L33 | `{"us_recognition","european_recognition"}` | +| `Date/rules/us_federal_rules_ed2023.py` | L33 | `{"us_recognition","european_recognition"}` | +| `Email/rules/rfc_5322_ed2008.py` | L36 | `{"standard_recognition","obfuscated_recognition"}` | +| `Email/rules/rfc_6761_ed2012.py` | L36 | `{"localhost_recognition"}` | +| `IP/rules/rfc_791_ed1981.py` | L33 | `{"ipv4_recognition"}` | +| `IP/rules/rfc_5952_ed2010.py` | L34 | `{"ipv6_recognition"}` | +| `ISBN/rules/iso_2108_ed2017.py` | L32, L55 | `{"isbn13_recognition"}` (both) | +| `ISBN/rules/isbn_users_manual_ed2012.py` | L34 | `{"isbn10_recognition"}` | +| `ISBN/rules/isbn_range_message_ed2026.py` | L37 | `{"isbn13_recognition","isbn10_recognition"}` | +| `Money/rules/iso_4217_ed2015.py` | L76 | `{"code_recognition"}` | +| `Money/rules/cldr_currencies_ed2025.py` | L136, L199 | `{"symbol_recognition"}` / `{"word_recognition"}` | +| `Phone/rules/rfc_3966_ed2004.py` | L34 | `{"tel_uri_recognition"}` | +| `Phone/rules/e164_ed2010.py` | L69, L111 | `{"e164_recognition","international_00_recognition"}` (both) | +| `Phone/rules/nanp_ed2024.py` | L93, L148 | `{"national_recognition"}` (both) | +| `URL/rules/whatwg_url_standard.py` | L43 | `{"absolute_uri_recognition"}` | + +All under `paxman/capabilities//`. + +Tests — rename the attribute everywhere (values unchanged). 12 files, 47 +hits: +| File | Sites | +|------|-------| +| `tests/unit/test_capability.py` | L38 `StubRule` | +| `tests/unit/test_extensions.py` | L85 `_DotDateRule`, L102 `_NamelessRule` | +| `tests/unit/test_rule_metadata.py` | `_RULE_METADATA_ATTRS` (L15-22, `"target_grammars"` at L20 → `"target_semantics"`), L97 conditional `if missing != "target_grammars":`, L118 class `_EmptyTargetGrammars` → `_EmptyTargetSemantics`, `match=` regexes embedding the attribute name (L85, L111 `"non-empty"`, L152 `"must be frozenset[str]"`) | +| `tests/integration/test_feature_gating.py` | L148 `_DanglingFeatureRule`, L214 `_DanglingGrammarRule`, docstring L58 | +| `tests/integration/test_recognition_seam.py` | L101 `_LongRule`, L126 `_ShortRule`, L403 `_CommunityRule` | +| `tests/integration/test_grammar_extensions.py` | L89 `DotDateRule`, L112 `SecondDateRule`, L135 `CommunityISO8601Rule`, L155 `DanglingDateRule`, docstring L149 | +| `tests/integration/test_format_value_seam.py` | L78 `_TokenRule`, L180 `_DualTokenRule` | +| `tests/integration/test_pipeline.py` | L167 `StubRule`, L192 `ExplodingRule`, L388 `_PhantomRule`, docstrings L374, L440 | +| `tests/capabilities/isbn/test_rules.py` | L182, L188, L194, L200-202 (`TestRuleConventions`) | +| `tests/capabilities/money/test_rules.py` | L145-147, L263-265, L378-380 + method names `test_target_grammars` → `test_target_semantics` | +| `tests/capabilities/currency/test_rules.py` | L100-102, L239-241, L309-311 + method names `test_target_grammars` → `test_target_semantics` | +| `tests/capabilities/url/test_rule.py` | L47 | + +Rule metadata values are grammar names today and remain grammar-name strings +in Phase 1 (D1/D2) — the value mapping is deferred to Phase 2 tasks. + +**Verify** +```bash +uv run pytest -q +uv run ruff check paxman/ tests/ +uv run pyright +``` +Also confirm the sweep is complete (zero hits inside `paxman/` and `tests/`): +```bash +grep -rn "target_grammars" paxman/ tests/ || echo "CLEAN" +``` + +**Commit** +``` +refactor: rename target_grammars to target_semantics +``` + +--- + +### Task 3 — `feat(core): enforce Grammar.semantics at class-definition time` + +The enforcement mirror (D3), landing together with the test-double sweep +(D4) — import-time failure otherwise. + +**Step 1 RED — extend `tests/unit/test_grammar_semantics_metadata.py`** +- Add `test_bare_grammar_subclass_raises_type_error`: a local + `class _BareGrammar(Grammar[Any])` with only `name` and `recognize()` + raises `TypeError` matching `"must define Grammar metadata"` (or the + exact message chosen in Step 2 — write the test to match it). +- Add `test_missing_semantics_raises`: parametrized — grammar with + everything but `semantics` → `TypeError` naming `semantics`. +- Add `test_empty_semantics_raises`: `semantics = ""` → `TypeError` + matching `"non-empty"`. +- Add `test_semantics_must_be_str`: `semantics = 42` / `semantics = + frozenset()` → `TypeError` matching `"semantics must be str"`. +- Add `test_inherited_semantics_satisfies_enforcement`: a subclass of a + compliant grammar (no own `semantics`) is accepted — locks the + `vars(cls).get` fallback (D3), covering `_CountingLongGrammar`. +- Run: `uv run pytest tests/unit/test_grammar_semantics_metadata.py -q` → + RED (no `__init_subclass__` yet). + +**Step 2 GREEN — one commit containing all three:** +1. `paxman/core/domain.py` — add to `Grammar` (L228-240): + - `semantics: ClassVar[str]` annotation (next to `name: str`). + - `__init_subclass__` mirroring `Rule`'s (L194-219): require `semantics` + via `hasattr`; type-check `vars(cls).get("semantics", + getattr(cls, "semantics"))` is `type(...) is str`; non-empty guard. + Error messages: `f"{cls.__name__} must define Grammar metadata: + semantics"`, `f"{cls.__name__}.semantics must be str"`, + `f"{cls.__name__}.semantics must be non-empty"`. + - `ClassVar` is already imported (`Rule` uses it at L191). +2. All 18 test-defined `Grammar` subclasses in 8 files get + `semantics = ""` (Phase-1 identity, D4): + `tests/unit/test_capability.py` L14 `StubGrammar`; `tests/unit/ + test_extensions.py` L42/51/60/69 (`_DotDateGrammar`, `_SecondGrammar`, + `_NamelessGrammar`, `_MixedCaseGrammar`); `tests/unit/test_discovery.py` + L61 `DotDateGrammar`; `tests/integration/test_feature_gating.py` L54 + `_NameRecognitionGrammar`; `tests/integration/test_grammar_extensions.py` + L35/54/73 (`DotDateGrammar`, `SecondDateGrammar`, `ClashingDateGrammar`); + `tests/integration/test_format_value_seam.py` L47/143 (`_TokenGrammar`, + `_DualTokenGrammar`); `tests/integration/test_pipeline.py` L127/136/357 + (`CrashGrammar`, `SimpleGrammar`, `_PhantomGrammar`); + `tests/integration/test_recognition_seam.py` L44/69/376 + (`_ProbeLongGrammar`, `_ProbeShortGrammar`, `_CommunityGrammar`). + Do NOT touch `_CountingLongGrammar` (L324 — inherits, D3) or the + lookalike test classes (`TestGrammarRule`, `TestGrammarDedup`, + `TestExtraGrammars` — not Grammar subclasses). +3. Ship the enforcement tests from Step 1. + +**Verify** +```bash +uv run pytest tests/unit/test_grammar_semantics_metadata.py -q +uv run pytest -q +uv run ruff check paxman/ tests/ +uv run pyright +``` + +**Commit** +``` +feat(core): enforce Grammar.semantics at class-definition time +``` + +--- + +### Task 4 — `refactor(engine): route on semantics, not grammar name` + +The three engine sites switch from name keys to semantics keys (D5). With +`semantics == name` (D1) the routing keys are set-equal, so this is +byte-identical — **no RED test is writable**; the full suite is the +regression net (a behavior change would surface as test failures). Do not +rename `GrammarRule.grammar_name`, `Candidate.recognition_rule`, or the +`_dedup_candidates` key — provenance stays name-based (ADR §4). + +**Step 1 GREEN — `paxman/engine/orchestrator.py`** +- `run_capability` (L54-89): after `all_grammars` is composed (L60-63), + build `semantics_by_name = {g.name: g.semantics for g in all_grammars}` + and pass it to `_validate_affinity`, `_collect_candidates`, and + `_activated_rules` (update signatures; `_activated_rules` currently takes + `(capability, contract)` at L385, `_collect_candidates` takes + `(capability, recognitions, rules)` at L295). +- `_validate_affinity` (L276-292): `known_grammars = {g.name ...}` → + `known_semantics = {g.semantics for g in all_grammars}`; iterate + `rule.target_semantics`; error message: "declares unknown semantics" + (keep the sorted-listing shape). +- `_collect_candidates` (L295-338): at L310-312, resolve the producing + grammar's semantics via the map — `semantics_by_name[grammar_name] not in + rule.target_semantics: continue`. Update the docstring (L302) to describe + semantic routing. +- `_activated_rules` (L385-400): replace `extra_grammars & + rule.target_grammars` with a semantics-keyed activation: + `extra_semantics = {semantics_by_name[n] for n in extra_grammars if n in + semantics_by_name}`; activate iff `extra_semantics & + rule.target_semantics`. Unknown extra names are silently skipped (existing + behavior preserved). +- Check no test asserts the OLD `_validate_affinity` error text verbatim + (grep `unknown grammar` in `tests/`); if one does, update it in this + commit to "unknown semantics". + +**Step 2 VERIFY (byte-identity proof)** +```bash +uv run pytest -q +uv run ruff check paxman/ tests/ +uv run pyright +``` +All green = routing keyed on semantics is behavior-identical (D1). + +**Commit** +``` +refactor(engine): route on semantics, not grammar name +``` + +--- + +### Task 5 — `refactor(capabilities): coalesce Date grammars to calendar-date semantics` + +First Phase 2 coalescing (ADR Migration #2, the ADR's own worked example). +Two RED drivers: the consistency-guard test's group lookup, and +`CommunityISO8601Rule`'s dangling semantics. + +**Step 1 RED** +- Create `tests/unit/test_grammar_semantics_consistency.py` (D8): a generic + guard that enumerates all shipped grammar classes, groups them by + `semantics`, and for each multi-member group runs shared probe rows + through each member's `recognize()` asserting identical notation fields + + canonicalization expectations. Seed it with the Date group's probe row: + `iso8601` input `"2026-01-15"` and `slash_iso` input `"2026/01/15"` both + produce `DateNotation(N1="2026", N2="01", N3="15")`, both canonicalize to + `"2026-01-15"` through the ISO rule. (Verify the exact `DateNotation` + field names from `paxman/capabilities/Date/notation.py` while writing.) +- In `tests/integration/test_grammar_extensions.py`, update + `CommunityISO8601Rule.target_semantics` (L135) from + `frozenset({"iso8601_recognition"})` to + `frozenset({"iso8601_calendar_date"})`. +- Run: `uv run pytest tests/unit/test_grammar_semantics_consistency.py + tests/integration/test_grammar_extensions.py -q` → RED: the guard test + cannot find a `"iso8601_calendar_date"` group (grammars still claim + identity ids) and `test_grammar_extensions.py` fails fast with + `ContractError` (dangling semantics — the coalesced id is not yet known). + +**Step 2 GREEN** +- Date grammars (4): `iso8601_recognition.py` and `slash_iso_recognition.py` + → `semantics = "iso8601_calendar_date"`; `us_recognition.py` → + `"us_calendar_date"`; `european_recognition.py` → `"european_calendar_date"`. +- Date rules (3): `iso_8601_ed2019.py` L36 → + `frozenset({"iso8601_calendar_date"})`; `en_50160_ed2010.py` L33 and + `us_federal_rules_ed2023.py` L33 → + `frozenset({"us_calendar_date", "european_calendar_date"})` (no widening, + D6). +- No other rule file changes (D7). + +**Verify** +```bash +uv run pytest tests/capabilities/date tests/integration/test_grammar_extensions.py \ + tests/unit/test_grammar_semantics_consistency.py -q +uv run pytest tests/integration -q +uv run ruff check paxman/capabilities/Date/ tests/ +uv run pyright +``` + +**Commit** +``` +refactor(capabilities): coalesce Date grammars to calendar-date semantics +``` + +--- + +### Task 6 — `refactor(capabilities): coalesce Email grammars to addr-spec semantics` + +Second coalescing (D6). The guard test is extended in the same commit — +its group lookup is the RED driver. + +**Step 1 RED — extend `tests/unit/test_grammar_semantics_consistency.py`** +- Add the `"rfc5322_addr_spec"` group's probe rows: `standard` input (e.g. + `"user@example.com"`) and `obfuscated` input (e.g. + `"user at example dot com"`) produce identical `EmailNotation` fields and + both canonicalize to `"user@example.com"` through the RFC 5322 rule. + (Verify exact `EmailNotation` fields from + `paxman/capabilities/Email/notation.py` while writing.) +- Run: `uv run pytest tests/unit/test_grammar_semantics_consistency.py -q` + → RED (no `"rfc5322_addr_spec"` group exists yet). + +**Step 2 GREEN** +- Email grammars (2): `standard_recognition.py` and + `obfuscated_recognition.py` → `semantics = "rfc5322_addr_spec"`. +- Email rule (1): `rfc_5322_ed2008.py` L36 → + `frozenset({"rfc5322_addr_spec"})`. +- Unchanged: `localhost_recognition.py` (identity), `rfc_6761_ed2012.py` L36 + `{"localhost_recognition"}`. + +**Verify** +```bash +uv run pytest tests/capabilities/email tests/unit/test_grammar_semantics_consistency.py -q +uv run pytest tests/integration -q +uv run ruff check paxman/capabilities/Email/ tests/ +uv run pyright +``` + +**Commit** +``` +refactor(capabilities): coalesce Email grammars to addr-spec semantics +``` + +--- + +### Task 7 — `refactor(capabilities): coalesce Phone grammars to E.164 semantics` + +Third coalescing (D6). Same pattern as Task 6. + +**Step 1 RED — extend `tests/unit/test_grammar_semantics_consistency.py`** +- Add the `"e164_international"` group's probe rows: `e164` input (e.g. + `"+15551234567"`) and `international_00` input (e.g. `"0015551234567"`) + produce identical `PhoneNotation` fields and both canonicalize to + `"+15551234567"` through the E.164 rule. (Verify exact `PhoneNotation` + fields from `paxman/capabilities/Phone/notation.py` while writing.) +- Run: `uv run pytest tests/unit/test_grammar_semantics_consistency.py -q` + → RED (no `"e164_international"` group yet). + +**Step 2 GREEN** +- Phone grammars (2): `e164_recognition.py` and + `international_00_recognition.py` → `semantics = "e164_international"`. +- Phone rules (2 classes, 1 file): `e164_ed2010.py` L69 and L111 → + `frozenset({"e164_international"})`. +- Unchanged: `tel_uri_recognition.py` + `rfc_3966_ed2004.py` L34 + (`{"tel_uri_recognition"}`), `national_recognition.py` + `nanp_ed2024.py` + L93/L148 (`{"national_recognition"}`). + +**Verify** +```bash +uv run pytest tests/capabilities/phone tests/unit/test_grammar_semantics_consistency.py -q +uv run pytest tests/integration -q +uv run ruff check paxman/capabilities/Phone/ tests/ +uv run pyright +``` + +**Commit** +``` +refactor(capabilities): coalesce Phone grammars to E.164 semantics +``` + +--- + +### Task 8 — `test: lock same-semantics field-mapping consistency` + +The guard's structural half: assert the no-coalesce and non-coalesced groups +stay singleton (D6/D7) and that every multi-member group is covered by probe +rows. This makes the guard a complete F1-style test-time guarantee (ADR +Migration #3). + +**Step 1 GREEN (test-only — the groups already exist after Tasks 5-7)** +- Extend `tests/unit/test_grammar_semantics_consistency.py`: + - A structural enumeration: every shipped grammar belongs to a semantics + group; every multi-member group MUST have probe rows defined in the + test's table (fails if a future coalescing adds a group without rows). + - Explicit singleton assertions for the no-coalesce set (D7): exactly one + grammar claims each of `"us_calendar_date"`, `"european_calendar_date"`, + `"name_recognition"`, `"alpha2_recognition"`, `"alpha3_recognition"`, + `"numeric_recognition"`, `"isbn13_recognition"`, `"isbn10_recognition"`. +- Run: `uv run pytest tests/unit/test_grammar_semantics_consistency.py -q` + → GREEN (all groups consistent after Tasks 5-7). If RED, a coalescing + drifted — investigate, do not weaken the test. + +**Verify** +```bash +uv run pytest tests/unit/test_grammar_semantics_consistency.py -q +uv run pytest tests/unit -q +``` + +**Commit** +``` +test: lock same-semantics field-mapping consistency +``` + +--- + +### Task 9 — `docs: sweep target_grammars and document semantic affinity` + +Docs sweep (ADR Migration #4, D9). **No RED step** — pure documentation. + +**Step 1 GREEN — rewrite the 6 in-scope repo-root files + 2 nested AGENTS.md** + +| File | References to update | +|------|----------------------| +| `README.md` | L458 (Community Extensions sample: `DotDateGrammar` gains `semantics = "dot_date_recognition"`, `DotDateRule` `target_grammars` → `target_semantics`), L485 + L488 ("Rules of the seam" bullets — opt-in and fail-fast now keyed on semantics) | +| `ARCHITECTURE.md` | L100 (composition guard), L174 (community opt-in), L176 (fail-fast `ContractError`) — reword to semantics vocabulary | +| `CONTEXT.md` | L140, L155 (six-metadata-attrs sentence → `target_semantics` + grammar `semantics` claim), L182, L207 | +| `HOW_TO_ADD_NEW_CAPABILITY.md` | L292 (Step 5 directive), L441, L449, L453 (rule template `target_semantics: ClassVar[frozenset[str]]`), L572 (orthogonality note), L1035, L1056, L1069 (checklist) | +| `HOW_TO_ADD_NEW_GRAMMAR.md` | L21, L23, L33, L174, L178, L192 (extended example), L200, L202, L287 (validation table) — **Step 4 is rewritten per ADR**: a grammar whose meaning is already shipped declares a shipped `semantics` id and stops (no rule edit); a genuinely new meaning requires a new rule | +| `capability_homogeneity_audit.md` | L63, L65 (proposed orchestrator line → semantics keyed), L233, L296, L307 (F1 addenda) | +| `paxman/core/AGENTS.md` | L25 (Rule six-attr list → `target_semantics`) + add the Grammar `semantics` convention | +| `paxman/capabilities/AGENTS.md` | L43 (rule-file conventions → `target_semantics`) + grammar `semantics` requirement | + +Note: `HOW_TO_ADD_NEW_GRAMMAR.md` and `HOW_TO_ADD_NEW_CAPABILITY.md` live at +the **repo root**, not under `docs/`. + +Do NOT touch (historical records, ADR-0002 precedent): `docs/superpowers/ +plans/*`, `docs/report/*`, `docs/research/*`, `docs/adr/*`. + +**Verify** (zero hits outside the excluded paths — generated dirs excluded, +verified command): +```bash +grep -rnE 'target_grammars' . \ + --exclude-dir=.git --exclude-dir=plans --exclude-dir=report \ + --exclude-dir=research --exclude-dir=adr --exclude-dir=paxman \ + --exclude-dir=tests --exclude-dir=htmlcov --exclude-dir=.hypothesis \ + --exclude-dir=.pytest_cache --exclude-dir=.venv || echo "CLEAN" +``` +(`--exclude-dir=paxman --exclude-dir=tests` are dropped if this task is +assigned the Task-2 sweep proof instead — the nested AGENTS.md are the only +`paxman/` matches.) + +**Commit** +``` +docs: sweep target_grammars and document semantic affinity +``` + +--- + +### Task 10 — Final gate (no commit) + +**Verify — full pre-PR gate** (authoritative per `.github/workflows/ci.yml`): +```bash +uv run ruff check . && uv run ruff format --check . && uv run pyright && \ + uv run import-linter lint && uv run pytest +``` +Coverage gate (one include pattern per package — the brace shorthand +`paxman/{core,capabilities,engine,api}/*` is not expanded by the installed +coverage version and reports "No data to report"): +```bash +uv run coverage report --include="paxman/core/*" --fail-under=95 +uv run coverage report --include="paxman/capabilities/*" --fail-under=95 +uv run coverage report --include="paxman/engine/*" --fail-under=95 +uv run coverage report --include="paxman/api/*" --fail-under=95 +``` +Zero-grep proof (Task 9 Verify) returns CLEAN; `target_grammars` appears +nowhere in `paxman/` or `tests/`; `semantics` is present on all 26 shipped +grammars (Task 1 test still green). + +If any gate fails, fix it in a follow-up commit — never by weakening a test, +never by restoring `target_grammars`, never by editing the excluded +historical docs. + +--- + +## §3 Traps + +1. **Ordering is load-bearing.** Task 2 must include `domain.py` AND all 21 + rule files AND the 12 test files in one commit — any split is an + import-time `TypeError` (D2). Task 3 must include `domain.py` AND the 18 + test doubles in one commit — enforcement import-fails every grammar + subclass lacking `semantics` (D4). Task 5 must include the + `CommunityISO8601Rule` fixture update in the same commit as the Date + coalescing, or `tests/integration/test_grammar_extensions.py` fails fast + with `ContractError` (dangling `"iso8601_calendar_date"`). +2. **Never widen a rule's meaning set.** `target_semantics` coalescing is + set-collapse only: `{"iso8601_recognition","slash_iso_recognition"}` → + `{"iso8601_calendar_date"}` is fine; `{"us_recognition", + "european_recognition"}` → `{"us_calendar_date","european_calendar_date"}` + keeps two ids. ISBN is deliberately NOT coalesced (D7) — collapsing + isbn13/isbn10 would widen `iso_2108_ed2017.py` to ISBN-10 input. If a + coalescing looks like it needs a set to grow, it is wrong. +3. **Error-message coupling.** `tests/unit/test_rule_metadata.py` matches + `TypeError` text that embeds the attribute name (`match=missing`, L111 + `"non-empty"`, L152 `"must be frozenset[str]"`) and `domain.py` L219 bakes + `target_grammars` into the message — the Task 2 sweep must rename source + and test in the same commit. Same coupling applies to the new + `Grammar.__init_subclass__` messages (Task 3) — write the tests to the + exact messages you ship. +4. **ADR's "55 files" is stale.** Verified ground truth is 44 files + (D10). Use this plan's tables; the ADR's counts are for reference only. +5. **Generated artifacts trip the zero-grep proof.** `htmlcov/` (11 files), + `.hypothesis/`, `.pytest_cache/` contain stale `target_grammars` matches + and are gitignored — the Task 9/10 proof command excludes them (D9). +6. **`HOW_TO_ADD_*` guides are at the repo root**, not under `docs/` — path + mistakes in the sweep are silent no-ops. +7. **The three engine routing sites have no direct unit coverage** — + `_validate_affinity`, `_collect_candidates`, `_activated_rules` are + covered through integration fixtures (recognition seam, feature gating, + grammar extensions, pipeline). Do not "add coverage" by weakening those + fixtures; Task 4's byte-identity proof IS the integration suite. +8. **Provenance stays name-based.** Do not "helpfully" rename + `GrammarRule.grammar_name`, `Candidate.recognition_rule`, or the + `_dedup_candidates` key in Task 4 — ADR §4 pins them; changing them is + an output change outside this ADR. +9. **Property and e2e suites have zero `target_grammars` hits** — no edits + needed, but they instantiate shipped grammars, so Task 3's enforcement + applies to them at import. They are exercised at the Task 10 gate. +10. **`_CountingLongGrammar` needs no edit** (D3 — inheritance satisfies + the `vars(cls).get` fallback), and the lookalike classes + (`TestGrammarRule`, `TestGrammarDedup`, `TestExtraGrammars`) are test + classes, not Grammar subclasses — do not touch them. + +--- + +## §4 Definition of Done + +- [ ] All 26 shipped grammars declare `semantics` (identity in Phase 1; + coalesced ids for Date/Email/Phone after Phase 2), enforced by + `Grammar.__init_subclass__` at class-definition time with tests. +- [ ] Zero `target_grammars` anywhere in `paxman/` or `tests/`; the Task 9 + zero-grep proof is CLEAN outside the excluded historical paths. +- [ ] Engine routes on semantics at all three sites; provenance and + candidate dedup remain name-based (`GrammarRule.grammar_name`, + `Candidate.recognition_rule` unchanged). +- [ ] Phase 2 coalescing landed for Date/Email/Phone exactly as D6/D7 + scope; no rule's `target_semantics` set grew. +- [ ] `tests/unit/test_grammar_semantics_consistency.py` covers every + multi-member semantics group with probe rows and locks the singleton + no-coalesce set. +- [ ] Docs swept (Task 9 files); README's community example shows + `semantics` on the grammar and `target_semantics` on the rule. +- [ ] Full pre-PR gate green: `ruff check . && ruff format --check . && + pyright && import-linter lint && pytest` and 95% coverage per package. + From 0ea772761b9740ad92e75ce925e4d05dd6d4ecc4 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 19:40:14 +0800 Subject: [PATCH 04/22] feat(core): enforce Grammar.semantics at class-definition time --- paxman/core/domain.py | 12 +++ tests/integration/test_feature_gating.py | 1 + tests/integration/test_format_value_seam.py | 2 + tests/integration/test_grammar_extensions.py | 3 + tests/integration/test_pipeline.py | 3 + tests/integration/test_recognition_seam.py | 3 + tests/unit/test_capability.py | 1 + tests/unit/test_discovery.py | 1 + tests/unit/test_extensions.py | 4 + tests/unit/test_grammar_semantics_metadata.py | 75 +++++++++++++++++++ 10 files changed, 105 insertions(+) diff --git a/paxman/core/domain.py b/paxman/core/domain.py index cf4b0f47..d644074f 100644 --- a/paxman/core/domain.py +++ b/paxman/core/domain.py @@ -229,6 +229,18 @@ class Grammar(ABC, Generic[NotationT]): """Base class for recognition grammars.""" name: str + semantics: ClassVar[str] + + def __init_subclass__(cls, **kwargs: object) -> None: + """Enforce Grammar metadata at class-definition time.""" + super().__init_subclass__(**kwargs) + if not hasattr(cls, "semantics"): + raise TypeError(f"{cls.__name__} must define Grammar metadata: semantics") + semantics: Any = vars(cls).get("semantics", cls.semantics) + if type(semantics) is not str: + raise TypeError(f"{cls.__name__}.semantics must be str") + if not cls.semantics: + raise TypeError(f"{cls.__name__}.semantics must be non-empty") @abstractmethod def recognize(self, text: str) -> list[RecognitionMatch[NotationT]]: diff --git a/tests/integration/test_feature_gating.py b/tests/integration/test_feature_gating.py index ff2fa690..b3267948 100644 --- a/tests/integration/test_feature_gating.py +++ b/tests/integration/test_feature_gating.py @@ -61,6 +61,7 @@ class _NameRecognitionGrammar(Grammar[CountryNotation]): """ name = "name_recognition" + semantics = "name_recognition" def recognize(self, text: str) -> list[RecognitionMatch[CountryNotation]]: return [ diff --git a/tests/integration/test_format_value_seam.py b/tests/integration/test_format_value_seam.py index de0d31bd..96c72010 100644 --- a/tests/integration/test_format_value_seam.py +++ b/tests/integration/test_format_value_seam.py @@ -48,6 +48,7 @@ class _TokenGrammar(Grammar[_TokenNotation]): """Grammar that recognizes a single fixed token.""" name = "token_grammar" + semantics = "token_grammar" def recognize(self, text: str) -> list[RecognitionMatch[_TokenNotation]]: return [ @@ -144,6 +145,7 @@ class _DualTokenGrammar(Grammar[_TokenNotation]): """Grammar that recognizes two distinct tokens.""" name = "dual_token_grammar" + semantics = "dual_token_grammar" def recognize(self, text: str) -> list[RecognitionMatch[_TokenNotation]]: return [ diff --git a/tests/integration/test_grammar_extensions.py b/tests/integration/test_grammar_extensions.py index 22fbc2f1..e40e3f59 100644 --- a/tests/integration/test_grammar_extensions.py +++ b/tests/integration/test_grammar_extensions.py @@ -36,6 +36,7 @@ class DotDateGrammar(Grammar[DateNotation]): """Community test double: recognizes YYYY.MM.DD (dot separator).""" name = "dot_date_recognition" + semantics = "dot_date_recognition" _PATTERN = re.compile(r"\b(\d{4})\.(\d{2})\.(\d{2})\b") def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: @@ -55,6 +56,7 @@ class SecondDateGrammar(Grammar[DateNotation]): """Second community test double — same shape, different normalization.""" name = "second_recognition" + semantics = "second_recognition" _PATTERN = re.compile(r"\b(\d{4})\.(\d{2})\.(\d{2})\b") def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: @@ -74,6 +76,7 @@ class ClashingDateGrammar(Grammar[DateNotation]): """Community grammar whose name collides with a shipped Date grammar.""" name = "iso8601_recognition" + semantics = "iso8601_recognition" def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: return [] diff --git a/tests/integration/test_pipeline.py b/tests/integration/test_pipeline.py index bf3a6216..cf723610 100644 --- a/tests/integration/test_pipeline.py +++ b/tests/integration/test_pipeline.py @@ -128,6 +128,7 @@ class CrashGrammar(Grammar[EmailNotation]): """Grammar whose recognize() always raises.""" name = "crash_grammar" + semantics = "crash_grammar" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: raise RuntimeError("grammar crashed") @@ -137,6 +138,7 @@ class SimpleGrammar(Grammar[EmailNotation]): """Grammar that returns a fixed notation.""" name = "simple_grammar" + semantics = "simple_grammar" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: return [ @@ -358,6 +360,7 @@ class _PhantomGrammar(Grammar[EmailNotation]): """Grammar referenced by a rule that does not exist in the capability.""" name = "phantom_grammar" + semantics = "phantom_grammar" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: return [ diff --git a/tests/integration/test_recognition_seam.py b/tests/integration/test_recognition_seam.py index e84c4849..14b76edd 100644 --- a/tests/integration/test_recognition_seam.py +++ b/tests/integration/test_recognition_seam.py @@ -49,6 +49,7 @@ class _ProbeLongGrammar(Grammar[_ProbeNotation]): """ name = "probe_long" + semantics = "probe_long" _patterns = (re.compile(r"AAAA"), re.compile(r"AA")) def recognize(self, text: str) -> list[RecognitionMatch[_ProbeNotation]]: @@ -70,6 +71,7 @@ class _ProbeShortGrammar(Grammar[_ProbeNotation]): """Recognizes 'AA' only.""" name = "probe_short" + semantics = "probe_short" def recognize(self, text: str) -> list[RecognitionMatch[_ProbeNotation]]: return [ @@ -375,6 +377,7 @@ def test_fallback_never_activates_community_grammars(self) -> None: class _CommunityGrammar(Grammar[_ProbeNotation]): name = "probe_community" + semantics = "probe_community" def recognize(self, text: str) -> list[RecognitionMatch[_ProbeNotation]]: calls.append(self.name) diff --git a/tests/unit/test_capability.py b/tests/unit/test_capability.py index 27fb6390..2f4ddcb6 100644 --- a/tests/unit/test_capability.py +++ b/tests/unit/test_capability.py @@ -15,6 +15,7 @@ class StubGrammar(Grammar): """Minimal concrete grammar for testing Capability.""" name: str = "stub_grammar" + semantics = "stub_grammar" def recognize(self, text: str) -> list[Notation]: return [] diff --git a/tests/unit/test_discovery.py b/tests/unit/test_discovery.py index f54e7ad5..c316cadc 100644 --- a/tests/unit/test_discovery.py +++ b/tests/unit/test_discovery.py @@ -62,6 +62,7 @@ class DotDateGrammar(Grammar): """Minimal community grammar for extension delegation tests.""" name = "dot_date_recognition" + semantics = "dot_date_recognition" def recognize(self, text: str) -> list[RecognitionMatch]: return [] diff --git a/tests/unit/test_extensions.py b/tests/unit/test_extensions.py index afbb18cb..618e31ba 100644 --- a/tests/unit/test_extensions.py +++ b/tests/unit/test_extensions.py @@ -43,6 +43,7 @@ class _DotDateGrammar(Grammar[Any]): """Minimal community grammar test double.""" name = "dot_date_recognition" + semantics = "dot_date_recognition" def recognize(self, text: str) -> list[RecognitionMatch[Any]]: return [] @@ -52,6 +53,7 @@ class _SecondGrammar(Grammar[Any]): """Second grammar — tests registration order.""" name = "second_recognition" + semantics = "second_recognition" def recognize(self, text: str) -> list[RecognitionMatch[Any]]: return [] @@ -61,6 +63,7 @@ class _NamelessGrammar(Grammar[Any]): """Grammar with an empty name — tests name validation.""" name = "" + semantics = "nameless_grammar" def recognize(self, text: str) -> list[RecognitionMatch[Any]]: return [] @@ -70,6 +73,7 @@ class _MixedCaseGrammar(Grammar[Any]): """Grammar with a mixed-case name — tests lowercase enforcement.""" name = "DotDateRecognition" + semantics = "DotDateRecognition" def recognize(self, text: str) -> list[RecognitionMatch[Any]]: return [] diff --git a/tests/unit/test_grammar_semantics_metadata.py b/tests/unit/test_grammar_semantics_metadata.py index 84c33f5c..7c92b622 100644 --- a/tests/unit/test_grammar_semantics_metadata.py +++ b/tests/unit/test_grammar_semantics_metadata.py @@ -2,6 +2,8 @@ from __future__ import annotations +from typing import Any + import pytest from paxman.capabilities import ( @@ -15,6 +17,7 @@ Money, Phone, ) +from paxman.core.domain import Grammar, RecognitionMatch class TestGrammarSemanticsMetadata: @@ -27,3 +30,75 @@ def test_shipped_grammars_declare_semantics_identity(self) -> None: assert isinstance(grammar.semantics, str) assert grammar.semantics != "" assert grammar.semantics == grammar.name + + +class TestGrammarSemanticsEnforcement: + @pytest.mark.unit + def test_bare_grammar_subclass_raises_type_error(self) -> None: + """A Grammar subclass missing ``semantics`` fails at class-definition time.""" + + with pytest.raises(TypeError, match="must define Grammar metadata"): + + class _BareGrammar(Grammar[Any]): + name = "bare_grammar" + + def recognize(self, text: str) -> list[RecognitionMatch[Any]]: + return [] + + @pytest.mark.unit + def test_missing_semantics_raises(self) -> None: + """The missing-``semantics`` error message names the attribute.""" + + with pytest.raises(TypeError, match="semantics"): + + class _MissingSemanticsGrammar(Grammar[Any]): + name = "missing_semantics_grammar" + + def recognize(self, text: str) -> list[RecognitionMatch[Any]]: + return [] + + @pytest.mark.unit + def test_empty_semantics_raises(self) -> None: + """An empty ``semantics`` string is valid type-wise but a bug: + the grammar would carry no semantics. The runtime guard must reject it.""" + with pytest.raises(TypeError, match="non-empty"): + + class _EmptySemanticsGrammar(Grammar[Any]): + name = "empty_semantics_grammar" + semantics = "" + + def recognize(self, text: str) -> list[RecognitionMatch[Any]]: + return [] + + @pytest.mark.unit + @pytest.mark.parametrize("value", [42, frozenset()]) + def test_semantics_must_be_str(self, value: object) -> None: + """A non-str ``semantics`` value fails during Grammar subclass creation.""" + namespace: dict[str, Any] = { + "name": "test_grammar", + "recognize": lambda self, text: [], + "semantics": value, + } + + with pytest.raises(TypeError, match="semantics must be str"): + type("_InvalidSemantics", (Grammar,), namespace) + + @pytest.mark.unit + def test_inherited_semantics_satisfies_enforcement(self) -> None: + """A subclass inheriting ``semantics`` needs no own declaration. + + Locks the ``vars(cls).get`` fallback: the resolved value is read from + the class namespace first, then from the MRO. + """ + + class _CompliantGrammar(Grammar[Any]): + name = "compliant_grammar" + semantics = "compliant_grammar" + + def recognize(self, text: str) -> list[RecognitionMatch[Any]]: + return [] + + class _InheritedSemanticsGrammar(_CompliantGrammar): + pass + + assert _InheritedSemanticsGrammar.semantics == "compliant_grammar" From 0f0f9d1864822e4dfe755231eeedcf6c8a18fcca Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 19:58:54 +0800 Subject: [PATCH 05/22] refactor(engine): route on semantics, not grammar name --- paxman/engine/orchestrator.py | 59 ++++++++++++-------- tests/integration/test_grammar_extensions.py | 4 +- 2 files changed, 37 insertions(+), 26 deletions(-) diff --git a/paxman/engine/orchestrator.py b/paxman/engine/orchestrator.py index 39622789..cd9ca769 100644 --- a/paxman/engine/orchestrator.py +++ b/paxman/engine/orchestrator.py @@ -62,9 +62,13 @@ def run_capability(text: str, contract: Contract) -> ExecutionResult: *get_extended_grammars(capability.name), ] _assert_unique_names("grammar", all_grammars) - all_rules = [*capability.get_rules(), *_activated_rules(capability, contract)] + semantics_by_name = {g.name: g.semantics for g in all_grammars} + all_rules = [ + *capability.get_rules(), + *_activated_rules(capability, contract, semantics_by_name), + ] _assert_unique_names("rule", all_rules) - _validate_affinity(all_grammars, all_rules) + _validate_affinity(semantics_by_name, all_rules) recognitions = _recognize( text, all_grammars, @@ -74,7 +78,7 @@ def run_capability(text: str, contract: Contract) -> ExecutionResult: had_recognitions = len(recognitions) > 0 rules = _filter_rules(all_rules, contract) - candidates = _collect_candidates(capability, recognitions, rules) + candidates = _collect_candidates(capability, recognitions, rules, semantics_by_name) status = _determine_status(candidates, had_recognitions) canonical_value = _extract_canonical_value(candidates, status) @@ -93,8 +97,9 @@ def _assert_unique_names(kind: str, items: Sequence[Grammar[Any] | Rule[Any]]) - """Fail fast when a composed grammar or rule name is duplicated. Shipped names must never be shadowed or duplicated by community - extensions: a duplicate would make grammar-name routing and provenance - attribution ambiguous, so reject it at composition time (D4). + extensions: a duplicate would make provenance attribution ambiguous + (routing is semantic, but names remain the audit identity), so reject it + at composition time (D4). """ names = [item.name for item in items] duplicates = sorted({name for name in names if names.count(name) > 1}) @@ -274,21 +279,21 @@ def _filter_rules(all_rules: list[Rule[Any]], contract: Contract) -> list[Rule[A def _validate_affinity( - all_grammars: Sequence[Grammar[Any]], rules: list[Rule[Any]] + semantics_by_name: dict[str, str], rules: list[Rule[Any]] ) -> None: - """Ensure every rule's declared grammars exist in the composition. + """Ensure every rule's declared semantics exist in the composition. The composition covers shipped and community grammars alike; a dangling - grammar name would silently exclude a rule from ever running, so fail + semantics would silently exclude a rule from ever running, so fail fast at pipeline start rather than producing a wrong (e.g. INVALID) result. """ - known_grammars = {g.name for g in all_grammars} + known_semantics = set(semantics_by_name.values()) for rule in rules: - unknown = [g for g in rule.target_semantics if g not in known_grammars] + unknown = [s for s in rule.target_semantics if s not in known_semantics] if unknown: raise ContractError( - f"Rule {rule.name!r} declares unknown grammar(s) " - f"{sorted(unknown)}; available: {sorted(known_grammars)}" + f"Rule {rule.name!r} declares unknown semantics " + f"{sorted(unknown)}; available: {sorted(known_semantics)}" ) @@ -296,20 +301,21 @@ def _collect_candidates( capability: Capability[Any], recognitions: list[RecognizedRep[Any]], rules: list[Rule[Any]], + semantics_by_name: dict[str, str], ) -> list[Candidate]: """Match recognitions against rules and collect candidates. - Routes each recognition only to rules whose ``target_semantics`` includes the - producing grammar's name (ARCHITECTURE.md:201), formats each validated - value through the capability's ``format_value()`` seam, then dedups - identical candidate tuples so the candidate multiset is stable regardless - of routing. + Routes each recognition only to rules whose ``target_semantics`` includes + the producing grammar's semantics (ARCHITECTURE.md:201), formats each + validated value through the capability's ``format_value()`` seam, then + dedups identical candidate tuples so the candidate multiset is stable + regardless of routing. """ candidates: list[Candidate] = [] for recognition in recognitions: grammar_name = recognition.grammar.grammar_name for rule in rules: - if grammar_name not in rule.target_semantics: + if semantics_by_name[grammar_name] not in rule.target_semantics: continue try: if rule.matches(recognition.notation, recognition.contract): @@ -383,18 +389,23 @@ def _extract_canonical_value( def _activated_rules( - capability: Capability[Any], contract: Contract + capability: Capability[Any], + contract: Contract, + semantics_by_name: dict[str, str], ) -> list[Rule[Any]]: """Community rules opt in like grammars: a rule runs only when the - contract names one of its ``target_semantics`` in ``extra_grammars``. + contract's ``extra_grammars`` resolve to one of its ``target_semantics``. - An un-opted community rule — even one targeting a shipped grammar — - never affects results, keeping extension behavior deterministic per - contract. + An unknown extra name keeps its own string as the semantics key, so a + rule declaring it (a dangling target) still fails fast in affinity + validation instead of being silently excluded. An un-opted community + rule — even one targeting a shipped grammar — never affects results, + keeping extension behavior deterministic per contract. """ extra_grammars = set(getattr(contract, "extra_grammars", ())) + extra_semantics = {semantics_by_name.get(n, n) for n in extra_grammars} return [ rule for rule in get_extended_rules(capability.name) - if extra_grammars & rule.target_semantics + if extra_semantics & rule.target_semantics ] diff --git a/tests/integration/test_grammar_extensions.py b/tests/integration/test_grammar_extensions.py index e40e3f59..ed5ab092 100644 --- a/tests/integration/test_grammar_extensions.py +++ b/tests/integration/test_grammar_extensions.py @@ -238,12 +238,12 @@ def test_community_grammar_collision_with_shipped_raises(self) -> None: @pytest.mark.integration def test_community_rule_dangling_target_raises(self) -> None: - """An opted-in community rule naming a missing grammar fails fast.""" + """An opted-in community rule naming a missing semantics fails fast.""" register_rule("date", DanglingDateRule) contract = DateContract( extra_grammars=("dot_date_recognition", "no_such_grammar") ) - with pytest.raises(ContractError, match="unknown grammar"): + with pytest.raises(ContractError, match="unknown semantics"): run_capability("2024.01.01", contract) @pytest.mark.integration From 3c592b8606b211e9cbc7c4253d93357e554cd579 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 20:24:32 +0800 Subject: [PATCH 06/22] refactor(capabilities): coalesce Date grammars to calendar-date semantics --- .../Date/grammar/european_recognition.py | 2 +- .../Date/grammar/iso8601_recognition.py | 2 +- .../Date/grammar/slash_iso_recognition.py | 2 +- .../Date/grammar/us_recognition.py | 2 +- .../Date/rules/en_50160_ed2010.py | 2 +- .../Date/rules/iso_8601_ed2019.py | 2 +- .../Date/rules/us_federal_rules_ed2023.py | 2 +- tests/integration/test_grammar_extensions.py | 2 +- .../test_grammar_semantics_consistency.py | 106 ++++++++++++++++++ tests/unit/test_grammar_semantics_metadata.py | 16 ++- 10 files changed, 128 insertions(+), 10 deletions(-) create mode 100644 tests/unit/test_grammar_semantics_consistency.py diff --git a/paxman/capabilities/Date/grammar/european_recognition.py b/paxman/capabilities/Date/grammar/european_recognition.py index 210f7a46..6ef5855c 100644 --- a/paxman/capabilities/Date/grammar/european_recognition.py +++ b/paxman/capabilities/Date/grammar/european_recognition.py @@ -21,7 +21,7 @@ class EuropeanDateGrammar(Grammar[DateNotation]): """ name = "european_recognition" - semantics = "european_recognition" + semantics = "european_calendar_date" def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: """Extract European date patterns from text. diff --git a/paxman/capabilities/Date/grammar/iso8601_recognition.py b/paxman/capabilities/Date/grammar/iso8601_recognition.py index 9ebe1d67..6772b638 100644 --- a/paxman/capabilities/Date/grammar/iso8601_recognition.py +++ b/paxman/capabilities/Date/grammar/iso8601_recognition.py @@ -20,7 +20,7 @@ class ISO8601DateGrammar(Grammar[DateNotation]): """ name = "iso8601_recognition" - semantics = "iso8601_recognition" + semantics = "iso8601_calendar_date" def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: """Extract ISO 8601 date patterns from text.""" diff --git a/paxman/capabilities/Date/grammar/slash_iso_recognition.py b/paxman/capabilities/Date/grammar/slash_iso_recognition.py index 17be5453..e3e48a12 100644 --- a/paxman/capabilities/Date/grammar/slash_iso_recognition.py +++ b/paxman/capabilities/Date/grammar/slash_iso_recognition.py @@ -26,7 +26,7 @@ class SlashISODateGrammar(Grammar[DateNotation]): """ name = "slash_iso_recognition" - semantics = "slash_iso_recognition" + semantics = "iso8601_calendar_date" def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: """Extract YYYY/MM/DD date patterns from text.""" diff --git a/paxman/capabilities/Date/grammar/us_recognition.py b/paxman/capabilities/Date/grammar/us_recognition.py index 7fbb1253..d85a6b99 100644 --- a/paxman/capabilities/Date/grammar/us_recognition.py +++ b/paxman/capabilities/Date/grammar/us_recognition.py @@ -21,7 +21,7 @@ class USDateGrammar(Grammar[DateNotation]): """ name = "us_recognition" - semantics = "us_recognition" + semantics = "us_calendar_date" def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: """Extract US date patterns from text. diff --git a/paxman/capabilities/Date/rules/en_50160_ed2010.py b/paxman/capabilities/Date/rules/en_50160_ed2010.py index 15cb29c4..5d910ec2 100644 --- a/paxman/capabilities/Date/rules/en_50160_ed2010.py +++ b/paxman/capabilities/Date/rules/en_50160_ed2010.py @@ -30,7 +30,7 @@ class Section4DateFormat(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 4 (date format)" - target_semantics = frozenset({"us_recognition", "european_recognition"}) + target_semantics = frozenset({"us_calendar_date", "european_calendar_date"}) requires_features = frozenset() def _interpret_two_digit_year(self, year_str: str, contract: Contract) -> int: diff --git a/paxman/capabilities/Date/rules/iso_8601_ed2019.py b/paxman/capabilities/Date/rules/iso_8601_ed2019.py index d15d32d6..16f6746c 100644 --- a/paxman/capabilities/Date/rules/iso_8601_ed2019.py +++ b/paxman/capabilities/Date/rules/iso_8601_ed2019.py @@ -33,7 +33,7 @@ class Section431CalendarDate(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 4.3.1 (calendar date)" - target_semantics = frozenset({"iso8601_recognition", "slash_iso_recognition"}) + target_semantics = frozenset({"iso8601_calendar_date"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: diff --git a/paxman/capabilities/Date/rules/us_federal_rules_ed2023.py b/paxman/capabilities/Date/rules/us_federal_rules_ed2023.py index 1d627ad8..4abde9fb 100644 --- a/paxman/capabilities/Date/rules/us_federal_rules_ed2023.py +++ b/paxman/capabilities/Date/rules/us_federal_rules_ed2023.py @@ -30,7 +30,7 @@ class Section1DateFormat(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 1 (date format)" - target_semantics = frozenset({"us_recognition", "european_recognition"}) + target_semantics = frozenset({"us_calendar_date", "european_calendar_date"}) requires_features = frozenset() def _interpret_two_digit_year(self, year_str: str, contract: Contract) -> int: diff --git a/tests/integration/test_grammar_extensions.py b/tests/integration/test_grammar_extensions.py index ed5ab092..81781334 100644 --- a/tests/integration/test_grammar_extensions.py +++ b/tests/integration/test_grammar_extensions.py @@ -135,7 +135,7 @@ class CommunityISO8601Rule(Rule[DateNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "community test double" - target_semantics = frozenset({"iso8601_recognition"}) + target_semantics = frozenset({"iso8601_calendar_date"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: diff --git a/tests/unit/test_grammar_semantics_consistency.py b/tests/unit/test_grammar_semantics_consistency.py new file mode 100644 index 00000000..480a7374 --- /dev/null +++ b/tests/unit/test_grammar_semantics_consistency.py @@ -0,0 +1,106 @@ +"""D8 — same-semantics grammars must produce identical notation field mappings +and canonicalization; guards semantic affinity routing. + +The affinity-routing engine treats every grammar claiming the same +``semantics`` id as interchangeable: any member of a group may recognize an +input, and its notation is routed to the group's shared rules. A group whose +members map the same input to different notation fields (or whose shared rule +canonicalizes differently) would resolve the same text differently depending +on which member happened to recognize it — silent nondeterminism. This guard +enumerates all shipped grammar classes, groups them by ``semantics``, and for +every seeded group runs shared probe rows through each member's +``recognize()`` asserting identical notation fields and canonical values. +""" + +from __future__ import annotations + +from typing import Any, NamedTuple + +import pytest + +from paxman.capabilities import ( + IP, + ISBN, + URL, + Country, + Currency, + Date, + Email, + Money, + Phone, +) +from paxman.capabilities.Date.contract import DateContract +from paxman.capabilities.Date.notation import DateNotation +from paxman.capabilities.Date.rules.iso_8601_ed2019 import Section431CalendarDate +from paxman.core.domain import Grammar, Rule + +_SHIPPED_CAPABILITIES = [Country, Currency, Date, Email, IP, ISBN, Money, Phone, URL] + + +class _ProbeRow(NamedTuple): + """One input run through every member of a semantics group.""" + + input: str + expected_notation: DateNotation + expected_canonical: str + + +# Probe rows keyed by semantics id. Each key must name a real group in the +# shipped grammar enumeration (test A); each member of a group must recognize +# the probe input into the identical notation, and the group's shared rule +# must canonicalize it identically (test B). +_PROBE_ROWS: dict[str, tuple[type[Rule[Any]], tuple[_ProbeRow, ...]]] = { + "iso8601_calendar_date": ( + Section431CalendarDate, + ( + _ProbeRow( + input="2026-01-15", + expected_notation=DateNotation(N1="2026", N2="01", N3="15"), + expected_canonical="2026-01-15", + ), + _ProbeRow( + input="2026/01/15", + expected_notation=DateNotation(N1="2026", N2="01", N3="15"), + expected_canonical="2026-01-15", + ), + ), + ), +} + + +def _group_shipped_grammars_by_semantics() -> dict[str, list[type[Grammar[Any]]]]: + """Group every shipped grammar class by its ``semantics`` id.""" + groups: dict[str, list[type[Grammar[Any]]]] = {} + for capability in _SHIPPED_CAPABILITIES: + for grammar in capability().get_grammars(): + groups.setdefault(grammar.semantics, []).append(type(grammar)) + return groups + + +@pytest.mark.unit +def test_probe_keys_name_real_semantics_groups() -> None: + """Every probe-table key must be a real semantics group in the enumeration.""" + groups = _group_shipped_grammars_by_semantics() + assert set(_PROBE_ROWS) <= set(groups) + + +@pytest.mark.unit +def test_same_semantics_grammars_agree_on_notation_and_canonical() -> None: + """Members of a seeded semantics group recognize probes identically. + + Groups without probe rows are skipped — the reverse coverage (that every + multi-member group is seeded) lands in a later task. + """ + groups = _group_shipped_grammars_by_semantics() + for semantics, (rule_cls, probes) in _PROBE_ROWS.items(): + rule = rule_cls() + for member_cls in groups.get(semantics, ()): + member = member_cls() + for probe in probes: + matches = member.recognize(probe.input) + if not matches: + continue + assert matches[0].notation == probe.expected_notation + assert rule.normalize(matches[0].notation, DateContract()) == ( + probe.expected_canonical + ) diff --git a/tests/unit/test_grammar_semantics_metadata.py b/tests/unit/test_grammar_semantics_metadata.py index 7c92b622..f5249154 100644 --- a/tests/unit/test_grammar_semantics_metadata.py +++ b/tests/unit/test_grammar_semantics_metadata.py @@ -19,17 +19,29 @@ ) from paxman.core.domain import Grammar, RecognitionMatch +# Semantics ids that intentionally coalesce several grammars onto one shared +# id (semantic affinity routing, ADR-0003): a grammar in this set legitimately +# declares ``semantics`` differing from its ``name``. +_COALESCED_SEMANTICS: frozenset[str] = frozenset( + {"iso8601_calendar_date", "us_calendar_date", "european_calendar_date"} +) + class TestGrammarSemanticsMetadata: @pytest.mark.unit def test_shipped_grammars_declare_semantics_identity(self) -> None: - """Every shipped grammar declares ``semantics`` equal to its name.""" + """Every shipped grammar declares ``semantics``: identity with its name + for non-coalesced grammars, or one of the coalesced ids (an explicit + allowlist) for grammars sharing a semantic group.""" capabilities = [Country, Currency, Date, Email, IP, ISBN, Money, Phone, URL] for capability in capabilities: for grammar in capability().get_grammars(): assert isinstance(grammar.semantics, str) assert grammar.semantics != "" - assert grammar.semantics == grammar.name + assert ( + grammar.semantics == grammar.name + or grammar.semantics in _COALESCED_SEMANTICS + ) class TestGrammarSemanticsEnforcement: From 93175960af08ab2ae9b3999e049fc9e2f4fd9c00 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 20:30:19 +0800 Subject: [PATCH 07/22] refactor(capabilities): coalesce Email grammars to addr-spec semantics --- .../Email/grammar/obfuscated_recognition.py | 2 +- .../Email/grammar/standard_recognition.py | 2 +- .../Email/rules/rfc_5322_ed2008.py | 2 +- .../test_grammar_semantics_consistency.py | 21 +++++++++++++++++++ tests/unit/test_grammar_semantics_metadata.py | 7 ++++++- 5 files changed, 30 insertions(+), 4 deletions(-) diff --git a/paxman/capabilities/Email/grammar/obfuscated_recognition.py b/paxman/capabilities/Email/grammar/obfuscated_recognition.py index 30f53064..e219b0db 100644 --- a/paxman/capabilities/Email/grammar/obfuscated_recognition.py +++ b/paxman/capabilities/Email/grammar/obfuscated_recognition.py @@ -21,7 +21,7 @@ class ObfuscatedEmailGrammar(Grammar[EmailNotation]): """Obfuscated email: 'user at domain dot tld' or 'user at domain.tld'.""" name = "obfuscated_recognition" - semantics = "obfuscated_recognition" + semantics = "rfc5322_addr_spec" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: """Extract obfuscated email patterns from text. diff --git a/paxman/capabilities/Email/grammar/standard_recognition.py b/paxman/capabilities/Email/grammar/standard_recognition.py index cbb156f3..9505a8cb 100644 --- a/paxman/capabilities/Email/grammar/standard_recognition.py +++ b/paxman/capabilities/Email/grammar/standard_recognition.py @@ -16,7 +16,7 @@ class StandardEmailGrammar(Grammar[EmailNotation]): """Standard email recognition: user@domain.tld.""" name = "standard_recognition" - semantics = "standard_recognition" + semantics = "rfc5322_addr_spec" def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: matches: list[RecognitionMatch[EmailNotation]] = [] diff --git a/paxman/capabilities/Email/rules/rfc_5322_ed2008.py b/paxman/capabilities/Email/rules/rfc_5322_ed2008.py index fc681110..9742e214 100644 --- a/paxman/capabilities/Email/rules/rfc_5322_ed2008.py +++ b/paxman/capabilities/Email/rules/rfc_5322_ed2008.py @@ -33,7 +33,7 @@ class Section341AddrSpec(Rule[EmailNotation]): strategy = RuleStrategy.REGEX provenance = PUBLICATION citation = "Section 3.4.1 (addr-spec)" - target_semantics = frozenset({"standard_recognition", "obfuscated_recognition"}) + target_semantics = frozenset({"rfc5322_addr_spec"}) requires_features = frozenset() def matches(self, notation: EmailNotation, contract: Contract) -> bool: diff --git a/tests/unit/test_grammar_semantics_consistency.py b/tests/unit/test_grammar_semantics_consistency.py index 480a7374..f1c46ae2 100644 --- a/tests/unit/test_grammar_semantics_consistency.py +++ b/tests/unit/test_grammar_semantics_consistency.py @@ -32,6 +32,8 @@ from paxman.capabilities.Date.contract import DateContract from paxman.capabilities.Date.notation import DateNotation from paxman.capabilities.Date.rules.iso_8601_ed2019 import Section431CalendarDate +from paxman.capabilities.Email.notation import EmailNotation +from paxman.capabilities.Email.rules.rfc_5322_ed2008 import Section341AddrSpec from paxman.core.domain import Grammar, Rule _SHIPPED_CAPABILITIES = [Country, Currency, Date, Email, IP, ISBN, Money, Phone, URL] @@ -65,6 +67,25 @@ class _ProbeRow(NamedTuple): ), ), ), + "rfc5322_addr_spec": ( + Section341AddrSpec, + ( + _ProbeRow( + input="user@example.com", + expected_notation=EmailNotation( + local_part="user", domain_part="example.com" + ), + expected_canonical="user@example.com", + ), + _ProbeRow( + input="user at example dot com", + expected_notation=EmailNotation( + local_part="user", domain_part="example.com" + ), + expected_canonical="user@example.com", + ), + ), + ), } diff --git a/tests/unit/test_grammar_semantics_metadata.py b/tests/unit/test_grammar_semantics_metadata.py index f5249154..c4464fb0 100644 --- a/tests/unit/test_grammar_semantics_metadata.py +++ b/tests/unit/test_grammar_semantics_metadata.py @@ -23,7 +23,12 @@ # id (semantic affinity routing, ADR-0003): a grammar in this set legitimately # declares ``semantics`` differing from its ``name``. _COALESCED_SEMANTICS: frozenset[str] = frozenset( - {"iso8601_calendar_date", "us_calendar_date", "european_calendar_date"} + { + "iso8601_calendar_date", + "us_calendar_date", + "european_calendar_date", + "rfc5322_addr_spec", + } ) From fd22152c9ce37d756f507864ba925776ad46a372 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 20:32:54 +0800 Subject: [PATCH 08/22] test: widen _ProbeRow notation type for multi-capability groups --- tests/unit/test_grammar_semantics_consistency.py | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/tests/unit/test_grammar_semantics_consistency.py b/tests/unit/test_grammar_semantics_consistency.py index f1c46ae2..bf2a4f10 100644 --- a/tests/unit/test_grammar_semantics_consistency.py +++ b/tests/unit/test_grammar_semantics_consistency.py @@ -40,10 +40,14 @@ class _ProbeRow(NamedTuple): - """One input run through every member of a semantics group.""" + """One input run through every member of a semantics group. + + ``expected_notation`` is deliberately untyped: the probe table spans + capabilities, so each row's notation type is the group's own. + """ input: str - expected_notation: DateNotation + expected_notation: object expected_canonical: str From abae3e4e96cc3cdad18c4a2d20a5e80b5886c75b Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 20:36:25 +0800 Subject: [PATCH 09/22] refactor(capabilities): coalesce Phone grammars to E.164 semantics --- .../Phone/grammar/e164_recognition.py | 2 +- .../grammar/international_00_recognition.py | 2 +- paxman/capabilities/Phone/rules/e164_ed2010.py | 4 ++-- .../unit/test_grammar_semantics_consistency.py | 17 +++++++++++++++++ tests/unit/test_grammar_semantics_metadata.py | 1 + 5 files changed, 22 insertions(+), 4 deletions(-) diff --git a/paxman/capabilities/Phone/grammar/e164_recognition.py b/paxman/capabilities/Phone/grammar/e164_recognition.py index f518d70c..bcf4f153 100644 --- a/paxman/capabilities/Phone/grammar/e164_recognition.py +++ b/paxman/capabilities/Phone/grammar/e164_recognition.py @@ -57,7 +57,7 @@ class E164Grammar(Grammar[PhoneNotation]): """ name = "e164_recognition" - semantics = "e164_recognition" + semantics = "e164_international" def recognize(self, text: str) -> list[RecognitionMatch[PhoneNotation]]: """Extract e164 patterns from text. diff --git a/paxman/capabilities/Phone/grammar/international_00_recognition.py b/paxman/capabilities/Phone/grammar/international_00_recognition.py index eb180fa9..d66f909a 100644 --- a/paxman/capabilities/Phone/grammar/international_00_recognition.py +++ b/paxman/capabilities/Phone/grammar/international_00_recognition.py @@ -37,7 +37,7 @@ class International00Grammar(Grammar[PhoneNotation]): """ name = "international_00_recognition" - semantics = "international_00_recognition" + semantics = "e164_international" def recognize(self, text: str) -> list[RecognitionMatch[PhoneNotation]]: """Extract 00-prefixed international patterns from text. diff --git a/paxman/capabilities/Phone/rules/e164_ed2010.py b/paxman/capabilities/Phone/rules/e164_ed2010.py index 48334612..2c36ec89 100644 --- a/paxman/capabilities/Phone/rules/e164_ed2010.py +++ b/paxman/capabilities/Phone/rules/e164_ed2010.py @@ -66,7 +66,7 @@ class Section6_1InternationalNumber(Rule[PhoneNotation]): strategy = RuleStrategy.PARSER provenance = PUBLICATION citation = "Section 6.1 (number structure)" - target_semantics = frozenset({"e164_recognition", "international_00_recognition"}) + target_semantics = frozenset({"e164_international"}) requires_features = frozenset() def matches(self, notation: PhoneNotation, contract: Contract) -> bool: @@ -108,7 +108,7 @@ class Section6_2CountryCode(Rule[PhoneNotation]): strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION citation = "Annex A (table of assigned country codes)" - target_semantics = frozenset({"e164_recognition", "international_00_recognition"}) + target_semantics = frozenset({"e164_international"}) requires_features = frozenset() def matches(self, notation: PhoneNotation, contract: Contract) -> bool: diff --git a/tests/unit/test_grammar_semantics_consistency.py b/tests/unit/test_grammar_semantics_consistency.py index bf2a4f10..b87c01c2 100644 --- a/tests/unit/test_grammar_semantics_consistency.py +++ b/tests/unit/test_grammar_semantics_consistency.py @@ -34,6 +34,8 @@ from paxman.capabilities.Date.rules.iso_8601_ed2019 import Section431CalendarDate from paxman.capabilities.Email.notation import EmailNotation from paxman.capabilities.Email.rules.rfc_5322_ed2008 import Section341AddrSpec +from paxman.capabilities.Phone.notation import PhoneNotation +from paxman.capabilities.Phone.rules.e164_ed2010 import Section6_1InternationalNumber from paxman.core.domain import Grammar, Rule _SHIPPED_CAPABILITIES = [Country, Currency, Date, Email, IP, ISBN, Money, Phone, URL] @@ -90,6 +92,21 @@ class _ProbeRow(NamedTuple): ), ), ), + "e164_international": ( + Section6_1InternationalNumber, + ( + _ProbeRow( + input="+15551234567", + expected_notation=PhoneNotation(shape="e164", value="15551234567"), + expected_canonical="+15551234567", + ), + _ProbeRow( + input="0015551234567", + expected_notation=PhoneNotation(shape="e164", value="15551234567"), + expected_canonical="+15551234567", + ), + ), + ), } diff --git a/tests/unit/test_grammar_semantics_metadata.py b/tests/unit/test_grammar_semantics_metadata.py index c4464fb0..7e20e3f1 100644 --- a/tests/unit/test_grammar_semantics_metadata.py +++ b/tests/unit/test_grammar_semantics_metadata.py @@ -28,6 +28,7 @@ "us_calendar_date", "european_calendar_date", "rfc5322_addr_spec", + "e164_international", } ) From 36315dc569666e787881eba8de9257088bb16d1c Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 20:42:08 +0800 Subject: [PATCH 10/22] test: lock same-semantics field-mapping consistency --- .../test_grammar_semantics_consistency.py | 66 +++++++++++++++++++ 1 file changed, 66 insertions(+) diff --git a/tests/unit/test_grammar_semantics_consistency.py b/tests/unit/test_grammar_semantics_consistency.py index b87c01c2..3c162546 100644 --- a/tests/unit/test_grammar_semantics_consistency.py +++ b/tests/unit/test_grammar_semantics_consistency.py @@ -40,6 +40,20 @@ _SHIPPED_CAPABILITIES = [Country, Currency, Date, Email, IP, ISBN, Money, Phone, URL] +# D7 no-coalesce ids: groups that must NEVER grow a second member. The +# date formats and the six identity singletons are distinct enough that +# coalescing them would silently change what they resolve. +_NO_COALESCE_SEMANTICS = ( + "us_calendar_date", + "european_calendar_date", + "name_recognition", + "alpha2_recognition", + "alpha3_recognition", + "numeric_recognition", + "isbn13_recognition", + "isbn10_recognition", +) + class _ProbeRow(NamedTuple): """One input run through every member of a semantics group. @@ -146,3 +160,55 @@ def test_same_semantics_grammars_agree_on_notation_and_canonical() -> None: assert rule.normalize(matches[0].notation, DateContract()) == ( probe.expected_canonical ) + + +@pytest.mark.unit +def test_every_shipped_grammar_belongs_to_one_semantics_group() -> None: + """No shipped grammar is dropped or duplicated by the semantics grouping. + + Every grammar enumerated via ``get_grammars()`` must land in exactly one + group with a non-empty semantics id; a dropped or double-counted grammar + would break the member-count equality. + """ + groups = _group_shipped_grammars_by_semantics() + shipped_count = sum( + len(capability().get_grammars()) for capability in _SHIPPED_CAPABILITIES + ) + assert sum(len(members) for members in groups.values()) == shipped_count + assert all(semantics for semantics in groups) + + +@pytest.mark.unit +def test_every_multi_member_semantics_group_has_probe_rows() -> None: + """A coalesced group must be seeded or the guard fails loudly. + + The affinity-routing engine only treats grammars as interchangeable within + one capability, so a multi-member group arises only from a coalescing + inside a capability. Cross-capability id reuse (Currency and Money both + declaring ``code_recognition`` etc.) is per-capability identity — those + grammars never co-route — and must not demand probe rows. A future + coalescing that adds a group without probe rows bypasses the same-notation + field-mapping guarantee and fails here. + """ + multi_member_ids: set[str] = set() + for capability in _SHIPPED_CAPABILITIES: + counts: dict[str, int] = {} + for grammar in capability().get_grammars(): + counts[grammar.semantics] = counts.get(grammar.semantics, 0) + 1 + multi_member_ids.update( + semantics for semantics, count in counts.items() if count > 1 + ) + assert multi_member_ids <= set(_PROBE_ROWS) + + +@pytest.mark.unit +def test_d7_no_coalesce_semantics_groups_stay_singleton() -> None: + """The D7-locked groups must never grow a second member. + + ``us_calendar_date``/``european_calendar_date`` are renamed singletons and + the other six are identity singletons; coalescing any of them would change + what the shared semantics resolves to. + """ + groups = _group_shipped_grammars_by_semantics() + for semantics in _NO_COALESCE_SEMANTICS: + assert len(groups[semantics]) == 1 From f6f7790f79cf23c4a135d9cdf46f5ee29342ce21 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 21:04:04 +0800 Subject: [PATCH 11/22] docs: sweep target_grammars and document semantic affinity --- ARCHITECTURE.md | 6 ++--- CONTEXT.md | 9 ++++--- HOW_TO_ADD_NEW_CAPABILITY.md | 17 +++++++------ HOW_TO_ADD_NEW_GRAMMAR.md | 44 ++++++++++++++------------------- README.md | 9 ++++--- capability_homogeneity_audit.md | 14 +++++------ paxman/capabilities/AGENTS.md | 4 +-- paxman/core/AGENTS.md | 4 +-- paxman/engine/orchestrator.py | 12 ++++++--- 9 files changed, 60 insertions(+), 59 deletions(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 4d8beab4..393ae5ff 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -97,7 +97,7 @@ The engine is the orchestration layer that coordinates the full pipeline. It: The engine is capability-agnostic. It does not know what a "grammar" or "rule" does — it only knows that grammars produce span-bearing recognition matches and rules produce candidates. -Before the recognition phase, the engine composes each capability's shipped grammars and rules with any community extensions registered for that capability (see "Community Extensions" below). Composition is guarded: duplicate names fail fast, and every rule's declared `target_grammars` must resolve within the composed set. +Before the recognition phase, the engine composes each capability's shipped grammars and rules with any community extensions registered for that capability (see "Community Extensions" below). Composition is guarded: duplicate names fail fast, and every rule's declared `target_semantics` must resolve within the composed set. ### Public API @@ -171,9 +171,9 @@ Capabilities are closed for modification but open for extension. Community contr A contract opts a registered grammar in by naming it in `extra_grammars`, a base `CapabilityContract` field surfaced on every `create_contract` factory. The engine composes the shipped active set with the opted-in extras, deduplicating names while preserving order — shipped slots first, extras after (unknown extra names are silently skipped). The shipped slots are `contract.active_grammars` when the contract implements it (the gated capabilities), or every shipped grammar in `get_grammars()` order when it returns `None` (the base default). Opt-in preserves determinism: a contract that names no extras composes to exactly the shipped set, so non-opt-in behavior is byte-identical. -Community rules follow the same opt-in discipline: a registered rule runs only when the contract names one of its `target_grammars` in `extra_grammars`. An un-opted community rule — even one targeting a shipped grammar — never affects results, so a default contract resolves with shipped rules only. +Community rules follow the same opt-in discipline: a registered rule runs only when the contract's `extra_grammars` resolve to one of its `target_semantics` ids. An un-opted community rule — even one targeting a shipped grammar's semantics — never affects results, so a default contract resolves with shipped rules only. -Composition is guarded at pipeline start: a community grammar name colliding with a shipped name raises `CapabilityError`, and an opted-in community rule whose `target_grammars` names a missing grammar raises `ContractError` — failing fast rather than producing a silently wrong result. Community grammars and rules are pure functions of their inputs, and the composed set is fixed once the registries freeze, so the determinism guarantees of "Determinism by Construction" extend unchanged. +Composition is guarded at pipeline start: a community grammar name colliding with a shipped name raises `CapabilityError`, and an opted-in community rule whose `target_semantics` names an id no grammar claims raises `ContractError` — failing fast rather than producing a silently wrong result. Community grammars and rules are pure functions of their inputs, and the composed set is fixed once the registries freeze, so the determinism guarantees of "Determinism by Construction" extend unchanged. --- diff --git a/CONTEXT.md b/CONTEXT.md index 4da7da5f..1d1ec887 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -137,7 +137,7 @@ class Section341AddrSpec(Rule[EmailNotation]): strategy = RuleStrategy.REGEX provenance = PUBLICATION citation = "Section 3.4.1 (addr-spec)" # Human-readable citation - target_grammars = frozenset({"standard_recognition", "obfuscated_recognition"}) + target_semantics = frozenset({"rfc5322_addr_spec"}) requires_features = frozenset() # authority features this rule gates on def matches(self, notation: EmailNotation, contract: Contract) -> bool: @@ -152,7 +152,7 @@ class Section341AddrSpec(Rule[EmailNotation]): return f"{notation.local_part.lower()}@{notation.domain_part.lower()}" ``` -Every rule declares six metadata attrs — `name`, `strategy`, `provenance`, `citation`, `target_grammars` (non-empty `frozenset[str]`), `requires_features` (`frozenset[str]`) — enforced at import time by `Rule.__init_subclass__`. Rules never raise and never read `output_format`. +Every rule declares six metadata attrs — `name`, `strategy`, `provenance`, `citation`, `target_semantics` (non-empty `frozenset[str]`), `requires_features` (`frozenset[str]`) — enforced at import time by `Rule.__init_subclass__`. Every grammar declares a `semantics` string — the meaning id its recognized notations carry (its identity id by default; a coalesced id shared by same-meaning grammars, e.g. the standard and obfuscated Email grammars both declare `"rfc5322_addr_spec"`) — enforced as a non-empty `str` at import time by `Grammar.__init_subclass__`. Rules never raise and never read `output_format`. ### Notation Purpose Notation exists for **placement-sensitive rules**: @@ -179,7 +179,7 @@ class SectionCode(Rule[CurrencyNotation]): name = "Section 3-code" strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION - target_grammars = frozenset({"code_recognition", "symbol_recognition", "word_recognition"}) + target_semantics = frozenset({"code_recognition", "symbol_recognition", "word_recognition"}) requires_features = frozenset() def matches(self, notation: CurrencyNotation, contract: Contract) -> bool: @@ -204,7 +204,7 @@ class Section431CalendarDate(Rule[DateNotation]): name = "Section 4.3.1-calendar-date" strategy = RuleStrategy.PARSER provenance = PUBLICATION - target_grammars = frozenset({"iso8601_recognition"}) + target_semantics = frozenset({"iso8601_calendar_date"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: @@ -244,6 +244,7 @@ class StandardEmailGrammar(Grammar[EmailNotation]): """Standard email recognition: user@domain.tld""" name = "standard_recognition" + semantics = "rfc5322_addr_spec" # coalesced — shared with obfuscated_recognition def recognize(self, text: str) -> list[RecognitionMatch[EmailNotation]]: """Extract span-bearing email matches from text.""" diff --git a/HOW_TO_ADD_NEW_CAPABILITY.md b/HOW_TO_ADD_NEW_CAPABILITY.md index 278770c9..b2750d88 100644 --- a/HOW_TO_ADD_NEW_CAPABILITY.md +++ b/HOW_TO_ADD_NEW_CAPABILITY.md @@ -169,6 +169,7 @@ class StandardMyDomainGrammar(Grammar[MyDomainNotation]): """Standard recognition for the MyDomain capability.""" name = "standard_recognition" + semantics = "standard_recognition" def recognize(self, text: str) -> list[RecognitionMatch[MyDomainNotation]]: """Extract span-bearing matches from raw text. @@ -289,7 +290,7 @@ Codebase examples: IP's IPv4 grammar is a loose regex (`\d{1,3}` octets) and `rf 3. Set `provenance` to the `PUBLICATION` constant defined above 4. Set `citation` to a human-readable citation (e.g., "Section 3.4.1 (addr-spec)") -5. Set `target_grammars` to the `frozenset[str]` of grammar names whose notations this rule validates (e.g., `frozenset({"standard_recognition"})`) +5. Set `target_semantics` to the `frozenset[str]` of grammar semantics whose notations this rule validates (e.g., `frozenset({"standard_recognition"})`) 6. Set `requires_features` to the `frozenset[str]` of contract fields that must be truthy for the rule to run (`frozenset()` when it always runs) All six attributes are enforced by `Rule.__init_subclass__` at class-definition time; see the Rule metadata section in Step 7. @@ -438,7 +439,7 @@ Capability-specific parameters come after the common block. Every capability sat **3. Rule metadata** -Every `Rule` subclass must declare six class attributes: `name`, `strategy`, `provenance`, `citation`, `target_grammars`, and `requires_features`. `Rule.__init_subclass__` enforces this at class-definition time, and a subclass missing any of them fails to import with a `TypeError`: +Every `Rule` subclass must declare six class attributes: `name`, `strategy`, `provenance`, `citation`, `target_semantics`, and `requires_features`. `Rule.__init_subclass__` enforces this at class-definition time, and a subclass missing any of them fails to import with a `TypeError`: ```python class SectionYourRule(Rule[YourDomainNotation]): @@ -446,11 +447,11 @@ class SectionYourRule(Rule[YourDomainNotation]): strategy = RuleStrategy.REGEX provenance = PUBLICATION citation = "Section 1 (your-rule)" - target_grammars = frozenset({"your_recognition"}) + target_semantics = frozenset({"your_recognition"}) requires_features = frozenset() ``` -- **`target_grammars: ClassVar[frozenset[str]]`** is the non-empty set of grammar names whose notations this rule validates. The engine uses it for affinity routing: each recognition is validated only by rules whose `target_grammars` includes the producing grammar's name, and a rule declaring a grammar the capability does not have fails fast with a `ContractError` before any candidate is produced. `Rule.__init_subclass__` also rejects an empty set at import time, since such a rule could never match a recognition. +- **`target_semantics: ClassVar[frozenset[str]]`** is the non-empty set of grammar `semantics` ids whose notations this rule validates. The engine uses it for affinity routing: each recognition is validated only by rules whose `target_semantics` includes the producing grammar's `semantics`, and a rule declaring a semantics id no grammar claims fails fast with a `ContractError` before any candidate is produced. `Rule.__init_subclass__` also rejects an empty set at import time, since such a rule could never match a recognition. - **`requires_features: ClassVar[frozenset[str]]`** is the set of Contract field names that must be truthy for the rule to run. An empty set is valid and is the common case: it means the rule always runs once selected. The engine validates that every named feature exists on the contract (a missing name raises `ContractError`) and applies the final feature filter *after* pinning, exclusion, and year selection: a rule whose required feature is present but `False` is dropped. **Feature gating has two loci, and they produce different `Resolution` statuses:** @@ -569,7 +570,7 @@ The key invariant: `None`, `"default"`, and the default format string are **trea Example — Date input `"01/02/2026"` is recognized by both the US and European grammars and validated by both rules, yielding two distinct canonical values (`2026-01-02` and `2026-02-01`). The result is `AMBIGUOUS` regardless of `output_format`. `output_format="US"` merely renders those two values as `01/02/2026` and `02/01/2026`; it cannot and must not decide which interpretation is "correct". -> Note: the grammar→rule routing decision (which rule validates which recognized notation) is an entirely separate concern from `output_format`. Routing is declared on the rule (e.g. `Rule.target_grammars`); it operates in the recognition→validation stage and never touches formatting. Keep the two orthogonal. +> Note: the grammar→rule routing decision (which rule validates which recognized notation) is an entirely separate concern from `output_format`. Routing is declared on the rule (`Rule.target_semantics`) and matched against each grammar's `semantics`; it operates in the recognition→validation stage and never touches formatting. Keep the two orthogonal. Example wiring — inherited from `CapabilityContract`, you only set the class variables: @@ -1032,7 +1033,7 @@ If your rule needs to read a capability-specific parameter (like `two_digit_base A shipped capability is closed for modification but open for extension: you can add recognition and validation without touching the capability package. -1. **Author** a `Grammar` subclass (Step 4) and a `Rule` subclass (Step 5) for the capability's notation, exactly as you would for a new capability — the same contracts apply, including span-bearing `RecognitionMatch` output, `target_grammars`, and `requires_features`. +1. **Author** a `Grammar` subclass (Step 4) and a `Rule` subclass (Step 5) for the capability's notation, exactly as you would for a new capability — the same contracts apply, including span-bearing `RecognitionMatch` output, `semantics` on the grammar, `target_semantics` on the rule, and `requires_features`. 2. **Register** them before the first `canonicalize()` call: ```python @@ -1053,7 +1054,7 @@ Semantics to rely on: - The extension registries freeze with the capability registry — registration after the first pipeline run raises `CapabilityError`. - Opt-in only: an un-named registered grammar never affects results, keeping shipped behavior byte-identical for non-opt-in contracts. -- Community rules are opt-in too: a registered rule runs only when the contract names one of its `target_grammars` in `extra_grammars`; an un-opted rule — even one targeting a shipped grammar — never affects results. +- Community rules are opt-in too: a registered rule runs only when the contract's `extra_grammars` resolve to one of its `target_semantics`; an un-opted rule — even one targeting a shipped grammar's semantics — never affects results. - Unknown `extra_grammars` names are silently skipped; shipped names listed in `extra_grammars` are deduplicated. - Composition is guarded: a grammar name colliding with a shipped name, or an opted-in community rule naming a missing grammar, fails fast at pipeline start. @@ -1066,7 +1067,7 @@ Use this checklist to verify your capability is complete: - [ ] Notation is a frozen dataclass with `as_list()` method - [ ] Each grammar extends `Grammar[YourDomainNotation]` and implements `recognize(text) -> list[RecognitionMatch[YourDomainNotation]]` - [ ] Each rule extends `Rule[YourDomainNotation]` and implements `matches(notation, contract) -> bool` and `normalize(notation, contract) -> str` -- [ ] Each rule declares `target_grammars` (non-empty `frozenset[str]`) and `requires_features` (`frozenset()` when the rule always runs) +- [ ] Each rule declares `target_semantics` (non-empty `frozenset[str]`) and `requires_features` (`frozenset()` when the rule always runs) - [ ] Each rule file has a `PUBLICATION` provenance constant - [ ] Capability extends `Capability` and implements `get_grammars()` and `get_rules()` - [ ] Contract inherits `CapabilityContract` (frozen dataclass, no `slots=True`) and satisfies the `Contract` protocol diff --git a/HOW_TO_ADD_NEW_GRAMMAR.md b/HOW_TO_ADD_NEW_GRAMMAR.md index e4e3dad4..432efbcd 100644 --- a/HOW_TO_ADD_NEW_GRAMMAR.md +++ b/HOW_TO_ADD_NEW_GRAMMAR.md @@ -18,9 +18,10 @@ Before starting, understand these concepts (all defined in depth in HOW_TO_ADD_N - **Rule** — a validation unit that checks a notation against an authoritative specification and produces the canonical value with provenance. Semantics. - **Notation** — the intermediate token grammars produce and rules consume. - **`active_grammars`** — the *optional* contract property naming which grammars run. A contract that does not implement it runs every shipped grammar returned by `get_grammars()`; only the gated capabilities (Email, IP, ISBN) implement it to name a subset. -- **`target_grammars`** — the rule metadata declaring which grammar(s) a rule validates. A recognition only routes to rules that name its producing grammar. +- **`semantics`** — the grammar metadata declaring the *meaning* the grammar assigns to its recognized notations: an identity id by default, or a coalesced id shared with grammars that carry the same meaning (e.g. both the ISO and slash-ISO Date grammars declare `"iso8601_calendar_date"`). +- **`target_semantics`** — the rule metadata declaring which grammar semantics a rule validates. A recognition only routes to rules whose `target_semantics` includes its producing grammar's `semantics`. -**The one sentence that matters:** a new grammar changes behavior only when it is (1) returned by `get_grammars()`, and (2) targeted by at least one rule via `target_grammars` — plus (3) named in `active_grammars`, but **only for the gated capabilities (Email, IP, ISBN) that implement it**. For the other six capabilities, the contract has no `active_grammars` and the engine runs every shipped grammar, so `get_grammars()` alone activates the new grammar. Miss any condition and the grammar silently never runs — so a shipped grammar ships with a test that proves the difference (Step 5). +**The one sentence that matters:** a new grammar changes behavior only when it is (1) returned by `get_grammars()`, and (2) its `semantics` is claimed by at least one rule via `target_semantics` — plus (3) named in `active_grammars`, but **only for the gated capabilities (Email, IP, ISBN) that implement it**. For the other six capabilities, the contract has no `active_grammars` and the engine runs every shipped grammar, so `get_grammars()` alone activates the new grammar. Miss any condition and the grammar silently never runs — so a shipped grammar ships with a test that proves the difference (Step 5). --- @@ -30,7 +31,7 @@ Before writing code, answer these questions: 1. **What new representation are you recognizing?** Write down examples of real human input, including edge cases (`2024/1/5`, not just `2024/01/01`). 2. **Does an existing notation already fit it?** If the representation decomposes into the same fields as an existing grammar (e.g., Date's `DateNotation(N1, N2, N3)`), reuse it. Only extend the notation when the new format genuinely carries different components. -3. **Does an existing rule already assign the same meaning?** If the new format means the same thing and normalizes the same way as an already-validated format (e.g., `2024/01/01` *is* ISO 8601's calendar date), you may only need to extend an existing rule's `target_grammars` (Step 4, option A). +3. **Does an existing rule already assign the same meaning?** If the new format means the same thing and normalizes the same way as an already-validated format (e.g., `2024/01/01` *is* ISO 8601's calendar date), declare that meaning's shipped `semantics` id on the new grammar and stop — no rule edit (Step 4, option A). 4. **Which recognition strategy fits the representation?** Every grammar follows one of two core strategies (see HOW_TO_ADD_NEW_CAPABILITY.md Step 4 for the extended set): @@ -90,7 +91,7 @@ Create `paxman/capabilities//grammar/_recognition.py`: 1. Import `Grammar` and `RecognitionMatch` from `paxman.core.domain`, and the capability's notation. 2. Compile the regex once at module scope — never inside `recognize()` (it runs for every input). -3. Define a class extending `Grammar[Notation]` with a `name` of the form `{format}_recognition` (snake_case, unique within the capability). +3. Define a class extending `Grammar[Notation]` with a `name` of the form `{format}_recognition` (snake_case, unique within the capability) and a non-empty `semantics` string declaring the meaning its notations carry. 4. Implement `recognize(text) -> list[RecognitionMatch[Notation]]` — one `RecognitionMatch` per `finditer()` hit, carrying the notation, the half-open `[start, end)` span, and `raw_text`. ```python @@ -113,6 +114,7 @@ class SlashISODateGrammar(Grammar[DateNotation]): """Slash-delimited ISO date recognition: YYYY/MM/DD.""" name = "slash_iso_recognition" + semantics = "iso8601_calendar_date" # same meaning as the dash ISO grammar (Step 4) def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: """Extract YYYY/MM/DD date patterns from text.""" @@ -171,35 +173,27 @@ Recognition order (and the same-span tiebreak) follows the runnable set — `get ## Step 4: Make a Rule Validate It -A recognition only becomes a candidate when a rule names its grammar in `target_grammars`. Two options: +A recognition only becomes a candidate when a rule's `target_semantics` includes the producing grammar's `semantics`. Whether you need to touch a rule at all depends on whether the new grammar's *meaning* is genuinely new. Two options: -### Option A — Extend an existing rule (same meaning, same normalization) +### Option A — Reuse an existing rule's meaning (declare the shipped semantics id and stop) -When the new representation means the same thing and normalizes to the same canonical form as a format an existing rule already validates (the slash-ISO date *is* ISO 8601's calendar date), extend that rule's `target_grammars`. This is the minimal change: `matches()` and `normalize()` already handle the notation. +When the new representation means the same thing and normalizes to the same canonical form as a format an existing rule already validates (the slash-ISO date *is* ISO 8601's calendar date), declare the *shipped* `semantics` id on the new grammar — and **stop**. No rule edit and no new rule: the existing rule's `target_semantics` already includes that id, so its `matches()` and `normalize()` — which already handle the notation — validate the new grammar's recognitions unchanged. This is the minimal change: ```python -class Section431CalendarDate(Rule[DateNotation]): - """ISO 8601 Section 4.3.1 — Calendar date. - - Validates both the dash-delimited ISO grammar and the slash-delimited - variant (which shares the same position mapping and canonical form). - """ +class SlashISODateGrammar(Grammar[DateNotation]): + """Slash-delimited ISO date recognition: YYYY/MM/DD.""" - name = "Section 4.3.1-calendar-date" - strategy = RuleStrategy.PARSER - provenance = PUBLICATION - citation = "Section 4.3.1 (calendar date)" - target_grammars = frozenset( - {"iso8601_recognition", "slash_iso_recognition"} # extended - ) - requires_features = frozenset() + name = "slash_iso_recognition" + semantics = "iso8601_calendar_date" # the shipped id — same meaning as the dash ISO grammar ``` +Same-meaning grammars **share** a semantics id (a *coalesced* id). The shipped Date rule `Section431CalendarDate` already declares `target_semantics = frozenset({"iso8601_calendar_date"})`, so the slash-ISO grammar joins the ISO grammar under that one id and nothing in `rules/` changes. + ### Option B — Add a new rule (new meaning, different normalization, or different authority) -When the new format needs its own validation or provenance, add a rule file — one file per publication, one class per spec section (see HOW_TO_ADD_NEW_CAPABILITY.md Step 5 for the full rule template). `Rule.__init_subclass__` enforces the six metadata attributes (`name`, `strategy`, `provenance`, `citation`, `target_grammars`, `requires_features`) at class-definition time, and `target_grammars` must be a non-empty `frozenset[str]`. +When the new format means something genuinely new, give the grammar its own identity `semantics` id and add a rule file whose `target_semantics` names it — one file per publication, one class per spec section (see HOW_TO_ADD_NEW_CAPABILITY.md Step 5 for the full rule template). `Rule.__init_subclass__` enforces the six metadata attributes (`name`, `strategy`, `provenance`, `citation`, `target_semantics`, `requires_features`) at class-definition time, and `target_semantics` must be a non-empty `frozenset[str]`. -Whichever option you choose, the engine **fails fast** if you get it wrong: `_validate_affinity` raises `ContractError` when a rule's `target_grammars` names a grammar that is not in the composed set (shipped + opted-in community), so a dangling name can never silently disable a rule. +Whichever option you choose, the engine **fails fast** if you get it wrong: `_validate_affinity` raises `ContractError` when a rule's `target_semantics` names an id that no grammar claims in the composed set (shipped + opted-in community), so a dangling target can never silently disable a rule. --- @@ -281,10 +275,10 @@ This guide's every snippet is drawn from a real change: the **slash-ISO date gra | Step | File | Change | |------|------|--------| | 2a | `tests/capabilities/date/test_grammar.py` | `TestSlashISODateGrammar` (failing first) | -| 2b | `paxman/capabilities/Date/grammar/slash_iso_recognition.py` | New `SlashISODateGrammar` (`name = "slash_iso_recognition"`) | +| 2b | `paxman/capabilities/Date/grammar/slash_iso_recognition.py` | New `SlashISODateGrammar` (`name = "slash_iso_recognition"`, `semantics = "iso8601_calendar_date"`) | | 3 | `paxman/capabilities/Date/capability.py` | `SlashISODateGrammar()` appended to `get_grammars()` | | 3 | `paxman/capabilities/Date/contract.py` | No change — Date is all-active; the engine runs every `get_grammars()` entry | -| 4 | `paxman/capabilities/Date/rules/iso_8601_ed2019.py` | `Section431CalendarDate.target_grammars` extended to include the new grammar | +| 4 | `paxman/capabilities/Date/rules/iso_8601_ed2019.py` | No change — `Section431CalendarDate.target_semantics` already covers the shared `"iso8601_calendar_date"` semantics | | 5 | `tests/capabilities/date/test_capability.py` | Grammar count 3 → 4; new name wired | | 5 | `tests/integration/test_date_capability.py` | `"2024/01/01"` → `SUCCESS "2024-01-01"` (was `MISSING`) | | 6 | `README.md`, `CONTEXT.md` | Counts, format list, grammar table row | diff --git a/README.md b/README.md index 250daf44..c42fe720 100644 --- a/README.md +++ b/README.md @@ -425,6 +425,7 @@ class DotDateGrammar(Grammar[DateNotation]): """Recognize YYYY.MM.DD dates with dot separators.""" name = "dot_date_recognition" + semantics = "dot_date_recognition" _PATTERN = re.compile(r"\b(\d{4})\.(\d{2})\.(\d{2})\b") def recognize(self, text: str) -> list[RecognitionMatch[DateNotation]]: @@ -455,7 +456,7 @@ class DotDateRule(Rule[DateNotation]): publication_year=2019, ) citation = "Section 4.3.1 (calendar date)" - target_grammars = frozenset({"dot_date_recognition"}) + target_semantics = frozenset({"dot_date_recognition"}) requires_features = frozenset() def matches(self, notation: DateNotation, contract: Contract) -> bool: @@ -482,10 +483,10 @@ print(result.canonicalized_value) # → "2024-01-01" Rules of the seam: - **Register before the first `canonicalize()` call** — the extension registries freeze with the capability registry. -- **Opt-in only** — a registered grammar runs only when named in `extra_grammars` (available on every `create_contract` factory), and a registered rule runs only when the contract names one of its `target_grammars` there; un-named grammars and un-opted rules never affect results. -- **Unknown `extra_grammars` names are silently skipped**, so a contract naming an uninstalled grammar still runs byte-identically. +- **Opt-in only** — a registered grammar runs only when named in `extra_grammars` (available on every `create_contract` factory), and a registered rule runs only when the contract's `extra_grammars` resolve to one of its `target_semantics` ids; un-named grammars and un-opted rules never affect results. +- **Unknown `extra_grammars` names are silently skipped** for grammar activation, so a contract naming an uninstalled grammar still runs byte-identically; the unknown name is kept as-is as the semantics key for rule activation. - **Names must be unique in the composed set** — a community grammar colliding with a shipped name fails fast with `CapabilityError`. -- **Community rules declare `target_grammars`** and activate only when the contract names one of them in `extra_grammars`; an opted-in rule naming a missing grammar fails fast with `ContractError`. +- **Community rules declare `target_semantics`** and activate only when the contract's `extra_grammars` resolve to one of those ids; a rule opted in via an id that no grammar claims fails fast with `ContractError`, while a rule that is not opted in stays inert regardless of any dangling targets. --- diff --git a/capability_homogeneity_audit.md b/capability_homogeneity_audit.md index 4170d2f0..a763d613 100644 --- a/capability_homogeneity_audit.md +++ b/capability_homogeneity_audit.md @@ -60,11 +60,11 @@ validate it." Each capability invented its own self-filter: This contradicts `ARCHITECTURE.md:201` ("Each grammar's notation flows to its corresponding validation rule"). -**Unanimous ideal:** declare affinity on the rule — `Rule.target_grammars: +**Unanimous ideal:** declare affinity on the rule — `Rule.target_semantics: frozenset[str]` (ClassVar, enforced by the existing `__init_subclass__` metadata -check); orchestrator adds one line `if grammar_name not in rule.target_grammars: -continue`. Engine stays capability-agnostic (reads declared names, no shape -knowledge). Replay-safe *if* `_collect_candidates` also dedups identical +check); orchestrator adds one line `if semantics_by_name[grammar_name] not in +rule.target_semantics: continue`. Engine stays capability-agnostic (reads declared +semantics ids, no shape knowledge). Replay-safe *if* `_collect_candidates` also dedups identical `(value, recognition_rule, validation_rule)` tuples — otherwise Date candidate multiplicity changes the hash (semantics/status unchanged). Preserves Date ambiguity (each rule sees only its grammar's notation → 2 candidates → @@ -230,7 +230,7 @@ Ranked defects (from the rule-comparison agent): ## Unanimous ideals — recommended build order -1. **`Rule.target_grammars`** + one-line orchestrator filter (fixes F1; makes +1. **`Rule.target_semantics`** + one-line orchestrator filter (fixes F1; makes `ARCHITECTURE.md:201` true; deterministic output with candidate dedup). 2. **Move `include_*` feature-gating** to engine-enforced declared metadata, split by feature kind (fixes F2). **[DONE 2026-08-03]** — enforced via @@ -293,7 +293,7 @@ formatter seam described in the centralize-output-format addendum below. ### B. F1 candidate multiset is identical — byte-identical output requires ordered comparison The audit's *Watch out for* warns that moving to grammar→rule affinity could change -the candidate multiset via candidate multiplicity. In practice `target_grammars` was set equal +the candidate multiset via candidate multiplicity. In practice `target_semantics` was set equal to each rule's *effective acceptance domain* (the affinity map in F1), so the candidate multiset is identical to the cartesian product for every capability (Email / Date / Country / IP / Phone). That multiset equality alone does not prove byte-identical output: @@ -304,7 +304,7 @@ rules and provenance — as asserted by the repeated-run determinism tests (e.g. `ExecutionResult` equality). Where only multiset equality is established, the claim is limited to multiset equality, not byte-identity. The `_dedup_candidates` step is a pure safety net for future over-declaration, not a behavior change. The plan's Step 6.7 -determinism gate was satisfied *structurally* (target_grammars == effective domain ⇒ +determinism gate was satisfied *structurally* (target_semantics == effective domain ⇒ identical multiset) rather than by captured pre-change constants; the full 782-test suite passing is the empirical confirmation. diff --git a/paxman/capabilities/AGENTS.md b/paxman/capabilities/AGENTS.md index c2e7cdef..1cd221b4 100644 --- a/paxman/capabilities/AGENTS.md +++ b/paxman/capabilities/AGENTS.md @@ -39,8 +39,8 @@ Every capability must conform to the same structural surface. `CapabilityContrac - **Notation** — `@dataclass(frozen=True, slots=True)`; one `str` field per component; the sole type parameter of the capability's `Grammar[NotationT]` / `Rule[NotationT]`. - **Contract** — `@dataclass(frozen=True)` extending `CapabilityContract`, NO `slots=True` (incompatible with the base `super()` pattern). Sets `DEFAULT_OUTPUT_FORMAT` / `OFFERED_OUTPUT_FORMATS` class vars; `capability_name` via `field(init=False)`; inherits `output_format` (always optional; base `__post_init__` resolves it and validates offered alternatives); `active_grammars` is optional — only feature-gated capabilities (Email, IP, ISBN) override it, and the base `None` default runs every shipped grammar. -- **Grammar** — one file = one recognizer; `name` = `{format}_recognition` (snake_case, unique per capability); emits span-bearing `RecognitionMatch` (half-open `[start, end)`, `raw_text`); syntax-only — extraction and sanitization, never validation, dedup, ordering, or token→canonical mapping. Two sanctioned strategies: Regex (shape) and Lexicon (key-only tables, kept in `grammar/data/` apart from recognition logic); see HOW_TO for the extended set. -- **Rule** — one file = one publication (module-level `PUBLICATION` provenance constant); class = one spec section; declares `name` (`Section {X.Y.Z}-{description}`), `strategy`, `provenance`, `citation`, `target_grammars` (non-empty), `requires_features` — all six enforced at import time. Rule classes sharing one publication live in the same file; authority-backed lookup tables live in `rules/data/`, separated from rule logic. +- **Grammar** — one file = one recognizer; `name` = `{format}_recognition` (snake_case, unique per capability); `semantics` = non-empty string declaring the meaning the grammar's notations carry (identity id unless the format shares another grammar's meaning, in which case declare that coalesced id — shared meaning ⇒ shared id); emits span-bearing `RecognitionMatch` (half-open `[start, end)`, `raw_text`); syntax-only — extraction and sanitization, never validation, dedup, ordering, or token→canonical mapping. Two sanctioned strategies: Regex (shape) and Lexicon (key-only tables, kept in `grammar/data/` apart from recognition logic); see HOW_TO for the extended set. +- **Rule** — one file = one publication (module-level `PUBLICATION` provenance constant); class = one spec section; declares `name` (`Section {X.Y.Z}-{description}`), `strategy`, `provenance`, `citation`, `target_semantics` (non-empty), `requires_features` — all six enforced at import time. Rule classes sharing one publication live in the same file; authority-backed lookup tables live in `rules/data/`, separated from rule logic. - **Feature gating — two loci, two statuses** — input-shape features toggle grammars via `active_grammars` (disabled grammar → `MISSING`), implemented only by the gated capabilities (Email, IP, ISBN) — other contracts inherit the `None` default, which runs every shipped grammar; authority features gate rules via `requires_features` (dropped rule → `INVALID`). Never gate inside `matches()`; never cast to read `include_*` flags. `typing.cast` is only for validity-affecting parameters. - **Presentation-only invariant** — rules never reference `output_format` (CI-scanned); `normalize()` always returns the default canonical form; `format_value()` is the only presentation seam, overridden only when `OFFERED_OUTPUT_FORMATS` is non-empty; formatting adds no provenance; offered formats must preserve the capability's ambiguity contract. - **`create_contract()`** — static, keyword-only; fixed common block first (`excluded_rules`, `pinned_rules`, `year`, `output_format`), capability-specific params after. diff --git a/paxman/core/AGENTS.md b/paxman/core/AGENTS.md index 87239351..ae74a0e4 100644 --- a/paxman/core/AGENTS.md +++ b/paxman/core/AGENTS.md @@ -22,8 +22,8 @@ Core owns the domain vocabulary (pipeline value objects + `Rule`/`Grammar` ABCs) ## CONVENTIONS - **Layer discipline:** `paxman.core` must never import from `paxman.api`, `paxman.engine`, or `paxman.capabilities`. If a new core type needs something from outside, it does not belong here. - **Value objects** (`domain.py`): `@dataclass(frozen=True, slots=True)`. Spans are half-open with `len(raw_text) == end - start`, enforced in `__post_init__`. `GrammarRule` enforces lowercase names; `RecognizedRep.__hash__` handles unhashable list notations; `Candidate._provenance` is `init=False` and tuple-ized in `__init__`. -- **`Rule` subclasses** must declare `name`, `strategy`, `provenance`, `citation`, `target_grammars` (non-empty `frozenset[str]`), `requires_features` (`frozenset[str]`) as class attrs. `Rule.__init_subclass__` raises `TypeError` at import time for missing/mistyped metadata — keep it a hard import-time failure. `matches()`/`normalize()` never raise. -- **`Grammar` subclasses** must declare `name`; `recognize()` returns span-bearing `RecognitionMatch` only, never bare notation. +- **`Rule` subclasses** must declare `name`, `strategy`, `provenance`, `citation`, `target_semantics` (non-empty `frozenset[str]`), `requires_features` (`frozenset[str]`) as class attrs. `Rule.__init_subclass__` raises `TypeError` at import time for missing/mistyped metadata — keep it a hard import-time failure. `matches()`/`normalize()` never raise. +- **`Grammar` subclasses** must declare `name` and `semantics` — a non-empty string, the meaning id the grammar's notations carry (identity id by default; same-meaning grammars share one coalesced id). `Grammar.__init_subclass__` raises `TypeError` at import time for a missing, non-string, or empty `semantics`. `recognize()` returns span-bearing `RecognitionMatch` only, never bare notation. - **Contracts:** subclass `CapabilityContract`, never `Contract` directly. Set `DEFAULT_OUTPUT_FORMAT`/`OFFERED_OUTPUT_FORMATS`, set `capability_name` via `field(default=..., init=False)`. `active_grammars` is optional: it returns `None` by default (the engine then runs every shipped `get_grammars()` entry); only feature-gated capabilities (Email, IP, ISBN) override it. - **`output_format`** is always optional: `None`/`"default"`/`DEFAULT_OUTPUT_FORMAT` resolve to the default, offered formats resolve to themselves, anything else raises `ContractError`. Resolved once in `CapabilityContract.__post_init__`; subclasses with their own `__post_init__` call `super().__post_init__()` first. Note: `resolve_output_format` is imported lazily there to break the `capability_contract` ↔ `contract` import cycle. - **`pinned_rules` wins over `excluded_rules`** (non-`None` pins; empty tuple pins to nothing); `year` filtering still applies after pinning. diff --git a/paxman/engine/orchestrator.py b/paxman/engine/orchestrator.py index cd9ca769..000c7ba0 100644 --- a/paxman/engine/orchestrator.py +++ b/paxman/engine/orchestrator.py @@ -306,10 +306,14 @@ def _collect_candidates( """Match recognitions against rules and collect candidates. Routes each recognition only to rules whose ``target_semantics`` includes - the producing grammar's semantics (ARCHITECTURE.md:201), formats each - validated value through the capability's ``format_value()`` seam, then - dedups identical candidate tuples so the candidate multiset is stable - regardless of routing. + the producing grammar's semantics, formats each validated value through + the capability's ``format_value()`` seam, then dedups identical candidate + tuples so the candidate multiset is stable regardless of routing. + + The ``semantics_by_name[grammar_name]`` lookup cannot KeyError: + recognitions are produced only by grammars in the composed ``all_grammars`` + (``_recognize`` filters against ``supported_names``), the same list the map + is built from. """ candidates: list[Candidate] = [] for recognition in recognitions: From 125c06dbd35934fedc0a701f27610650803fce3f Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 21:10:47 +0800 Subject: [PATCH 12/22] docs: ruff-format code blocks in swept documentation --- CONTEXT.md | 5 ++++- HOW_TO_ADD_NEW_GRAMMAR.md | 4 +++- 2 files changed, 7 insertions(+), 2 deletions(-) diff --git a/CONTEXT.md b/CONTEXT.md index 1d1ec887..4cba1350 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -173,13 +173,16 @@ The resolver **consumes notation** and outputs a canonical_value (not notation). ```python # capabilities/Currency/rules/iso_4217_ed2015.py + class SectionCode(Rule[CurrencyNotation]): """ISO 4217 Section 3 - Currency and funds codes""" name = "Section 3-code" strategy = RuleStrategy.LOOKUP_TABLE provenance = PUBLICATION - target_semantics = frozenset({"code_recognition", "symbol_recognition", "word_recognition"}) + target_semantics = frozenset( + {"code_recognition", "symbol_recognition", "word_recognition"} + ) requires_features = frozenset() def matches(self, notation: CurrencyNotation, contract: Contract) -> bool: diff --git a/HOW_TO_ADD_NEW_GRAMMAR.md b/HOW_TO_ADD_NEW_GRAMMAR.md index 432efbcd..faa9015f 100644 --- a/HOW_TO_ADD_NEW_GRAMMAR.md +++ b/HOW_TO_ADD_NEW_GRAMMAR.md @@ -184,7 +184,9 @@ class SlashISODateGrammar(Grammar[DateNotation]): """Slash-delimited ISO date recognition: YYYY/MM/DD.""" name = "slash_iso_recognition" - semantics = "iso8601_calendar_date" # the shipped id — same meaning as the dash ISO grammar + semantics = ( + "iso8601_calendar_date" # the shipped id — same meaning as the dash ISO grammar + ) ``` Same-meaning grammars **share** a semantics id (a *coalesced* id). The shipped Date rule `Section431CalendarDate` already declares `target_semantics = frozenset({"iso8601_calendar_date"})`, so the slash-ISO grammar joins the ISO grammar under that one id and nothing in `rules/` changes. From 42fd19e937f75330a47b29765743fbaf90bdce18 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 21:38:54 +0800 Subject: [PATCH 13/22] test: tighten same-semantics guard coverage (oracle review) - Per-capability semantics grouping matching the engine's routing scope (cross-capability id reuse no longer mixes members, and seeded groups must stay single-capability) - Members must each recognize >=1 probe and every probe must be matched by >=1 member; dead members/probes fail loudly instead of passing via the previous bare 'continue' - Assert all matches agree with the expected notation, not just matches[0] - Drive each group's shared rule with its own contract class (Date/ Email/Phone) instead of a hardcoded DateContract - Correct the allowlist comment: coalesced ids vs renamed singletons --- .../test_grammar_semantics_consistency.py | 117 ++++++++++++++---- tests/unit/test_grammar_semantics_metadata.py | 9 +- 2 files changed, 100 insertions(+), 26 deletions(-) diff --git a/tests/unit/test_grammar_semantics_consistency.py b/tests/unit/test_grammar_semantics_consistency.py index 3c162546..6066c2ee 100644 --- a/tests/unit/test_grammar_semantics_consistency.py +++ b/tests/unit/test_grammar_semantics_consistency.py @@ -7,9 +7,13 @@ members map the same input to different notation fields (or whose shared rule canonicalizes differently) would resolve the same text differently depending on which member happened to recognize it — silent nondeterminism. This guard -enumerates all shipped grammar classes, groups them by ``semantics``, and for -every seeded group runs shared probe rows through each member's -``recognize()`` asserting identical notation fields and canonical values. +enumerates all shipped grammar classes, groups them by ``semantics`` within +each capability, and pins every seeded group's members to a single canonical +mapping: each member must recognize at least one probe row into the group's +expected notation, and the group's shared rule must canonicalize every match +identically. Group members may recognize disjoint input sets (e.g. the dash +and slash ISO grammars), so agreement is asserted member-vs-table — no +cross-member comparison is claimed. """ from __future__ import annotations @@ -32,10 +36,13 @@ from paxman.capabilities.Date.contract import DateContract from paxman.capabilities.Date.notation import DateNotation from paxman.capabilities.Date.rules.iso_8601_ed2019 import Section431CalendarDate +from paxman.capabilities.Email.contract import EmailContract from paxman.capabilities.Email.notation import EmailNotation from paxman.capabilities.Email.rules.rfc_5322_ed2008 import Section341AddrSpec +from paxman.capabilities.Phone.contract import PhoneContract from paxman.capabilities.Phone.notation import PhoneNotation from paxman.capabilities.Phone.rules.e164_ed2010 import Section6_1InternationalNumber +from paxman.core.capability_contract import CapabilityContract from paxman.core.domain import Grammar, Rule _SHIPPED_CAPABILITIES = [Country, Currency, Date, Email, IP, ISBN, Money, Phone, URL] @@ -54,6 +61,15 @@ "isbn10_recognition", ) +# Contract class used to drive each seeded group's shared rule ``normalize()``. +# The right contract per group keeps the guard honest if a rule ever starts +# reading contract fields. +_CONTRACTS: dict[str, type[CapabilityContract]] = { + "iso8601_calendar_date": DateContract, + "rfc5322_addr_spec": EmailContract, + "e164_international": PhoneContract, +} + class _ProbeRow(NamedTuple): """One input run through every member of a semantics group. @@ -124,42 +140,84 @@ class _ProbeRow(NamedTuple): } -def _group_shipped_grammars_by_semantics() -> dict[str, list[type[Grammar[Any]]]]: - """Group every shipped grammar class by its ``semantics`` id.""" - groups: dict[str, list[type[Grammar[Any]]]] = {} +def _group_shipped_grammars_by_capability_semantics() -> dict[ + str, dict[str, list[type[Grammar[Any]]]] +]: + """Group shipped grammar classes per capability by their ``semantics`` id. + + The affinity-routing engine treats grammars as interchangeable only within + one capability, so groups are scoped per capability: a semantics id reused + across capabilities (Currency and Money both declaring ``code_recognition`` + etc.) yields separate per-capability groups that never co-route and must + not be probed as one unit. + """ + groups: dict[str, dict[str, list[type[Grammar[Any]]]]] = {} for capability in _SHIPPED_CAPABILITIES: + per_capability: dict[str, list[type[Grammar[Any]]]] = {} for grammar in capability().get_grammars(): - groups.setdefault(grammar.semantics, []).append(type(grammar)) + per_capability.setdefault(grammar.semantics, []).append(type(grammar)) + groups[capability.__name__] = per_capability return groups @pytest.mark.unit def test_probe_keys_name_real_semantics_groups() -> None: """Every probe-table key must be a real semantics group in the enumeration.""" - groups = _group_shipped_grammars_by_semantics() - assert set(_PROBE_ROWS) <= set(groups) + groups = _group_shipped_grammars_by_capability_semantics() + assert all( + any(key in per_capability for per_capability in groups.values()) + for key in _PROBE_ROWS + ) @pytest.mark.unit def test_same_semantics_grammars_agree_on_notation_and_canonical() -> None: - """Members of a seeded semantics group recognize probes identically. + """Members of a seeded semantics group pin the group's canonical mapping. - Groups without probe rows are skipped — the reverse coverage (that every - multi-member group is seeded) lands in a later task. + Every member must recognize at least one probe row into the group's + expected notation, and the group's shared rule must canonicalize every + match to the expected canonical value. Because members may recognize + disjoint input sets, agreement is asserted member-vs-table (against the + group's single expected mapping), not by comparing members on one input. + A member that recognizes none of the probes — or a probe that no member + recognizes — fails loudly instead of passing silently. """ - groups = _group_shipped_grammars_by_semantics() + groups = _group_shipped_grammars_by_capability_semantics() for semantics, (rule_cls, probes) in _PROBE_ROWS.items(): rule = rule_cls() - for member_cls in groups.get(semantics, ()): + member_lists = [ + per_capability.get(semantics, ()) for per_capability in groups.values() + ] + assert sum(1 for members in member_lists if members) == 1, ( + f"semantics {semantics!r} spans multiple capabilities; probe rows " + "must stay scoped to one capability" + ) + members = [member for group_members in member_lists for member in group_members] + for member_cls in members: member = member_cls() + matched_any = False for probe in probes: matches = member.recognize(probe.input) if not matches: continue - assert matches[0].notation == probe.expected_notation - assert rule.normalize(matches[0].notation, DateContract()) == ( - probe.expected_canonical + matched_any = True + assert all(m.notation == probe.expected_notation for m in matches), ( + f"{member_cls.__name__} mapped {probe.input!r} to " + f"{[m.notation for m in matches]}, expected " + f"{probe.expected_notation!r}" ) + assert ( + rule.normalize(matches[0].notation, _CONTRACTS[semantics]()) + == probe.expected_canonical + ) + assert matched_any, ( + f"{member_cls.__name__} (semantics {semantics!r}) recognized none " + "of the probe rows" + ) + for probe in probes: + assert any(member_cls().recognize(probe.input) for member_cls in members), ( + f"probe {probe.input!r} matched by no member of {semantics!r}" + ) @pytest.mark.unit @@ -170,12 +228,21 @@ def test_every_shipped_grammar_belongs_to_one_semantics_group() -> None: group with a non-empty semantics id; a dropped or double-counted grammar would break the member-count equality. """ - groups = _group_shipped_grammars_by_semantics() + groups = _group_shipped_grammars_by_capability_semantics() shipped_count = sum( len(capability().get_grammars()) for capability in _SHIPPED_CAPABILITIES ) - assert sum(len(members) for members in groups.values()) == shipped_count - assert all(semantics for semantics in groups) + assert ( + sum( + len(members) + for per_capability in groups.values() + for members in per_capability.values() + ) + == shipped_count + ) + assert all( + semantics for per_capability in groups.values() for semantics in per_capability + ) @pytest.mark.unit @@ -207,8 +274,12 @@ def test_d7_no_coalesce_semantics_groups_stay_singleton() -> None: ``us_calendar_date``/``european_calendar_date`` are renamed singletons and the other six are identity singletons; coalescing any of them would change - what the shared semantics resolves to. + what the shared semantics resolves to. Each id must total exactly one + member across all capabilities. """ - groups = _group_shipped_grammars_by_semantics() + groups = _group_shipped_grammars_by_capability_semantics() for semantics in _NO_COALESCE_SEMANTICS: - assert len(groups[semantics]) == 1 + total = sum( + len(per_capability.get(semantics, ())) for per_capability in groups.values() + ) + assert total == 1, f"{semantics!r} must stay a singleton, found {total}" diff --git a/tests/unit/test_grammar_semantics_metadata.py b/tests/unit/test_grammar_semantics_metadata.py index 7e20e3f1..4b414367 100644 --- a/tests/unit/test_grammar_semantics_metadata.py +++ b/tests/unit/test_grammar_semantics_metadata.py @@ -19,9 +19,12 @@ ) from paxman.core.domain import Grammar, RecognitionMatch -# Semantics ids that intentionally coalesce several grammars onto one shared -# id (semantic affinity routing, ADR-0003): a grammar in this set legitimately -# declares ``semantics`` differing from its ``name``. +# Semantics ids that legitimately differ from a grammar's ``name`` (semantic +# affinity routing, ADR-0003): coalesced ids shared by several grammars +# (``iso8601_calendar_date``, ``rfc5322_addr_spec``, ``e164_international``) +# and renamed singletons (``us_calendar_date``, ``european_calendar_date``). +# A grammar in this set declares ``semantics`` differing from its ``name`` +# without failing the identity check. _COALESCED_SEMANTICS: frozenset[str] = frozenset( { "iso8601_calendar_date", From 73b04f4de779cb02658ead2964684caa7599e081 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 21:38:54 +0800 Subject: [PATCH 14/22] docs: clarify semantics-id opt-in in extra_grammars (oracle review) --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index c42fe720..5f18521b 100644 --- a/README.md +++ b/README.md @@ -484,7 +484,7 @@ Rules of the seam: - **Register before the first `canonicalize()` call** — the extension registries freeze with the capability registry. - **Opt-in only** — a registered grammar runs only when named in `extra_grammars` (available on every `create_contract` factory), and a registered rule runs only when the contract's `extra_grammars` resolve to one of its `target_semantics` ids; un-named grammars and un-opted rules never affect results. -- **Unknown `extra_grammars` names are silently skipped** for grammar activation, so a contract naming an uninstalled grammar still runs byte-identically; the unknown name is kept as-is as the semantics key for rule activation. +- **Unknown `extra_grammars` names are silently skipped** for grammar activation, so a contract naming an uninstalled grammar still runs byte-identically; the unknown name is kept as-is as the semantics key for rule activation. A name that is not a grammar name but matches a known semantics id therefore activates that semantics's rules without opting in a grammar — those rules fire only on recognitions carrying that semantics (fail-fast `ContractError` applies only to ids no grammar claims). - **Names must be unique in the composed set** — a community grammar colliding with a shipped name fails fast with `CapabilityError`. - **Community rules declare `target_semantics`** and activate only when the contract's `extra_grammars` resolve to one of those ids; a rule opted in via an id that no grammar claims fails fast with `ContractError`, while a rule that is not opted in stays inert regardless of any dangling targets. From 9ce5fdb4cb98361b04c86d8988e2b23e2c095cf4 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 21:58:59 +0800 Subject: [PATCH 15/22] fix: bound slash-ISO pattern, clarify activation docs (coderabbit review) --- HOW_TO_ADD_NEW_GRAMMAR.md | 6 +++--- .../Date/grammar/slash_iso_recognition.py | 7 +++++-- tests/capabilities/date/test_grammar.py | 11 +++++++++++ 3 files changed, 19 insertions(+), 5 deletions(-) diff --git a/HOW_TO_ADD_NEW_GRAMMAR.md b/HOW_TO_ADD_NEW_GRAMMAR.md index faa9015f..173d7e56 100644 --- a/HOW_TO_ADD_NEW_GRAMMAR.md +++ b/HOW_TO_ADD_NEW_GRAMMAR.md @@ -21,7 +21,7 @@ Before starting, understand these concepts (all defined in depth in HOW_TO_ADD_N - **`semantics`** — the grammar metadata declaring the *meaning* the grammar assigns to its recognized notations: an identity id by default, or a coalesced id shared with grammars that carry the same meaning (e.g. both the ISO and slash-ISO Date grammars declare `"iso8601_calendar_date"`). - **`target_semantics`** — the rule metadata declaring which grammar semantics a rule validates. A recognition only routes to rules whose `target_semantics` includes its producing grammar's `semantics`. -**The one sentence that matters:** a new grammar changes behavior only when it is (1) returned by `get_grammars()`, and (2) its `semantics` is claimed by at least one rule via `target_semantics` — plus (3) named in `active_grammars`, but **only for the gated capabilities (Email, IP, ISBN) that implement it**. For the other six capabilities, the contract has no `active_grammars` and the engine runs every shipped grammar, so `get_grammars()` alone activates the new grammar. Miss any condition and the grammar silently never runs — so a shipped grammar ships with a test that proves the difference (Step 5). +**The one sentence that matters:** a new grammar changes behavior only when it is (1) returned by `get_grammars()`, and (2) its `semantics` is claimed by at least one rule via `target_semantics` — plus (3) named in `active_grammars`, but **only for the gated capabilities (Email, IP, ISBN) that implement it**. For the other six capabilities, the contract has no `active_grammars` and the engine runs every shipped grammar, so `get_grammars()` alone activates the new grammar. Miss condition (1) or (3) and the grammar never runs — input matching only it stays `MISSING`. Miss condition (2) and the grammar still runs, but its recognitions route to no rules, so input matching only it becomes `INVALID` (recognized, no authority rule validates) instead of resolving — never a candidate. Either way the resolved output is unchanged, which is why a shipped grammar ships with a test that proves the difference (Step 5). --- @@ -107,7 +107,7 @@ import re from paxman.capabilities.Date.notation import DateNotation from paxman.core.domain import Grammar, RecognitionMatch -_SLASH_ISO_PATTERN = re.compile(r"(\d{4})/(\d{1,2})/(\d{1,2})") +_SLASH_ISO_PATTERN = re.compile(r"(? None: assert grammar.recognize("07/26/2026") == [] assert grammar.recognize("26/07/2026") == [] + def test_does_not_match_embedded_in_digits(self) -> None: + """A date run glued to surrounding digits is not recognized. + + The digit lookarounds prevent partial matches inside longer digit + runs (e.g. IDs), mirroring the 2-digit US/European patterns. + """ + grammar = SlashISODateGrammar() + assert grammar.recognize("12026/07/26") == [] + assert grammar.recognize("2026/07/261") == [] + assert grammar.recognize("12026/07/261") == [] + def test_grammar_name(self) -> None: grammar = SlashISODateGrammar() assert grammar.name == "slash_iso_recognition" From 78794cc8f7ae831419bfafd339fc63442b317c98 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 23:18:31 +0800 Subject: [PATCH 16/22] fix: bound remaining date grammars against embedded digits --- .../Date/grammar/european_recognition.py | 6 +++- .../Date/grammar/iso8601_recognition.py | 6 +++- .../Date/grammar/us_recognition.py | 6 +++- tests/capabilities/date/test_grammar.py | 35 +++++++++++++++++++ 4 files changed, 50 insertions(+), 3 deletions(-) diff --git a/paxman/capabilities/Date/grammar/european_recognition.py b/paxman/capabilities/Date/grammar/european_recognition.py index 6ef5855c..b38f3186 100644 --- a/paxman/capabilities/Date/grammar/european_recognition.py +++ b/paxman/capabilities/Date/grammar/european_recognition.py @@ -10,13 +10,17 @@ from paxman.capabilities.Date.notation import DateNotation from paxman.core.domain import Grammar, RecognitionMatch -_EUROPEAN_DATE_PATTERN_4DIGIT = re.compile(r"(\d{1,2})/(\d{1,2})/(\d{4})") +_EUROPEAN_DATE_PATTERN_4DIGIT = re.compile(r"(? None: result = grammar.recognize("No dates here") assert result == [] + def test_does_not_match_embedded_in_digits(self) -> None: + """A date glued to surrounding digits is not recognized. + + The digit lookarounds prevent partial matches inside longer digit + runs (e.g. IDs). + """ + grammar = ISO8601DateGrammar() + assert grammar.recognize("12026-07-26") == [] + assert grammar.recognize("2026-07-261") == [] + assert grammar.recognize("12026-07-261") == [] + def test_grammar_name(self) -> None: grammar = ISO8601DateGrammar() assert grammar.name == "iso8601_recognition" @@ -88,6 +99,18 @@ def test_grammar_name(self) -> None: grammar = USDateGrammar() assert grammar.name == "us_recognition" + def test_does_not_match_embedded_in_digits(self) -> None: + """A date glued to surrounding digits is not recognized. + + Both year-length variants carry digit lookarounds, preventing + partial matches inside longer digit runs (e.g. IDs). + """ + grammar = USDateGrammar() + assert grammar.recognize("1207/26/2026") == [] + assert grammar.recognize("07/26/20261") == [] + assert grammar.recognize("1207/26/26") == [] + assert grammar.recognize("07/26/261") == [] + def test_emits_spans(self) -> None: result = self.grammar.recognize("x 07/26/2026 y") assert len(result) == 1 @@ -126,6 +149,18 @@ def test_grammar_name(self) -> None: grammar = EuropeanDateGrammar() assert grammar.name == "european_recognition" + def test_does_not_match_embedded_in_digits(self) -> None: + """A date glued to surrounding digits is not recognized. + + Both year-length variants carry digit lookarounds, preventing + partial matches inside longer digit runs (e.g. IDs). + """ + grammar = EuropeanDateGrammar() + assert grammar.recognize("1226/07/2026") == [] + assert grammar.recognize("26/07/20261") == [] + assert grammar.recognize("1226/07/26") == [] + assert grammar.recognize("26/07/261") == [] + def test_emits_spans(self) -> None: result = self.grammar.recognize("x 26/07/2026 y") assert len(result) == 1 From 634aef8fda72bb8c17c674901674508be254f090 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 23:18:45 +0800 Subject: [PATCH 17/22] docs: sync slash-ISO docstring with bounded date grammar set --- paxman/capabilities/Date/grammar/slash_iso_recognition.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/paxman/capabilities/Date/grammar/slash_iso_recognition.py b/paxman/capabilities/Date/grammar/slash_iso_recognition.py index f37a1ae5..5746ead1 100644 --- a/paxman/capabilities/Date/grammar/slash_iso_recognition.py +++ b/paxman/capabilities/Date/grammar/slash_iso_recognition.py @@ -21,7 +21,7 @@ class SlashISODateGrammar(Grammar[DateNotation]): accepted and zero-padded by the validating rule. The leading 4-digit year keeps the pattern disjoint from the US and European grammars, which both require a leading month/day. Digit lookarounds keep the match disjoint - from surrounding digits, mirroring the 2-digit US/European patterns, so a + from surrounding digits, mirroring the other shipped date grammars, so a longer digit run (e.g. an ID like ``12026/01/15``) is never partially matched as a date. From 22d461ee6bc53a7bbc4578e981e720be00d7056d Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 23:54:33 +0800 Subject: [PATCH 18/22] docs(adr): correct target_grammars inventory to verified 54 files --- docs/adr/0003-semantic-affinity-routing.md | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/adr/0003-semantic-affinity-routing.md b/docs/adr/0003-semantic-affinity-routing.md index d65afa72..8cd9c0a0 100644 --- a/docs/adr/0003-semantic-affinity-routing.md +++ b/docs/adr/0003-semantic-affinity-routing.md @@ -202,10 +202,10 @@ is rule territory by design. never widen the set of meanings a rule validates. Mitigation: each coalescing step runs the per-capability pipeline tests; the migration lands one capability at a time. -- **Incomplete docs sweep.** 55 files reference `target_grammars` (24 rule - files, 3 engine sites, 1 domain ABC, extension registries docstring, test - suites, and 6 documentation surfaces). Mitigation: this ADR lands before - code; the sweep is enumerated in Migration #4. +- **Incomplete docs sweep.** 54 files referenced `target_grammars` at plan + time (pre-migration inventory; the plan's D10 re-count is authoritative — + superseded by the completed Migration #4 sweep). Mitigation: this ADR lands + before code; the sweep is enumerated in Migration #4. ## Alternatives Considered From e02c96dcdc42c66ef200ae72e23b6e61d16df81d Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Tue, 11 Aug 2026 23:54:33 +0800 Subject: [PATCH 19/22] test: guard grammar-semantics rule coverage, fix dangling-target docstring --- tests/integration/test_grammar_extensions.py | 3 ++- .../test_grammar_semantics_consistency.py | 21 +++++++++++++++++++ 2 files changed, 23 insertions(+), 1 deletion(-) diff --git a/tests/integration/test_grammar_extensions.py b/tests/integration/test_grammar_extensions.py index 81781334..5548f67a 100644 --- a/tests/integration/test_grammar_extensions.py +++ b/tests/integration/test_grammar_extensions.py @@ -149,7 +149,8 @@ def normalize(self, notation: DateNotation, contract: Contract) -> str: class DanglingDateRule(Rule[DateNotation]): - """Community rule whose target_semantics references a missing grammar.""" + """Community rule whose target_semantics references a semantics id that + no grammar claims.""" name = "dangling_date_rule" strategy = RuleStrategy.PARSER diff --git a/tests/unit/test_grammar_semantics_consistency.py b/tests/unit/test_grammar_semantics_consistency.py index 6066c2ee..1c300138 100644 --- a/tests/unit/test_grammar_semantics_consistency.py +++ b/tests/unit/test_grammar_semantics_consistency.py @@ -245,6 +245,27 @@ def test_every_shipped_grammar_belongs_to_one_semantics_group() -> None: ) +@pytest.mark.unit +def test_every_grammar_semantics_claimed_by_rule_target() -> None: + """Every shipped grammar's semantics is claimed by an in-capability rule. + + A grammar whose semantics no rule declares routes every recognition to + zero rules — input matching only it yields INVALID instead of resolving, + silently. Requiring in-capability rule-target coverage keeps + ``_collect_candidates()`` free of unroutable shipped grammars; a grammar + added without a claiming rule fails here at test time. + """ + for capability in _SHIPPED_CAPABILITIES: + instance = capability() + targets = {s for rule in instance.get_rules() for s in rule.target_semantics} + for grammar in instance.get_grammars(): + assert grammar.semantics in targets, ( + f"{capability.__name__} grammar {grammar.name!r} declares " + f"semantics {grammar.semantics!r} claimed by no shipped rule " + f"(rule targets: {sorted(targets)})" + ) + + @pytest.mark.unit def test_every_multi_member_semantics_group_has_probe_rows() -> None: """A coalesced group must be seeded or the guard fails loudly. From 98f4255df799c111594096ca9f90c9a233708dbe Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Wed, 12 Aug 2026 00:40:08 +0800 Subject: [PATCH 20/22] test: pin renamed singleton semantics ids, cover raw semantics-id opt-in Oracle review follow-up (ADR-0003): - test_grammar_semantics_metadata: assert us_recognition -> us_calendar_date and european_recognition -> european_calendar_date, closing the cross-swap hole in the identity allowlist (D7 guard now pins name->id mapping) - test_grammar_extensions: extra_grammars=("iso8601_calendar_date",) activates a rule targeting that semantics id without opting in any community grammar, locking the README-documented semantics_by_name.get(n, n) fallback --- tests/integration/test_grammar_extensions.py | 24 +++++++++++++++++++ tests/unit/test_grammar_semantics_metadata.py | 15 ++++++++++++ 2 files changed, 39 insertions(+) diff --git a/tests/integration/test_grammar_extensions.py b/tests/integration/test_grammar_extensions.py index 5548f67a..2e3bd35b 100644 --- a/tests/integration/test_grammar_extensions.py +++ b/tests/integration/test_grammar_extensions.py @@ -312,3 +312,27 @@ def test_community_rule_on_shipped_grammar_requires_activation(self) -> None: c.validation_rule == "community_iso8601_rule" for c in activated_result.candidates ) + + @pytest.mark.integration + def test_rule_opt_in_via_raw_semantics_id_without_grammar(self) -> None: + """A raw semantics id in ``extra_grammars`` activates rules targeting it + without opting in any community grammar. + + Locks the README-documented raw-name fallback + (``semantics_by_name.get(n, n)``): ``iso8601_calendar_date`` is a + known semantics id but not a grammar name, so the community rule + targeting it fires on shipped ISO recognitions while no community + grammar is activated (fail-fast ``ContractError`` applies only to ids + no grammar claims). + """ + register_rule("date", CommunityISO8601Rule) + + result = run_capability( + "2026-01-15", DateContract(extra_grammars=("iso8601_calendar_date",)) + ) + assert any( + c.validation_rule == "community_iso8601_rule" for c in result.candidates + ) + assert all( + c.recognition_rule == "iso8601_recognition" for c in result.candidates + ) diff --git a/tests/unit/test_grammar_semantics_metadata.py b/tests/unit/test_grammar_semantics_metadata.py index 4b414367..0f654093 100644 --- a/tests/unit/test_grammar_semantics_metadata.py +++ b/tests/unit/test_grammar_semantics_metadata.py @@ -52,6 +52,21 @@ def test_shipped_grammars_declare_semantics_identity(self) -> None: or grammar.semantics in _COALESCED_SEMANTICS ) + @pytest.mark.unit + def test_renamed_singletons_pin_exact_semantics_ids(self) -> None: + """The renamed singleton grammars pin their exact ``semantics`` ids. + + The allowlist above would accept a cross-swap (e.g. ``us_recognition`` + declaring ``european_calendar_date``), and with dual-target date rules + such a swap is behaviorally inert today — but the moment any rule + targets a single id, a wrong declaration silently mis-canonicalizes + US/EU dates. Pin the name→id mapping explicitly (ADR-0003 + consistency-guard rationale). + """ + by_name = {grammar.name: grammar.semantics for grammar in Date().get_grammars()} + assert by_name["us_recognition"] == "us_calendar_date" + assert by_name["european_recognition"] == "european_calendar_date" + class TestGrammarSemanticsEnforcement: @pytest.mark.unit From d307ae149342070a15c08ad349cca2e5be3f385e Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Wed, 12 Aug 2026 00:40:08 +0800 Subject: [PATCH 21/22] docs: correct ADR-0003 inventory count, note lookaround change Oracle review follow-up (ADR-0003): - ADR + plan D10: verified inventory is 55 files (44 sweep-relevant: 26 under paxman/, 12 tests, 6 repo-root docs incl HOW_TO_ADD_NEW_GRAMMAR.md) - ADR Phase 1 + plan out-of-scope: post-ADR addendum that the date grammars' digit-lookaround bounds deliberately stop digit-glued ids like 12026-01-15 from partially matching - plan: sync progress table (all ten tasks landed) --- docs/adr/0003-semantic-affinity-routing.md | 6 +- ...08-11-adr0003-semantic-affinity-routing.md | 91 +++++++++++++------ 2 files changed, 70 insertions(+), 27 deletions(-) diff --git a/docs/adr/0003-semantic-affinity-routing.md b/docs/adr/0003-semantic-affinity-routing.md index 8cd9c0a0..8723470e 100644 --- a/docs/adr/0003-semantic-affinity-routing.md +++ b/docs/adr/0003-semantic-affinity-routing.md @@ -137,6 +137,10 @@ is rule territory by design. (ruff, pyright, import-linter, pytest) stays green with no test edits. Community rule metadata (`target_grammars` → `target_semantics`) is a breaking rename at 0.x, acceptable under the same policy as ADR-0002. + *Post-ADR correction:* during Migration #4 the four date grammars' + digit-lookaround bounds (`(? **For agentic workers.** This plan is written to be executed by a worker @@ -19,26 +19,42 @@ > a verify-only gate with **no commit**, so the exact-commit-message > requirement does not apply to it. -> **Progress — START HERE at Task 3.** +> **Progress — COMPLETE.** All ten tasks landed. > > | Task | Status | Commit | > |------|--------|--------| > | Task 1 — declare `semantics` on all shipped grammars | ✅ landed | `ebf21bb` | > | Task 2 — rename `target_grammars` → `target_semantics` | ✅ landed | `f8ac1f6` | -> | Task 3 — enforce `Grammar.semantics` at class-definition time | ⬜ **pending** | — | -> | Task 4 — engine routes on semantics | ⬜ pending | — | -> | Tasks 5-7 — Phase 2 coalescing (Date, Email, Phone) | ⬜ pending | — | -> | Task 8 — consistency guard | ⬜ pending | — | -> | Task 9 — docs sweep | ⬜ pending | — | -> | Task 10 — final gate (no commit) | ⬜ pending | — | +> | Task 3 — enforce `Grammar.semantics` at class-definition time | ✅ landed | `0ea7727` | +> | Task 4 — engine routes on semantics | ✅ landed | `0f0f9d1` | +> | Task 5 — Phase 2: Date coalescing | ✅ landed | `3c592b8` | +> | Task 6 — Phase 2: Email coalescing | ✅ landed | `9317596` | +> | Task 7 — Phase 2: Phone coalescing | ✅ landed | `abae3e4` (`fd22152` prep fix) | +> | Task 8 — consistency guard | ✅ landed | `36315dc` | +> | Task 9 — docs sweep | ✅ landed | `f6f7790` | +> | Task 10 — final gate (no commit) | ✅ green | follow-up `125c06d` | > -> Tasks 1 and 2 are **done — do not re-execute them.** `Grammar` still lacks -> `semantics` and `__init_subclass__` (`paxman/core/domain.py:228-240`); the -> engine still routes on grammar names (`paxman/engine/orchestrator.py:287`, -> `:312`, `:399`) — those line numbers are current and verified. Task 3's -> inventory is verified: exactly 18 test-defined `Grammar` subclasses in 8 -> test files need `semantics = ""`, at the exact locations listed in -> Task 3. +> Execution notes for the follow-up session that ran Tasks 5-10: +> +> - Task 5 also updated `tests/unit/test_grammar_shipped_grammars_declare_semantics_identity` +> (now `test_shipped_grammars_declare_semantics_identity`) via a module-level +> `_COALESCED_SEMANTICS` allowlist, so every commit stayed green — approved +> deviation; the allowlist mirrors D6 exactly and is itself locked by the +> guard's enumeration-completeness test. +> - M1/M2 (orchestrator docstring KeyError invariant + drop the duplicate +> sentence at ARCHITECTURE.md:201) and M3 (README fail-fast mechanism, from +> Task 5 execution) were folded into Task 9. Task 5's RED driver 2 failed +> with a silent-exclusion `assert False`, not the plan-predicted +> `ContractError` — README's claim is accurate and now documented precisely. +> - Task 8's multi-member group count is scoped per capability (Currency and +> Money share `code_recognition`/`symbol_recognition`/`word_recognition` +> across capabilities; routing is per-capability, so groups are enumerated +> per capability). +> - Task 10 gate: `ruff format --check .` flags 8 pre-existing historical +> docs (`docs/research/*`, `docs/superpowers/plans/*` — untouched on this +> branch, forbidden to edit, out of CI scope); the CI-authoritative gate +> (`ruff check paxman/ tests/ && ruff format --check paxman/ tests/`) is +> fully green. All ten plan tasks are done — do not re-execute them. --- @@ -131,15 +147,21 @@ output is byte-identical to today for every existing input. exclude generated dirs (`htmlcov/`, `.hypothesis/`, `.pytest_cache/`, `.venv/`) or it fails on a dirty local checkout. - **D10 — Ground truth over ADR claims.** The ADR's "55 files reference - `target_grammars`" (L205) is stale. Verified ground truth: **44 files** + `target_grammars`" (L205) was a pre-migration estimate. Verified by + re-counting at the migration start commit: **55 files** — 44 sweep-relevant (26 under `paxman/` — 24 `.py` + 2 nested AGENTS.md, 12 test files, 6 - repo-root doc files). Where the ADR and this plan disagree on - counts/paths, this plan's verified inventory wins. + repo-root doc files, `HOW_TO_ADD_NEW_GRAMMAR.md` included) plus 8 plan + 3 + research files excluded by D9. Where the ADR and this plan disagree on + counts/paths, this plan's verified inventory wins (the ADR's count is + superseded as stale). ### Out of scope - No behavior change to recognition/validation/status semantics (Phase 1 is - byte-identical; Phase 2 coalesces declarations only). + byte-identical; Phase 2 coalesces declarations only). Post-plan correction: + the date grammars' digit-lookaround bounds were tightened so digit-glued + ids like `12026-01-15` no longer partially match — deliberate, so "no + behavior change" excludes that digit-glued class only. - No rename of `GrammarRule.grammar_name`, `RecognizedRep`, `Candidate`, or `_dedup_candidates` keys — provenance stays name-based (ADR §4). - No edits to historical plans/research/ADR files (D9). @@ -462,12 +484,21 @@ Two RED drivers: the consistency-guard test's group lookup, and `frozenset({"us_calendar_date", "european_calendar_date"})` (no widening, D6). - No other rule file changes (D7). +- **Plan deviation (approved 2026-08-11)**: `test_shipped_grammars_declare_semantics_identity` + (`tests/unit/test_grammar_semantics_metadata.py` L25-32) asserts `semantics == name` for + every shipped grammar; coalescing breaks it. Update it in this commit: keep the str / + non-empty assertions for all grammars, add a module-level `_COALESCED_SEMANTICS` frozenset + (seeded `{"iso8601_calendar_date", "us_calendar_date", "european_calendar_date"}`), and + assert `semantics == name or semantics in _COALESCED_SEMANTICS`. Tasks 6-7 extend the set + with the Email/Phone ids. Keeps every commit green — Task 8's `pytest tests/unit` verify + passes as written. **Verify** ```bash uv run pytest tests/capabilities/date tests/integration/test_grammar_extensions.py \ tests/unit/test_grammar_semantics_consistency.py -q uv run pytest tests/integration -q +uv run pytest tests/unit/test_grammar_semantics_metadata.py -q uv run ruff check paxman/capabilities/Date/ tests/ uv run pyright ``` @@ -589,6 +620,11 @@ test: lock same-semantics field-mapping consistency ### Task 9 — `docs: sweep target_grammars and document semantic affinity` +> **Follow-up items folded in here (M1-M2: Task 4 review; M3: Task 5 execution):** +> - **M1 (source docstring):** in `paxman/engine/orchestrator.py` `_collect_candidates`, add one sentence to the docstring stating the KeyError invariant for `semantics_by_name[grammar_name]`: recognitions are produced only by grammars in the composed `all_grammars` (`_recognize` filters against `supported_names`), the same list the map is built from. +> - **M2 (dead citation):** the `(ARCHITECTURE.md:201)` reference in that same docstring is stale (ARCHITECTURE.md has no routing/affinity content; L201 is "Quality Enforcement") — drop the dead line-number reference while updating the docstring. +> - **M3 (README fail-fast mechanism, verified against the engine 2026-08-11):** the "Rules of the seam" fail-fast bullet must state the real mechanism — `_activated_rules` resolves `extra_grammars` names via `semantics_by_name.get(n, n)`, so an unknown extra name keeps its own string and can activate a community rule targeting that (dangling) id, which `_validate_affinity` then rejects with `ContractError`. A rule that is NOT opted in is silently inert regardless of dangling targets. (Task 5's RED driver exercised the inert path, not the fail-fast path, hence the plan's original ContractError prediction did not match.) + Docs sweep (ADR Migration #4, D9). **No RED step** — pure documentation. **Step 1 GREEN — rewrite the 6 in-scope repo-root files + 2 nested AGENTS.md** @@ -708,21 +744,24 @@ historical docs. ## §4 Definition of Done -- [ ] All 26 shipped grammars declare `semantics` (identity in Phase 1; +- [x] All 26 shipped grammars declare `semantics` (identity in Phase 1; coalesced ids for Date/Email/Phone after Phase 2), enforced by `Grammar.__init_subclass__` at class-definition time with tests. -- [ ] Zero `target_grammars` anywhere in `paxman/` or `tests/`; the Task 9 +- [x] Zero `target_grammars` anywhere in `paxman/` or `tests/`; the Task 9 zero-grep proof is CLEAN outside the excluded historical paths. -- [ ] Engine routes on semantics at all three sites; provenance and +- [x] Engine routes on semantics at all three sites; provenance and candidate dedup remain name-based (`GrammarRule.grammar_name`, `Candidate.recognition_rule` unchanged). -- [ ] Phase 2 coalescing landed for Date/Email/Phone exactly as D6/D7 +- [x] Phase 2 coalescing landed for Date/Email/Phone exactly as D6/D7 scope; no rule's `target_semantics` set grew. -- [ ] `tests/unit/test_grammar_semantics_consistency.py` covers every +- [x] `tests/unit/test_grammar_semantics_consistency.py` covers every multi-member semantics group with probe rows and locks the singleton no-coalesce set. -- [ ] Docs swept (Task 9 files); README's community example shows +- [x] Docs swept (Task 9 files); README's community example shows `semantics` on the grammar and `target_semantics` on the rule. -- [ ] Full pre-PR gate green: `ruff check . && ruff format --check . && +- [x] Full pre-PR gate green: `ruff check . && ruff format --check . && pyright && import-linter lint && pytest` and 95% coverage per package. + (Gate as written is green under CI scope; `ruff format --check .` + additionally flags 8 pre-existing historical docs — see progress + table note.) From 500e9de04867de93bb87a7a4066a86928608cf98 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Wed, 12 Aug 2026 11:43:07 +0800 Subject: [PATCH 22/22] docs: align plan gate scope, digit-glued exclusion, sweep proofs Review follow-up (ADR-0003): - Task 10 gate command now matches CI: ruff check/format scoped to paxman/ and tests/ (the full-tree command never was CI-authoritative) - ADR section 4 + plan goal: byte-identical output claim explicitly excludes digit-glued dates (post-ADR lookaround tightening) - Sweep proofs: Task 9 command now searches paxman/ and tests/ (the nested AGENTS.md are swept there); drop the || echo "CLEAN" fallbacks in both commands so grep errors (exit >= 2) stay visible instead of being reported as a clean result; Task 10 and DoD statements updated to match --- docs/adr/0003-semantic-affinity-routing.md | 4 +- ...08-11-adr0003-semantic-affinity-routing.md | 39 +++++++++++-------- 2 files changed, 25 insertions(+), 18 deletions(-) diff --git a/docs/adr/0003-semantic-affinity-routing.md b/docs/adr/0003-semantic-affinity-routing.md index 8723470e..3cacf74b 100644 --- a/docs/adr/0003-semantic-affinity-routing.md +++ b/docs/adr/0003-semantic-affinity-routing.md @@ -115,7 +115,9 @@ target_semantics = frozenset({"iso8601_calendar_date"}) (`grammar_name`), and `_dedup_candidates` continues to collapse on `(value, recognition_rule, validation_rule)`. `semantics` is routing metadata; the grammar name remains the audit identity of the recognition. Provenance -output is byte-identical to today for every existing input. +output is byte-identical to today for every existing input, excluding +digit-glued dates — the date grammars' lookaround bounds were tightened +post-ADR (see the Migration note). ### 5. What a grammar-only addition now means diff --git a/docs/superpowers/plans/2026-08-11-adr0003-semantic-affinity-routing.md b/docs/superpowers/plans/2026-08-11-adr0003-semantic-affinity-routing.md index 322bcabf..43fb39f8 100644 --- a/docs/superpowers/plans/2026-08-11-adr0003-semantic-affinity-routing.md +++ b/docs/superpowers/plans/2026-08-11-adr0003-semantic-affinity-routing.md @@ -71,7 +71,8 @@ semantics at all three sites (`_validate_affinity`, `_collect_candidates`, for every shipped grammar); Phase 2 coalesces same-meaning grammars in Date, Email, and Phone; a consistency-guard test locks same-semantics/same-field- mapping; docs are swept. Provenance and candidate dedup stay name-based — -output is byte-identical to today for every existing input. +output is byte-identical to today for every existing input, excluding the +digit-glued date class (post-plan lookaround tightening; see Out of scope). ### D-Decisions (locked — do not revisit without a new ADR) @@ -324,8 +325,10 @@ uv run pyright ``` Also confirm the sweep is complete (zero hits inside `paxman/` and `tests/`): ```bash -grep -rn "target_grammars" paxman/ tests/ || echo "CLEAN" +grep -rn "target_grammars" paxman/ tests/ ``` +(No `|| echo "CLEAN"` fallback — zero matches prints nothing and exits 1; a +grep error exits ≥ 2 and stays visible.) **Commit** ``` @@ -646,18 +649,19 @@ the **repo root**, not under `docs/`. Do NOT touch (historical records, ADR-0002 precedent): `docs/superpowers/ plans/*`, `docs/report/*`, `docs/research/*`, `docs/adr/*`. -**Verify** (zero hits outside the excluded paths — generated dirs excluded, -verified command): +**Verify** (zero hits outside the excluded paths — the historical records and +generated dirs are excluded; `paxman/` and `tests/` are searched, because the +sweep covers the nested AGENTS.md there): ```bash grep -rnE 'target_grammars' . \ --exclude-dir=.git --exclude-dir=plans --exclude-dir=report \ - --exclude-dir=research --exclude-dir=adr --exclude-dir=paxman \ - --exclude-dir=tests --exclude-dir=htmlcov --exclude-dir=.hypothesis \ - --exclude-dir=.pytest_cache --exclude-dir=.venv || echo "CLEAN" + --exclude-dir=research --exclude-dir=adr \ + --exclude-dir=htmlcov --exclude-dir=.hypothesis \ + --exclude-dir=.pytest_cache --exclude-dir=.venv ``` -(`--exclude-dir=paxman --exclude-dir=tests` are dropped if this task is -assigned the Task-2 sweep proof instead — the nested AGENTS.md are the only -`paxman/` matches.) +No `|| echo "CLEAN"` fallback: zero matches is the expected result (grep +prints nothing, exits 1); any grep error (exit ≥ 2) stays visible instead of +being reported as "CLEAN". **Commit** ``` @@ -668,10 +672,11 @@ docs: sweep target_grammars and document semantic affinity ### Task 10 — Final gate (no commit) -**Verify — full pre-PR gate** (authoritative per `.github/workflows/ci.yml`): +**Verify — full pre-PR gate** (authoritative per `.github/workflows/ci.yml`; +ruff lint and format are CI-scoped to `paxman/ tests/`): ```bash -uv run ruff check . && uv run ruff format --check . && uv run pyright && \ - uv run import-linter lint && uv run pytest +uv run ruff check paxman/ tests/ && uv run ruff format --check paxman/ tests/ \ + && uv run pyright && uv run import-linter lint && uv run pytest ``` Coverage gate (one include pattern per package — the brace shorthand `paxman/{core,capabilities,engine,api}/*` is not expanded by the installed @@ -682,9 +687,9 @@ uv run coverage report --include="paxman/capabilities/*" --fail-under=95 uv run coverage report --include="paxman/engine/*" --fail-under=95 uv run coverage report --include="paxman/api/*" --fail-under=95 ``` -Zero-grep proof (Task 9 Verify) returns CLEAN; `target_grammars` appears -nowhere in `paxman/` or `tests/`; `semantics` is present on all 26 shipped -grammars (Task 1 test still green). +Zero-grep proof (Task 9 Verify) shows no matches outside the excluded +paths; `target_grammars` appears nowhere in `paxman/` or `tests/`; `semantics` +is present on all 26 shipped grammars (Task 1 test still green). If any gate fails, fix it in a follow-up commit — never by weakening a test, never by restoring `target_grammars`, never by editing the excluded @@ -748,7 +753,7 @@ historical docs. coalesced ids for Date/Email/Phone after Phase 2), enforced by `Grammar.__init_subclass__` at class-definition time with tests. - [x] Zero `target_grammars` anywhere in `paxman/` or `tests/`; the Task 9 - zero-grep proof is CLEAN outside the excluded historical paths. + zero-grep proof shows no matches outside the excluded historical paths. - [x] Engine routes on semantics at all three sites; provenance and candidate dedup remain name-based (`GrammarRule.grammar_name`, `Candidate.recognition_rule` unchanged).