From 7c33fa361d5a33665c459e564f743758d0dac2d9 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Wed, 19 Aug 2026 19:44:26 +0800 Subject: [PATCH 1/4] docs: add segmentation recipe for multi-entity input --- docs/recipes/segmentation.md | 158 +++++++++++++++++++++++++++++++++++ 1 file changed, 158 insertions(+) create mode 100644 docs/recipes/segmentation.md diff --git a/docs/recipes/segmentation.md b/docs/recipes/segmentation.md new file mode 100644 index 00000000..189d4010 --- /dev/null +++ b/docs/recipes/segmentation.md @@ -0,0 +1,158 @@ +# Segmentation Recipe — Multi-Entity Input + +> One `paxman.canonicalize()` call resolves one presumed entity. Multi-entity +> input is caller-owned segmentation: split, then canonicalize per mention. + +This recipe is the sanctioned pattern for "find all X in this text" demand +without bending Paxman's scope. See [ADR-0004](../adr/0004-single-value-invariant.md) +and the architecture review [§8 M1](../reports/2026-08-17-architecture-review.md) for the charter. + +--- + +## 1. The invariant + +One entity per `canonicalize()` call is the product contract ([ADR-0004](../adr/0004-single-value-invariant.md)). +Paxman operates at the *mention* level: the caller ensures the slice passed to +each call contains one presumed entity (or none). + +* **`AMBIGUOUS`** means a genuine single-mention spec conflict — one recognized + span, two authorities disagreeing on its canonical value (e.g. `01/02/2026` + as `2026-01-02` vs `2026-02-01`). +* **`MultipleMentionsError`** means your input contained two or more separate + mentions that resolved to different values. It is a segmentation-usage signal, + not a domain result — it fails fast instead of masquerading as ambiguity. + +Segmentation is **caller-owned by charter**, not a missing feature. This is +mandate [M1 in the architecture review §8](../reports/2026-08-17-architecture-review.md) +and the core decision of ADR-0004: multi-entity extraction belongs outside the +library. For the four resolution statuses see [README — Resolution Status](../../README.md#resolution-status). + +--- + +## 2. The recipe + +Segment → canonicalize per mention → reassemble. + +Your segmenter finds mention *candidates*; Paxman canonicalizes each candidate +and tells you whether it is `SUCCESS`, `INVALID`, `MISSING`, or `AMBIGUOUS`. +Spans and `MultipleMentionsError` make the loop robust — the error fires when +your segmenter let two mentions through. + +```python +import re + +import paxman +from paxman.capabilities import Email +from paxman.core.discovery import register_capability +from paxman.core.domain import Resolution +from paxman.core.errors import MultipleMentionsError + +register_capability(Email()) +contract = Email.create_contract() + +# Caller-owned segmentation: a coarse pattern finds mention candidates… +EMAIL_LIKE = re.compile(r"[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}") + +def canonicalize_emails(text: str) -> list[tuple[str, str, str]]: + """Return (raw, canonical, position) for every email mention.""" + out: list[tuple[str, str, str]] = [] + for m in EMAIL_LIKE.finditer(text): + try: + result = paxman.canonicalize(m.group(0), contract) + except MultipleMentionsError: + # Your segmenter let two mentions through — tighten it. + raise + if result.status is Resolution.SUCCESS: + out.append( + (m.group(0), result.canonicalized_value or "", str(m.start())) + ) + return out +``` + +The regex above is **deliberately coarse** — it is a caller-owned candidate +finder, not an RFC 5322 validator. Treat it as a cheap pre-filter; Paxman's +grammars and rules remain the authority on whether a candidate is `SUCCESS`, +`INVALID`, or `AMBIGUOUS`. This is pitfall (a) below: a coarse, +capability-shaped pattern beats naive splitting, but must not pretend to +replace capability validation. + +--- + +## 3. Span mechanics + +Every `ExecutionResult.span` and `Candidate.span` is a half-open `[start, end)` +offset into **the slice passed to THAT `canonicalize()` call**, not into the +original document. With per-mention calls, `result.span` is relative to the +SLICE (the single-mention string you handed to `canonicalize()`), e.g. `0` +means "start of this candidate string." + +To reassemble document positions, add the segmenter's offset: + +* `m.start()` / `m.end()` from your segmenter — document-absolute. +* `result.span` / `candidate.span` — mention-local, slice-relative. + +So the document position of a resolved mention is `m.start() + result.span[0]` +when you passed `m.group(0)` as the slice. For `SUCCESS` there is a single +resolved entity and `result.span` is set; for `MISSING`/`INVALID`/`AMBIGUOUS` +there is no single resolved entity and `result.span` is `None` — locate +mentions via per-`Candidate.span` on `AMBIGUOUS` instead. + +--- + +## 4. Signals, not failures + +Per mention, the four statuses keep their exact meanings (see +[README — Resolution Status](../../README.md#resolution-status)): + +* `SUCCESS` — one canonical value resolved. +* `INVALID` — recognized, but no authority validates it. +* `MISSING` — nothing recognized. +* `AMBIGUOUS` — one mention, multiple authorities disagree. + +`MultipleMentionsError` is **not** a Paxman status. It is a segmenter bug +detector: two mentions landed in one slice and they disagree on value. It never +represents Paxman state; it tells you to tighten the segmenter so each slice +holds at most one mention. Handle it as an invariant violation in the caller, +not as a domain outcome to branch on. + +--- + +## 5. Pitfalls + +(a) **Naive splitting vs capability-shaped patterns.** Splitting on commas, +newlines, or whitespace is brittle — addresses, display names, and surrounding +punctuation break naive delimiters. Prefer a capability-shaped coarse pattern +(like `EMAIL_LIKE` above) that approximates the capability's own grammars. Keep +it coarse and let Paxman decide validity; a too-strict pre-filter silently +drops mentions that would have been `INVALID`/`AMBIGUOUS` honestly. + +(b) **Segmenter vs grammar boundary disagreement.** Your segmenter and Paxman's +grammars may disagree on where a mention starts or ends. Always feed the +segmenter's slice to `canonicalize()` and **trust the returned status** — an +honest `INVALID` or `AMBIGUOUS` beats pre-filtering or trimming the slice to +force a `SUCCESS`. If a slice is rejected, it is the segmenter's candidate +that was wrong, not Paxman's verdict. + +(c) **Don't widen a segment to "give context."** Adding surrounding words to +help Paxman understand a mention backfires: extra text that contains another +mention triggers `MultipleMentionsError`. Keep slices tight to one presumed +entity; Paxman is stateless per call and needs no surrounding document context. + +--- + +## 6. Scope statement + +Extraction stays **caller-owned forever** ([M1](../reports/2026-08-17-architecture-review.md), +[ADR-0004](../adr/0004-single-value-invariant.md)). This recipe is the sanctioned +pattern for multi-entity input, and requests for built-in document extraction +are out of scope by charter, not by limitation. Paxman will not ship a +"find all emails/phones/dates in this document" API — the split-then-canonicalize +loop above is the intended interface. + +--- + +## References + +* [ADR-0004: Single-Value Invariant](../adr/0004-single-value-invariant.md) +* [Architecture Review §8 M1](../reports/2026-08-17-architecture-review.md) — "One entity per call, forever" +* [README — Resolution Status](../../README.md#resolution-status) — `MISSING` / `INVALID` / `SUCCESS` / `AMBIGUOUS` From d7b1732fc3d26a93f0c0217516f321b07a5f1f00 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Wed, 19 Aug 2026 19:45:53 +0800 Subject: [PATCH 2/4] docs: link the segmentation recipe from README and ARCHITECTURE --- ARCHITECTURE.md | 2 ++ README.md | 4 ++++ 2 files changed, 6 insertions(+) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 9e454244..871281b7 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -190,6 +190,8 @@ The system produces one of four resolution statuses: Ambiguity is detected at the value level, not the candidate level. Multiple candidates with the same canonical value still produce SUCCESS. Ambiguity requires genuinely different canonical outputs from different authoritative sources. +For multi-entity input, segmentation is caller-owned — see the [segmentation recipe](docs/recipes/segmentation.md) (ADR-0004 companion). + --- ## Error Handling diff --git a/README.md b/README.md index 04eaceab..315f6fc3 100644 --- a/README.md +++ b/README.md @@ -611,6 +611,10 @@ except ValidationError as e: print(f"Validation failed in {e.rule}: {e}") ``` +### Working with Multi-Entity Input + +Paxman resolves one mention per `canonicalize()` call; input containing multiple entities raises `MultipleMentionsError`. For the caller-owned split-then-canonicalize pattern, see [docs/recipes/segmentation.md](docs/recipes/segmentation.md). + --- ## Learn More From 28fca136ab1a3ed02a63f094d6f8fd1ac5eae7e8 Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Wed, 19 Aug 2026 20:00:46 +0800 Subject: [PATCH 3/4] fix(docs): address oracle and thermo review findings on segmentation recipe MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - docs/recipes/segmentation.md: clarify MultipleMentionsError is PaxmanError exception not Resolution status (oracle NIT) - docs/recipes/segmentation.md: narrow pitfall (c) to distinct-value predicate per ADR-0004 (identical values still coalesce to SUCCESS) (thermo LOW #1) Skipped: thermo LOW #2 registry guard note and str(m.start()) NITs — verbatim Email block is locked per D2, cannot change without breaking canonical text; span None NIT and ellipsis/promise NITs are non-blocking wording polish. --- docs/recipes/segmentation.md | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/docs/recipes/segmentation.md b/docs/recipes/segmentation.md index 189d4010..998a97d7 100644 --- a/docs/recipes/segmentation.md +++ b/docs/recipes/segmentation.md @@ -109,7 +109,7 @@ Per mention, the four statuses keep their exact meanings (see * `MISSING` — nothing recognized. * `AMBIGUOUS` — one mention, multiple authorities disagree. -`MultipleMentionsError` is **not** a Paxman status. It is a segmenter bug +`MultipleMentionsError` is **not** a Paxman status (it is a `PaxmanError` exception, not a `Resolution` status). It is a segmenter bug detector: two mentions landed in one slice and they disagree on value. It never represents Paxman state; it tells you to tighten the segmenter so each slice holds at most one mention. Handle it as an invariant violation in the caller, @@ -135,8 +135,10 @@ that was wrong, not Paxman's verdict. (c) **Don't widen a segment to "give context."** Adding surrounding words to help Paxman understand a mention backfires: extra text that contains another -mention triggers `MultipleMentionsError`. Keep slices tight to one presumed -entity; Paxman is stateless per call and needs no surrounding document context. +mention with a *different* canonical value triggers `MultipleMentionsError` +(identical values still coalesce to `SUCCESS` per ADR-0004). Keep slices tight +to one presumed entity; Paxman is stateless per call and needs no surrounding +document context. --- From 343f1c0cda1080a49f9a1a38bd4fc6641c41e21a Mon Sep 17 00:00:00 2001 From: Azahari Zaman Date: Wed, 19 Aug 2026 20:38:35 +0800 Subject: [PATCH 4/4] fix: address review comments on segmentation recipe and README - docs/recipes/segmentation.md: canonicalize_emails now returns a record for every EMAIL_LIKE candidate (status, value, absolute span), not just SUCCESS; converts slice-relative result.span to absolute via m.start() offset and retains raw mention data - docs/recipes/segmentation.md: clarify that only MultipleMentionsError indicates segmentation error; INVALID and AMBIGUOUS are valid per-mention outcomes to be handled as domain results, not discarded - README.md: narrow Working with Multi-Entity Input to state MultipleMentionsError occurs only when distinct mentions resolve to different canonical values (identical values coalesce to SUCCESS) All findings verified against paxman/core/domain.py, engine/orchestrator.py and ADR-0004; changes are docs-only and keep the verbatim Email block's coarse regex and register_capability usage per D4. --- README.md | 2 +- docs/recipes/segmentation.md | 19 ++++++++++--------- 2 files changed, 11 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index 315f6fc3..0d35a05d 100644 --- a/README.md +++ b/README.md @@ -613,7 +613,7 @@ except ValidationError as e: ### Working with Multi-Entity Input -Paxman resolves one mention per `canonicalize()` call; input containing multiple entities raises `MultipleMentionsError`. For the caller-owned split-then-canonicalize pattern, see [docs/recipes/segmentation.md](docs/recipes/segmentation.md). +Paxman resolves one mention per `canonicalize()` call; `MultipleMentionsError` occurs only when distinct recognized mentions in one slice resolve to different canonical values — identical canonical values still coalesce to `SUCCESS`. For the caller-owned split-then-canonicalize pattern, see [docs/recipes/segmentation.md](docs/recipes/segmentation.md). --- diff --git a/docs/recipes/segmentation.md b/docs/recipes/segmentation.md index 998a97d7..d9eaf625 100644 --- a/docs/recipes/segmentation.md +++ b/docs/recipes/segmentation.md @@ -53,19 +53,21 @@ contract = Email.create_contract() # Caller-owned segmentation: a coarse pattern finds mention candidates… EMAIL_LIKE = re.compile(r"[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}") -def canonicalize_emails(text: str) -> list[tuple[str, str, str]]: - """Return (raw, canonical, position) for every email mention.""" - out: list[tuple[str, str, str]] = [] +def canonicalize_emails(text: str) -> list[tuple[str, Resolution, str | None, tuple[int, int] | None]]: + """Return (raw, status, canonical, absolute_span) for every email candidate.""" + out: list[tuple[str, Resolution, str | None, tuple[int, int] | None]] = [] for m in EMAIL_LIKE.finditer(text): try: result = paxman.canonicalize(m.group(0), contract) except MultipleMentionsError: # Your segmenter let two mentions through — tighten it. raise - if result.status is Resolution.SUCCESS: - out.append( - (m.group(0), result.canonicalized_value or "", str(m.start())) - ) + abs_span = ( + (m.start() + result.span[0], m.start() + result.span[1]) + if result.span is not None + else None + ) + out.append((m.group(0), result.status, result.canonicalized_value, abs_span)) return out ``` @@ -130,8 +132,7 @@ drops mentions that would have been `INVALID`/`AMBIGUOUS` honestly. grammars may disagree on where a mention starts or ends. Always feed the segmenter's slice to `canonicalize()` and **trust the returned status** — an honest `INVALID` or `AMBIGUOUS` beats pre-filtering or trimming the slice to -force a `SUCCESS`. If a slice is rejected, it is the segmenter's candidate -that was wrong, not Paxman's verdict. +force a `SUCCESS`. Only `MultipleMentionsError` indicates a segmentation error; `INVALID` and `AMBIGUOUS` are valid per-mention outcomes that tell you the candidate was recognized but not validated or was ambiguous, so handle them as domain results rather than discarding them as segmentation failures. (c) **Don't widen a segment to "give context."** Adding surrounding words to help Paxman understand a mention backfires: extra text that contains another