From 1068f147421d29ba88793d37e8f8454d7bb38cba Mon Sep 17 00:00:00 2001 From: Seth Fitzsimmons Date: Wed, 19 Aug 2026 11:53:59 -0700 Subject: [PATCH 1/5] feat(common): tag Overture feature types with `overture` Discovery emitted `feature` and `overture:theme=*` but never the plain `overture` tag, so the filter the CLI's own help suggested matched nothing. Add an `overture_provider` in overture-schema-common that attaches the tag when any concrete arm of an entry point subclasses `OvertureFeature`, and reserve the tag to that package. "Is this built on Overture's feature model?" is now answerable by tag, so a consumer asking it needs neither a dependency on this package nor an `issubclass` call. The tag is not a claim that the type belongs to the Overture schema. Any package can subclass `OvertureFeature` and register an entry point, and this provider tags it like any other; reserving the tag governs who may emit it, not which models receive it. Signed-off-by: Seth Fitzsimmons --- .../changelog.d/668.feature.md | 1 + .../overture-schema-common/pyproject.toml | 1 + .../overture/schema/common/tag_providers.py | 49 +++++++++++++ .../tests/test_common_tag_providers.py | 71 +++++++++++++++---- packages/overture-schema-system/README.md | 4 +- .../changelog.d/668.misc.md | 1 + .../schema/system/discovery/discovery.py | 1 + .../tests/test_tag_providers.py | 19 +++++ 8 files changed, 133 insertions(+), 14 deletions(-) create mode 100644 packages/overture-schema-common/changelog.d/668.feature.md create mode 100644 packages/overture-schema-system/changelog.d/668.misc.md diff --git a/packages/overture-schema-common/changelog.d/668.feature.md b/packages/overture-schema-common/changelog.d/668.feature.md new file mode 100644 index 000000000..2499fe03d --- /dev/null +++ b/packages/overture-schema-common/changelog.d/668.feature.md @@ -0,0 +1 @@ +Added an `overture` tag provider that marks every entry point built on `OvertureFeature`, so consumers can select those types by tag instead of importing `OvertureFeature`. diff --git a/packages/overture-schema-common/pyproject.toml b/packages/overture-schema-common/pyproject.toml index 7c2448a44..3eca61aff 100644 --- a/packages/overture-schema-common/pyproject.toml +++ b/packages/overture-schema-common/pyproject.toml @@ -37,4 +37,5 @@ dev = [ ] [project.entry-points."overture.tag_providers"] +overture = "overture.schema.common.tag_providers:overture_provider" theme = "overture.schema.common.tag_providers:theme_provider" diff --git a/packages/overture-schema-common/src/overture/schema/common/tag_providers.py b/packages/overture-schema-common/src/overture/schema/common/tag_providers.py index 02b6ed161..57b1462fb 100644 --- a/packages/overture-schema-common/src/overture/schema/common/tag_providers.py +++ b/packages/overture-schema-common/src/overture/schema/common/tag_providers.py @@ -14,6 +14,55 @@ from overture.schema.system.discovery import ModelKey +def overture_provider( + types: Iterable[type[BaseModel]], + key: ModelKey, + tags: set[str], +) -> set[str]: + """Add `"overture"` when the entry point references an `OvertureFeature`. + + The tag says the model is built on Overture's feature model -- that it + carries the theme/type/id/version/sources contract `OvertureFeature` + defines -- as distinct from `feature`, which says only that it is a + `Feature`. It exists so that question is answerable by tag: + `overture-schema list-types --tag overture` selects those types, and a + consumer filtering on them needs no dependency on this package and no + `issubclass` call. + + It is *not* a claim that the type belongs to the Overture schema. Any + package can subclass `OvertureFeature` and register an entry point, and + this provider tags it like any other. Reserving the tag to this package + governs who may *emit* it, not which models receive it. + + One qualifying arm is enough: a discriminated union with any + `OvertureFeature` arm is an Overture feature type. + + This shares its predicate with `theme_provider` below, so `overture` is + also derivable as "carries any `overture:theme=` tag". It is emitted + separately because a selector a user can type is worth more than the + derivation, and because a future non-themed `OvertureFeature` would + break the equivalence. + + Parameters + ---------- + types + Concrete `BaseModel` subclasses for the entry point. For + discriminated-union features this is every arm. + key + Key identifying the model. + tags + Current tags; may be extended. + + Returns + ------- + set[str] + Updated tags, with `"overture"` added if applicable. + """ + if any(issubclass(tp, OvertureFeature) for tp in types): + tags.add("overture") + return tags + + def theme_provider( types: Iterable[type[BaseModel]], key: ModelKey, diff --git a/packages/overture-schema-common/tests/test_common_tag_providers.py b/packages/overture-schema-common/tests/test_common_tag_providers.py index feaa0064b..d323b63b4 100644 --- a/packages/overture-schema-common/tests/test_common_tag_providers.py +++ b/packages/overture-schema-common/tests/test_common_tag_providers.py @@ -7,11 +7,16 @@ from overture.schema.common import OvertureFeature from overture.schema.common.tag_providers import ( + overture_provider, theme_provider, ) from overture.schema.system.discovery import ModelKey from overture.schema.system.discovery.discovery import _generate_tags -from overture.schema.system.discovery.types import TagProviderDict, TagProviderKey +from overture.schema.system.discovery.types import ( + TagProvider, + TagProviderDict, + TagProviderKey, +) @pytest.fixture @@ -34,30 +39,41 @@ def _empty_key(name: str = "x", entry_point: str = "mod:X") -> ModelKey: return ModelKey(name=name, entry_point=entry_point, tags=frozenset()) -def test_theme_provider_plain_class(building: type[OvertureFeature]) -> None: - tags = theme_provider((building,), _empty_key(), set()) - assert tags == {"overture:theme=buildings"} - +@pytest.fixture +def transportation_union() -> object: + """A two-arm transportation union. `_generate_tags` walks it to the arms.""" -def test_theme_provider_discriminated_union() -> None: - # `_generate_tags` is responsible for walking the union to concrete arms. class Road(OvertureFeature[Literal["transportation"], Literal["road"]]): pass class Rail(OvertureFeature[Literal["transportation"], Literal["rail"]]): pass - union = Annotated[ + return Annotated[ Annotated[Road, Tag("road")] | Annotated[Rail, Tag("rail")], Field(discriminator="type"), ] - provider_key = TagProviderKey( - name="theme", - entry_point="common:theme_provider", + + +def _providers(provider: TagProvider) -> TagProviderDict: + """Register one provider under a key this package is allowed to use.""" + key = TagProviderKey( + name=provider.__name__.removesuffix("_provider"), + entry_point=f"common:{provider.__name__}", package_name="overture-schema-common", ) - providers: TagProviderDict = {provider_key: theme_provider} - tags = _generate_tags(union, _empty_key(), providers) + return {key: provider} + + +def test_theme_provider_plain_class(building: type[OvertureFeature]) -> None: + tags = theme_provider((building,), _empty_key(), set()) + assert tags == {"overture:theme=buildings"} + + +def test_theme_provider_discriminated_union(transportation_union: object) -> None: + tags = _generate_tags( + transportation_union, _empty_key(), _providers(theme_provider) + ) assert tags == {"overture:theme=transportation"} @@ -74,3 +90,32 @@ class BadFeature(OvertureFeature): # type: ignore[type-arg] with pytest.raises(TypeError, match="must be annotated Literal"): theme_provider((BadFeature,), _empty_key(), set()) + + +def test_overture_provider_plain_class(building: type[OvertureFeature]) -> None: + tags = overture_provider((building,), _empty_key(), set()) + assert tags == {"overture"} + + +def test_overture_provider_discriminated_union(transportation_union: object) -> None: + tags = _generate_tags( + transportation_union, _empty_key(), _providers(overture_provider) + ) + assert tags == {"overture"} + + +def test_overture_provider_skips_non_overture(not_overture: type[BaseModel]) -> None: + tags = overture_provider((not_overture,), _empty_key(), set()) + assert tags == set() + + +def test_overture_provider_partial_union_still_tags( + not_overture: type[BaseModel], +) -> None: + """One Overture arm is enough; a mixed union still counts as Overture.""" + + class Road(OvertureFeature[Literal["transportation"], Literal["road"]]): + pass + + tags = overture_provider((not_overture, Road), _empty_key(), set()) + assert tags == {"overture"} diff --git a/packages/overture-schema-system/README.md b/packages/overture-schema-system/README.md index 84ff7b86f..503c7b0ba 100644 --- a/packages/overture-schema-system/README.md +++ b/packages/overture-schema-system/README.md @@ -132,7 +132,7 @@ class Park(Identified): class ParkBench(Identified): - park_id: Annotated[Id, Reference(Relationship.BELONGS_TO, Park)] + park_id: Annotated[Id, Reference(Relationship.COMPOSITION, Park, role="part_of")] ``` ## Discovery @@ -222,6 +222,7 @@ Specific plain tags and namespaces are reserved for designated packages. For exa |---|---| | `feature` (tag) | `overture-schema-system` | | `system:` (namespace) | `overture-schema-system` | +| `overture` (tag) | `overture-schema-common` | | `overture:` (namespace) | `overture-schema-common` | When a provider attempts to set a reserved tag from an unauthorized package, discovery logs a warning and discards the tag. @@ -229,6 +230,7 @@ When a provider attempts to set a reserved tag from an unauthorized package, dis ### Built-in Providers - **`feature`** (in `system`) -- adds `feature` if any concrete arm is a `Feature` subclass. +- **`overture`** (in `common`) -- adds `overture` if any concrete arm is an `OvertureFeature` subclass: the model is built on Overture's feature model, as distinct from `feature`, which says only that it is a `Feature`. Consumers that need to ask that question read the tag rather than importing `OvertureFeature`. The tag does not assert that the type belongs to the Overture schema -- a third-party `OvertureFeature` subclass receives it too, and the reservation governs who may emit the tag, not which models get it. - **`theme`** (in `common`) -- adds `overture:theme={theme}` for each `OvertureFeature` referenced. A discriminated-union feature whose arms span multiple themes contributes one tag per distinct theme. ### Selecting Models by Tag diff --git a/packages/overture-schema-system/changelog.d/668.misc.md b/packages/overture-schema-system/changelog.d/668.misc.md new file mode 100644 index 000000000..b76bcab60 --- /dev/null +++ b/packages/overture-schema-system/changelog.d/668.misc.md @@ -0,0 +1 @@ +Reserved the plain `overture` tag to `overture-schema-common` and documented its provider in the README. diff --git a/packages/overture-schema-system/src/overture/schema/system/discovery/discovery.py b/packages/overture-schema-system/src/overture/schema/system/discovery/discovery.py index 2425a7551..fc3537e32 100644 --- a/packages/overture-schema-system/src/overture/schema/system/discovery/discovery.py +++ b/packages/overture-schema-system/src/overture/schema/system/discovery/discovery.py @@ -24,6 +24,7 @@ # Tags that are reserved and can only be set by specific packages. _RESERVED_TAGS: dict[str, set[str]] = { "feature": {"overture-schema-system"}, + "overture": {"overture-schema-common"}, } # Namespaces that are reserved and can only be set by specific packages. _RESERVED_NAMESPACES: dict[str, set[str]] = { diff --git a/packages/overture-schema-system/tests/test_tag_providers.py b/packages/overture-schema-system/tests/test_tag_providers.py index 33fd15392..84c1069c8 100644 --- a/packages/overture-schema-system/tests/test_tag_providers.py +++ b/packages/overture-schema-system/tests/test_tag_providers.py @@ -123,6 +123,25 @@ def test_allowed_reserved_tag( assert _generate_tags(any_model, any_key, system_providers) == {"feature"} +def test_reserved_overture_tag( + other_tag_provider: TagProviderKey, + any_key: ModelKey, + any_model: type[BaseModel], +) -> None: + providers = {other_tag_provider: fake_provider("overture", "valid")} + result = _generate_tags(any_model, any_key, providers) + assert result == {"valid"} + + +def test_allowed_reserved_overture_tag( + common_tag_provider: TagProviderKey, + any_key: ModelKey, + any_model: type[BaseModel], +) -> None: + common_providers = {common_tag_provider: fake_provider("overture")} + assert _generate_tags(any_model, any_key, common_providers) == {"overture"} + + def test_reserved_namespace( other_tag_provider: TagProviderKey, any_key: ModelKey, From 39a932bde7d827c12e2ddf03178ede474bc655e7 Mon Sep 17 00:00:00 2001 From: Seth Fitzsimmons Date: Wed, 19 Aug 2026 11:54:13 -0700 Subject: [PATCH 2/5] fix(cli): render `--help` example blocks and cite tags that exist The command docstrings were raw strings, so Click's `\b` no-rewrap marker stayed two literal characters and every example block collapsed into one paragraph. Drop the `r` prefix; pydocstyle's D301 pushes the other way and is a false positive for Click docstrings, so it is waived with the reason recorded once at the module level. The "official Overture types only" examples suggested `--tag overture --tag feature`, wrong twice over: the plain `overture` tag did not exist, and `--tag` is OR, so pairing it with `feature` widens the scope instead of narrowing it. Cite `--tag overture` alone, reworded -- the tag says a type is built on Overture's feature model, not that it is official. Added an example showing what the OR is for, and tag examples to `list-types`, which had none. The `--tag` syntax note illustrated namespaced tags with `overture:approved`, which no provider can register. No shipped model emits a bare namespaced tag, so the note describes the form instead. `test_help_cites_only_tags_that_exist` covers the whole surface: it splits rendered help per option and checks both citation forms -- an argument (`--tag X`) and an illustration (`(e.g. X)`) -- against the set that option accepts, since `--group-by` takes the key half of a `key=value` tag and `--tag` does not. Judged against a global union instead, a phantom passes in whichever option it was not written for. Signed-off-by: Seth Fitzsimmons --- .../changelog.d/668.bugfix.md | 1 + .../src/overture/schema/cli/commands.py | 48 +++++-- .../src/overture/schema/cli/tag_options.py | 9 +- .../tests/test_cli_commands.py | 117 ++++++++++++++++++ 4 files changed, 159 insertions(+), 16 deletions(-) create mode 100644 packages/overture-schema-cli/changelog.d/668.bugfix.md diff --git a/packages/overture-schema-cli/changelog.d/668.bugfix.md b/packages/overture-schema-cli/changelog.d/668.bugfix.md new file mode 100644 index 000000000..f3dcd1add --- /dev/null +++ b/packages/overture-schema-cli/changelog.d/668.bugfix.md @@ -0,0 +1 @@ +Fixed `--help` example blocks rendering a literal `\b` instead of Click's no-rewrap marker, and replaced the tag-filter examples that cited tags discovery never emitted. diff --git a/packages/overture-schema-cli/src/overture/schema/cli/commands.py b/packages/overture-schema-cli/src/overture/schema/cli/commands.py index 24ce225f1..15e9ad727 100644 --- a/packages/overture-schema-cli/src/overture/schema/cli/commands.py +++ b/packages/overture-schema-cli/src/overture/schema/cli/commands.py @@ -225,10 +225,17 @@ def get_source_name(filename: Path) -> str: return "" if str(filename) == "-" else str(filename) +# Every `# noqa: D301` below is the same waiver, against `pydocstyle` (see +# the docformat-only target). D301 wants a raw string wherever a docstring +# contains a backslash, but `\b` here is Click's no-rewrap marker: a raw +# string hands Click two literal characters and every example block collapses +# into one paragraph. Any new command with an Examples block needs the waiver +# too. Note the placement is pydocstyle's -- ruff reports D301 at the +# docstring line instead, so selecting ruff's `D` rules would need its own. @click.group() @click.version_option(package_name="overture-schema") -def cli() -> None: - r"""Overture Schema command-line interface. +def cli() -> None: # noqa: D301 + """Overture Schema command-line interface. Provides validation, schema generation, and type discovery for Overture Maps data. @@ -737,8 +744,8 @@ def validate( excludes: tuple[str, ...], types: tuple[str, ...], show_fields: tuple[str, ...], -) -> None: - r"""Validate Overture Maps data against schemas. +) -> None: # noqa: D301 + """Validate Overture Maps data against schemas. Read from FILENAME or stdin if FILENAME is '-'. Supports JSON, YAML, and GeoJSON formats. @@ -757,8 +764,12 @@ def validate( # Validate specific type $ overture-schema validate --type building data.json \b - # Official Overture types only - $ overture-schema validate --tag overture --tag feature data.json + # Two themes at once (repeatable; scope is their union) + $ overture-schema validate --tag overture:theme=buildings \\ + --tag overture:theme=places data.json + \b + # Only types built on the Overture feature model + $ overture-schema validate --tag overture data.json """ # Resolve model type first (errors here are ValueErrors, not ValidationErrors) try: @@ -803,8 +814,8 @@ def json_schema_command( filters: tuple[str, ...], excludes: tuple[str, ...], types: tuple[str, ...], -) -> None: - r"""Generate JSON schema for Overture Maps types. +) -> None: # noqa: D301 + """Generate JSON schema for Overture Maps types. Outputs a JSON Schema document to stdout that can be used for validation or documentation purposes. @@ -820,8 +831,12 @@ def json_schema_command( # Specific types $ overture-schema json-schema --type building \b - # Official Overture types only - $ overture-schema json-schema --tag overture --tag feature + # Two themes at once (repeatable; scope is their union) + $ overture-schema json-schema --tag overture:theme=buildings \\ + --tag overture:theme=places + \b + # Only types built on the Overture feature model + $ overture-schema json-schema --tag overture """ try: model_type = resolve_types( @@ -838,7 +853,8 @@ def json_schema_command( @tag_selection_options @click.option( "--group-by", - help="Group types by a key/value tag's key (e.g. 'overture:theme'). " + help="Group types by a key/value tag's key, as in " + "--group-by overture:theme. " "Plain and namespaced tags have no value to group by and are " "ignored here.", ) @@ -847,8 +863,8 @@ def list_types( filters: tuple[str, ...], excludes: tuple[str, ...], group_by: str | None, -) -> None: - r"""List all available types. +) -> None: # noqa: D301 + """List all available types. Displays all registered models and can be organized by grouping. @@ -856,6 +872,12 @@ def list_types( Examples: # List all types $ overture-schema list-types + \b + # One theme + $ overture-schema list-types --tag overture:theme=buildings + \b + # Group the listing by theme + $ overture-schema list-types --group-by overture:theme """ try: models = discover_models() diff --git a/packages/overture-schema-cli/src/overture/schema/cli/tag_options.py b/packages/overture-schema-cli/src/overture/schema/cli/tag_options.py index 3befb1143..cd1965f09 100644 --- a/packages/overture-schema-cli/src/overture/schema/cli/tag_options.py +++ b/packages/overture-schema-cli/src/overture/schema/cli/tag_options.py @@ -9,10 +9,13 @@ F = TypeVar("F", bound=Callable[..., object]) +# Every tag named here must be one discovery actually emits -- a tag in help +# text reads as runnable. The namespaced form has no shipped example, so it is +# described rather than illustrated. _TAG_SYNTAX_NOTE = ( - "Accepts plain tags (e.g. feature), namespaced tags " - "(e.g. overture:approved), or compound key/value tags " - "(e.g. overture:theme=buildings)." + "Accepts plain tags (e.g. feature, overture) and compound key/value tags " + "(e.g. overture:theme=buildings). A namespaced form, namespace:predicate, " + "is also accepted for tags that third-party packages register." ) diff --git a/packages/overture-schema-cli/tests/test_cli_commands.py b/packages/overture-schema-cli/tests/test_cli_commands.py index 9fb088c7d..3a9c50a30 100644 --- a/packages/overture-schema-cli/tests/test_cli_commands.py +++ b/packages/overture-schema-cli/tests/test_cli_commands.py @@ -1,6 +1,7 @@ """Tests for CLI commands (validate, list-types, json-schema).""" import json +import re from io import StringIO import pytest @@ -8,6 +9,32 @@ from conftest import build_feature from overture.schema.cli.commands import cli +from overture.schema.system.discovery import discover_models + +_HELP_INVOCATIONS = [["--help"]] + [[name, "--help"] for name in sorted(cli.commands)] + +# Click renders each option as a line starting with two spaces and the flag, +# its help indented under it. +_OPTION_START = re.compile(r"^ (--[\w-]+)", re.MULTILINE) + +# Options whose `(e.g. ...)` illustrations name something other than a tag. +_NOT_TAGS = {"--type", "--show-field"} + + +def _help_segments(output: str) -> list[tuple[str | None, str]]: + """Split `--help` output into (option, text), so a citation keeps its option. + + The leading segment -- description and examples, before the first option + line -- carries `None`. + """ + bounds = list(_OPTION_START.finditer(output)) + if not bounds: + return [(None, output)] + segments: list[tuple[str | None, str]] = [(None, output[: bounds[0].start()])] + for current, following in zip(bounds, [*bounds[1:], None], strict=True): + end = following.start() if following else len(output) + segments.append((current[1], output[current.start() : end])) + return segments class TestListTypesCommand: @@ -29,6 +56,96 @@ def test_list_types_command_help(self, cli_runner: CliRunner) -> None: assert "list-types" in result.output.lower() +class TestHelpFormatting: + """Tests for `--help` rendering and the tags it cites.""" + + @pytest.mark.parametrize("argv", _HELP_INVOCATIONS, ids=lambda a: " ".join(a)) + def test_help_has_no_literal_escape( + self, cli_runner: CliRunner, argv: list[str] + ) -> None: + """Click's no-rewrap marker is interpreted, not printed.""" + result = cli_runner.invoke(cli, argv) + assert result.exit_code == 0 + assert "\\b" not in result.output + + def test_help_keeps_examples_on_separate_lines(self, cli_runner: CliRunner) -> None: + """Each example command occupies its own line rather than reflowing.""" + result = cli_runner.invoke(cli, ["validate", "--help"]) + assert result.exit_code == 0 + commands = [ + line.strip() + for line in result.output.splitlines() + if line.strip().startswith("$ overture-schema") + ] + assert len(commands) >= 4 + assert "$ overture-schema validate data.json" in commands + + def test_help_cites_only_tags_that_exist(self, cli_runner: CliRunner) -> None: + """Tags named in `--help` work in the option they are named for. + + Two citation forms, because a phantom has hidden in each: an + argument (`--tag X`) and an illustration (`(e.g. X)`). Both are + found wherever they appear in the rendered output, including + across a terminal wrap. + + Each is checked against the set that option accepts, not a global + union: `--group-by` takes the *key* half of a `key=value` tag, so + `--tag overture:theme` selects nothing and `--group-by feature` + groups nothing, and neither may pass. Attribution is by splitting + the Options block per option, so an illustration is judged by the + option whose help it sits in. + + Not checked: a tag named in running prose outside both forms. + `namespace:predicate` in the `--tag` syntax note is deliberately + such a placeholder -- describing the grammar without writing + something that looks runnable is why it is worded that way. Nor are + options in `_NOT_TAGS`, whose illustrations name something else -- + `--type`'s "(e.g., building, segment)" names types. + """ + emitted = {tag for key in discover_models() for tag in key.tags} + keys = {tag.split("=", 1)[0] for tag in emitted if "=" in tag} + # What each tag-taking option accepts. + accepts: dict[str | None, set[str]] = { + "--tag": emitted, + "--filter": emitted, + "--exclude": emitted, + "--group-by": keys, + } + argument_form = re.compile(r"--(tag|filter|exclude|group-by) ([\w.:=-]+)") + illustration_form = re.compile(r"e\.g\. ([^)]+)\)") + + cited: list[tuple[str, str, set[str]]] = [] + for argv in _HELP_INVOCATIONS: + result = cli_runner.invoke(cli, argv) + assert result.exit_code == 0, f"{argv}: --help failed" + where = " ".join(argv) + for option, segment in _help_segments(result.output): + # Rejoin words the terminal wrapper split, so a tag broken + # across lines is still seen whole. + flowed = " ".join(segment.split()) + cited += [ + # A citation ending a sentence picks up its period; the + # tag grammar allows `.` inside a name, never at the end. + (where, value.rstrip(".,"), accepts[f"--{flag}"]) + for flag, value in argument_form.findall(flowed) + if value != "TEXT" # the option signature's metavar + ] + if option in _NOT_TAGS: + continue + # An illustration inside a tag option's help is judged by + # that option; one in the description, the examples, or the + # Commands block belongs to no option, so accept either. + accepted = accepts.get(option, emitted | keys) + cited += [ + (where, token.strip("'\"").rstrip(".,"), accepted) + for group in illustration_form.findall(flowed) + for token in re.split(r",\s*", group.strip()) + ] + assert cited, "no tag cited anywhere in --help -- the sweep found nothing" + for where, tag, accepted in cited: + assert tag in accepted, f"{where}: {tag!r} matches no model" + + class TestJsonSchemaCommand: """Tests for the json-schema command.""" From 9635641ce9f8a1977e0bd6d81c2e5aa53b9313fc Mon Sep 17 00:00:00 2001 From: Seth Fitzsimmons Date: Wed, 19 Aug 2026 11:54:13 -0700 Subject: [PATCH 3/5] fix(codegen): name union entry points in `overture-codegen list` The listing fell back to `str(model)` for anything without `__name__`, so `Segment` printed its whole `typing.Annotated[...]` type expression -- several hundred characters of discriminator internals in place of a name. Read the name off the entry point instead. Every registered model has one, class or union alias alike, and it is the name the rest of the toolchain already uses. Signed-off-by: Seth Fitzsimmons --- .../changelog.d/668.bugfix.md | 1 + .../src/overture/schema/codegen/cli.py | 14 +++++++++----- packages/overture-schema-codegen/tests/test_cli.py | 7 +++++++ 3 files changed, 17 insertions(+), 5 deletions(-) create mode 100644 packages/overture-schema-codegen/changelog.d/668.bugfix.md diff --git a/packages/overture-schema-codegen/changelog.d/668.bugfix.md b/packages/overture-schema-codegen/changelog.d/668.bugfix.md new file mode 100644 index 000000000..d46f6bef7 --- /dev/null +++ b/packages/overture-schema-codegen/changelog.d/668.bugfix.md @@ -0,0 +1 @@ +Fixed `overture-codegen list` printing a raw `typing.Annotated[...]` expression for discriminated-union entry points such as `Segment`; entries now list by their entry-point class name. diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py b/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py index dc5a7da5a..a5687d817 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py @@ -11,6 +11,7 @@ from overture.schema.system.discovery import ( discover_models, filter_models, + split_entry_point, ) from .extraction.specs import ModelSpec, SupplementarySpec, TypeIdentity @@ -59,11 +60,14 @@ def cli() -> None: def list_models() -> None: """List all discovered models.""" models = discover_models() - names = sorted( - model.__name__ if isinstance(model, type) else str(model) - for model in models.values() - ) - for name in names: + # Name every entry from its entry point, not the loaded object: a + # discriminated union loads as an `Annotated[...]` alias with no + # `__name__`, so `str(model)` would print the whole type expression. + names = [] + for key in models: + _, class_name = split_entry_point(key.entry_point) + names.append(class_name) + for name in sorted(names): click.echo(name) diff --git a/packages/overture-schema-codegen/tests/test_cli.py b/packages/overture-schema-codegen/tests/test_cli.py index 18124ed87..dd90c6880 100644 --- a/packages/overture-schema-codegen/tests/test_cli.py +++ b/packages/overture-schema-codegen/tests/test_cli.py @@ -26,6 +26,13 @@ def test_list_shows_discovered_models(self, cli_runner: CliRunner) -> None: assert "Building" in result.output assert "Place" in result.output + def test_list_names_union_alias_entry_points(self, cli_runner: CliRunner) -> None: + """A union alias lists by its entry-point class name, not its repr.""" + result = cli_runner.invoke(cli, ["list"]) + + assert "Segment" in result.output.split() + assert "typing.Annotated" not in result.output + class TestCliGenerate: """Tests for the generate command.""" From 4cf99f87569fc1ed2ddb09872935793716137fcf Mon Sep 17 00:00:00 2001 From: Seth Fitzsimmons Date: Wed, 19 Aug 2026 11:55:38 -0700 Subject: [PATCH 4/5] docs: correct the Python package documentation Fixes the inaccuracies catalogued in #604 and #668. `packages/overture-schema/README.md` documented `from overture.schema import Building, parse, discover_models, json_schema`. None of it resolves: `overture.schema` is a namespace root shipping only `py.typed` since #622 moved the validation API out of it, and `parse()` exists nowhere. Rewritten against the real surface -- models from their theme packages, `validate`/`validate_json` from `overture.schema.validation`, discovery and `json_schema` from `overture.schema.system`. Three READMEs claimed `model_validate()` accepts GeoJSON. It does not, and the framing above it was also wrong: Overture publishes one shape, flat and tabular. The models additionally read and write GeoJSON so the schema works with tools that expect features rather than rows, and because that is the representation the generated JSON Schema describes. The two modes are not interchangeable, and the error a GeoJSON dict produces ("theme Field required") does not point at the cause. Round-trip examples now pass `by_alias=True`, without which `class_` is emitted and the output will not re-validate. The codegen README described `analyze_type()` returning a `TypeInfo` with `.kind`/`.base_type` and imported `TypeKind`. Neither exists; it returns a `(FieldShape, bool, str | None)` tuple. Replaced with a worked example whose output is transcribed from a run, and corrected the layer diagram and the package credited with `discover_models`. The pyspark README's S3 examples named release 2026-06-17.0, which the bucket no longer holds. Bumped, with a note on reading the current identifier from the STAC catalog, since a hardcoded release goes stale by design. PYDANTIC_GUIDE.md prescribed a `models.py` / `enums.py` / `types.py` split inside a per-type subdirectory (#604). No theme package has such a file. Rewritten against the settled layout: one module per feature type at the theme root, `_common.py` for what a theme shares, a subpackage only for a type large enough to split. The container-to-mixin example now uses the real `Named` / `Appearance` mixins `Building` inherits. It also named two enum members that do not exist -- `Relationship` has `COMPOSITION`, `AGGREGATION`, `HIERARCHY`, `ASSOCIATION`, and `CONNECTS_TO`/`BELONGS_TO` are *roles*. A third instance was in overture-schema-system's README. And its `$defs` migration example placed an `Address` with `freeform`/`locality` in the addresses theme, claiming `Building` carries one; that model is `places.Address`, and `Place` is what holds it. Removed the Reference section's five "Complete Templates": they prescribed the dead file-name convention, and their content survives in Quick Start, Quick Reference, and the Relationship Patterns section. Python embedded in Markdown is now formatted with `ruff format`, so documented snippets match the style of the code they describe. That needs ruff 0.16, so the dependency floor moves from 0.13. Signed-off-by: Seth Fitzsimmons --- PYDANTIC_GUIDE.md | 447 +++++++++--------- README.pydantic.md | 33 +- packages/overture-schema-cli/pyproject.toml | 2 +- packages/overture-schema-codegen/README.md | 52 +- .../changelog.d/668.docs.md | 1 + packages/overture-schema-pyspark/README.md | 20 +- .../changelog.d/668.docs.md | 1 + .../overture-schema-system/pyproject.toml | 2 +- packages/overture-schema-validation/README.md | 10 +- .../changelog.d/668.docs.md | 1 + packages/overture-schema/README.md | 124 +++-- .../overture-schema/changelog.d/668.docs.md | 1 + pyproject.toml | 2 +- uv.lock | 6 +- 14 files changed, 383 insertions(+), 319 deletions(-) create mode 100644 packages/overture-schema-codegen/changelog.d/668.docs.md create mode 100644 packages/overture-schema-pyspark/changelog.d/668.docs.md create mode 100644 packages/overture-schema-validation/changelog.d/668.docs.md create mode 100644 packages/overture-schema/changelog.d/668.docs.md diff --git a/PYDANTIC_GUIDE.md b/PYDANTIC_GUIDE.md index 4571b2afe..3e61c3191 100644 --- a/PYDANTIC_GUIDE.md +++ b/PYDANTIC_GUIDE.md @@ -21,7 +21,6 @@ This guide helps you work with Overture Maps Pydantic schemas - Python models th - [Project Architecture](#project-architecture) - [Migrating from JSON Schema](#migrating-from-json-schema) - [Reference](#reference) - - [Complete Templates](#complete-templates) - [Quick Reference](#quick-reference) --- @@ -42,7 +41,11 @@ from pydantic import BaseModel, Field # Overture common models from overture.schema.common import OvertureFeature -from overture.schema.system.geometric import Geometry, GeometryType, GeometryTypeConstraint +from overture.schema.system.geometric import ( + Geometry, + GeometryType, + GeometryTypeConstraint, +) # Validation system from overture.schema.system.field_constraint import UniqueItemsConstraint @@ -59,9 +62,14 @@ from overture.schema.system.string import LanguageTag # Numeric types (use these instead of int/float) from overture.schema.system.numeric import ( - int8, int32, int64, - uint8, uint16, uint32, - float32, float64 + int8, + int32, + int64, + uint8, + uint16, + uint32, + float32, + float64, ) ``` @@ -73,6 +81,7 @@ from pydantic import BaseModel, Field from overture.schema.system.model_constraint import no_extra_fields from overture.schema.system.numeric import int8, float64 + @no_extra_fields class MyCustomType(BaseModel): """Brief description of what this represents.""" @@ -88,10 +97,8 @@ class MyCustomType(BaseModel): priority: Annotated[ int8 | None, Field( - ge=1, - le=10, - description="Priority level from 1 (lowest) to 10 (highest)" - ) + ge=1, le=10, description="Priority level from 1 (lowest) to 10 (highest)" + ), ] = None ``` @@ -101,7 +108,12 @@ class MyCustomType(BaseModel): from typing import Annotated, Literal from pydantic import Field from overture.schema.common import OvertureFeature -from overture.schema.system.geometric import Geometry, GeometryType, GeometryTypeConstraint +from overture.schema.system.geometric import ( + Geometry, + GeometryType, + GeometryTypeConstraint, +) + class MyFeature(OvertureFeature[Literal["my_theme"], Literal["my_type"]]): """Description of what this feature represents.""" @@ -138,9 +150,11 @@ Pydantic models are Python classes that define data structures and their constra ```python from overture.schema.system.model_constraint import no_extra_fields + @no_extra_fields class Address(BaseModel): """A postal address - no extra fields allowed.""" + street: str city: str postal_code: str | None = None @@ -154,8 +168,10 @@ from typing import Literal from overture.schema.common import OvertureFeature from overture.schema.system.numeric import float64 + class Building(OvertureFeature[Literal["buildings"], Literal["building"]]): """A building feature with strongly-typed theme and type.""" + # Inherits: id, theme, type, geometry, bbox, version, sources height: float64 | None = None ``` @@ -182,7 +198,10 @@ from overture.schema.common.level import Stacked from overture.schema.common.names import Named from overture.schema.system.numeric import float64 -class Building(OvertureFeature[Literal["buildings"], Literal["building"]], Named, Stacked): + +class Building( + OvertureFeature[Literal["buildings"], Literal["building"]], Named, Stacked +): # Gets fields from Feature: id, theme, type, geometry, etc. # Gets fields from Named: names # Gets fields from Stacked: level @@ -198,6 +217,7 @@ Sometimes you need a field name that conflicts with Python keywords or conventio from typing import Annotated from pydantic import Field + class Building(OvertureFeature): # Use class_ in Python code, but "class" in the actual data class_: Annotated[str | None, Field(alias="class")] = None @@ -246,7 +266,7 @@ class Building(OvertureFeature): ```python access_policy: Annotated[ str | None, - Field(description="Access policy for the place. When absent, assume 'open'") + Field(description="Access policy for the place. When absent, assume 'open'"), ] = None ``` @@ -266,26 +286,32 @@ Keep the schema separate from business logic. The schema describes the shape of ```python from overture.schema.system.model_constraint import no_extra_fields from overture.schema.system.numeric import ( - int8, int32, int64, # Signed integers - uint8, uint16, uint32, # Unsigned integers - float32, float64 # Floating point + int8, + int32, + int64, # Signed integers + uint8, + uint16, + uint32, # Unsigned integers + float32, + float64, # Floating point ) + @no_extra_fields class MyModel(BaseModel): # Signed integers with specific ranges - level: int8 | None = None # -128 to 127 - year: int32 | None = None # -2,147,483,648 to 2,147,483,647 - timestamp: int64 | None = None # Full 64-bit range + level: int8 | None = None # -128 to 127 + year: int32 | None = None # -2,147,483,648 to 2,147,483,647 + timestamp: int64 | None = None # Full 64-bit range # Unsigned integers (0 and positive only) - red_value: uint8 | None = None # 0 to 255 (like RGB values) - port: uint16 | None = None # 0 to 65,535 (like network ports) - population: uint32 | None = None # 0 to 4,294,967,295 + red_value: uint8 | None = None # 0 to 255 (like RGB values) + port: uint16 | None = None # 0 to 65,535 (like network ports) + population: uint32 | None = None # 0 to 4,294,967,295 # Floating point numbers - height: float64 | None = None # Double precision (recommended) - ratio: float32 | None = None # Single precision + height: float64 | None = None # Double precision (recommended) + ratio: float32 | None = None # Single precision ``` **When to use each:** @@ -316,6 +342,7 @@ Union types allow a field to accept multiple different types. The `|` symbol mea ```python from typing import Literal + class Building(OvertureFeature): # This field can be either a string OR None (most common union) name: str | None = None @@ -328,7 +355,7 @@ class Building(OvertureFeature): ```python # Optional field (most common union) -height: float64 | None = None # Can be a number or missing +height: float64 | None = None # Can be a number or missing # Specific string values (an alternative to enums where descriptions aren't needed) priority: Literal["low", "medium", "high"] | None = None @@ -364,8 +391,8 @@ height: float64 | None = None # With Annotated - type + extra information height: Annotated[ - float64 | None, # The actual type (what kind of data) - Field(description="Height in meters") # Extra metadata + float64 | None, # The actual type (what kind of data) + Field(description="Height in meters"), # Extra metadata ] = None ``` @@ -384,17 +411,16 @@ Use Pydantic's `Field()` function to add constraints and descriptions: from typing import Annotated from pydantic import Field + class Building(OvertureFeature): # Range constraints height: Annotated[ - float64 | None, - Field(ge=0, le=1000, description="Height in meters (0-1000m)") + float64 | None, Field(ge=0, le=1000, description="Height in meters (0-1000m)") ] = None # Integer constraints floors: Annotated[ - int32 | None, - Field(gt=0, lt=200, description="Number of floors (1-199)") + int32 | None, Field(gt=0, lt=200, description="Number of floors (1-199)") ] = None ``` @@ -412,13 +438,12 @@ class Place(OvertureFeature): # Length constraints name: Annotated[ str | None, - Field(min_length=1, max_length=100, description="Place name (1-100 chars)") + Field(min_length=1, max_length=100, description="Place name (1-100 chars)"), ] = None # Pattern matching postal_code: Annotated[ - str | None, - Field(pattern=r"^\d{5}(-\d{4})?$", description="US postal code") + str | None, Field(pattern=r"^\d{5}(-\d{4})?$", description="US postal code") ] = None ``` @@ -446,12 +471,13 @@ class Building(Feature): ```python from overture.schema.system.field_constraint import UniqueItemsConstraint + class Building(OvertureFeature): # List with size and uniqueness constraints categories: Annotated[ list[str] | None, Field(min_length=1, max_length=10, description="1-10 categories"), - UniqueItemsConstraint() # Must come AFTER Field() + UniqueItemsConstraint(), # Must come AFTER Field() ] = None ``` @@ -487,6 +513,7 @@ Enums define a fixed set of allowed values: ```python from enum import Enum + class BuildingClass(str, Enum): """Further delineation of the building's built purpose.""" @@ -495,6 +522,7 @@ class BuildingClass(str, Enum): INDUSTRIAL = "industrial" CIVIC = "civic" + # Usage in a model class Building(OvertureFeature): class_: Annotated[BuildingClass | None, Field(alias="class")] = None @@ -509,6 +537,7 @@ Use `DocumentedEnum` from `overture.schema.system.doc` when enum members need th ```python from overture.schema.system.doc import DocumentedEnum + class VehicleType(str, DocumentedEnum): """Types of vehicles for transportation.""" @@ -630,8 +659,12 @@ parent_division_id: Annotated[Id, Reference(Relationship.HIERARCHY, Division)] capital_division_ids: Annotated[list[Id], Reference(Relationship.HIERARCHY, Division)] # With role: unambiguous -parent_division_id: Annotated[Id, Reference(Relationship.HIERARCHY, Division, role="child_of")] -capital_division_ids: Annotated[list[Id], Reference(Relationship.HIERARCHY, Division, role="has_as_capital")] +parent_division_id: Annotated[ + Id, Reference(Relationship.HIERARCHY, Division, role="child_of") +] +capital_division_ids: Annotated[ + list[Id], Reference(Relationship.HIERARCHY, Division, role="has_as_capital") +] ``` The `role` must be a non-empty snake_case string (lowercase letters, digits, underscores). It describes the source's role relative to the target using source-perspective phrasing. @@ -650,30 +683,35 @@ from pydantic import Field from overture.schema.common import OvertureFeature from overture.schema.system.ref import Id, Reference, Relationship + # COMPOSITION — part points to its whole class BuildingPart(OvertureFeature[Literal["buildings"], Literal["building_part"]]): """A structural part of a building.""" + building_id: Annotated[ Id, Reference(Relationship.COMPOSITION, Building, role="part_of"), - Field(description="The building to which this part belongs") + Field(description="The building to which this part belongs"), ] + # HIERARCHY — child points to parent class DivisionArea(OvertureFeature[Literal["divisions"], Literal["division_area"]]): """Area polygon nested under a division.""" + division_id: Annotated[ Id, Reference(Relationship.HIERARCHY, Division, role="child_of"), - Field(description="Division ID of the parent division of this area.") + Field(description="Division ID of the parent division of this area."), ] + # ASSOCIATION — peer reference, no ownership class ConnectorReference(BaseModel): """Reference to a connector feature.""" + connector_id: Annotated[ - Id, - Reference(Relationship.ASSOCIATION, Connector, role="connects_to") + Id, Reference(Relationship.ASSOCIATION, Connector, role="connects_to") ] ``` @@ -693,14 +731,8 @@ class AdminCityCenterAssociation( ): """Describes how an administrative area relates to a city center.""" - admin_area_id: Annotated[ - Id, - Reference(Relationship.ASSOCIATION, AdminArea) - ] - city_center_id: Annotated[ - Id, - Reference(Relationship.ASSOCIATION, CityCenter) - ] + admin_area_id: Annotated[Id, Reference(Relationship.ASSOCIATION, AdminArea)] + city_center_id: Annotated[Id, Reference(Relationship.ASSOCIATION, CityCenter)] # Information about the relationship itself relationship_type: Literal["primary_center", "secondary_center"] = "primary_center" @@ -719,16 +751,25 @@ When a feature needs to reference multiple other features, use a list of referen ```python # COMPOSITION — boundary defines two divisions -class DivisionBoundary(OvertureFeature[Literal["divisions"], Literal["division_boundary"]]): +class DivisionBoundary( + OvertureFeature[Literal["divisions"], Literal["division_boundary"]] +): """A boundary line between two divisions.""" + division_ids: Annotated[ - list[Annotated[Id, Reference(Relationship.COMPOSITION, Division, role="boundary_of")]], + list[ + Annotated[ + Id, Reference(Relationship.COMPOSITION, Division, role="boundary_of") + ] + ], Field(min_length=2, max_length=2, description="Left and right divisions"), ] + # AGGREGATION — route groups segments class Route(OvertureFeature[Literal["transportation"], Literal["route"]]): """A transportation route passing through multiple segments.""" + segment_ids: Annotated[ list[Id], Reference(Relationship.AGGREGATION, TransportationSegment, role="groups"), @@ -750,7 +791,7 @@ Include `Reference` annotations for semantic clarity and documentation: division_id: Annotated[ Id, Reference(Relationship.HIERARCHY, Division, role="child_of"), - Field(description="Division ID of the parent division of this area.") + Field(description="Division ID of the parent division of this area."), ] # Avoid — missing semantic information @@ -774,11 +815,15 @@ from typing import Annotated, Literal from pydantic import Field from overture.schema.common import OvertureFeature + # Base class with common fields -class TransportationSegment(OvertureFeature[Literal["transportation"], Literal["segment"]]): +class TransportationSegment( + OvertureFeature[Literal["transportation"], Literal["segment"]] +): subtype: Subtype # This is the discriminator field # ... common fields for all segments + # Specific segment types class RoadSegment(TransportationSegment): subtype: Literal[Subtype.ROAD] # Must be "road" @@ -786,16 +831,17 @@ class RoadSegment(TransportationSegment): speed_limits: SpeedLimits | None = None # ... road-specific fields + class RailSegment(TransportationSegment): subtype: Literal[Subtype.RAIL] # Must be "rail" class_: Annotated[RailClass, Field(alias="class")] rail_flags: RailFlags | None = None # ... rail-specific fields + # Union type that automatically picks the right model based on subtype Segment = Annotated[ - RoadSegment | RailSegment | WaterSegment, - Field(discriminator="subtype") + RoadSegment | RailSegment | WaterSegment, Field(discriminator="subtype") ] ``` @@ -847,7 +893,7 @@ road_segment = RoadSegment(subtype=Subtype.ROAD, ...) # Valid Instead of making classes abstract, we use **entry point registration** where only specific concrete types are discoverable as map features: -```python +```toml # In packages/overture-schema-theme-transportation/pyproject.toml [project.entry-points."overture.models"] connector = "overture.schema.transportation:Connector" @@ -874,25 +920,29 @@ segment = "overture.schema.transportation:Segment" from typing import Annotated from pydantic import BaseModel, Field + @no_extra_fields class Names(BaseModel): primary: str # Keys (strings) must match a language tag pattern, values are strings - common: Annotated[ - dict[ - # The key type - Annotated[ - str, - Field( - pattern=r"^[a-z]{2,3}(-[A-Z]{2})?$", - description="Language tag (e.g., 'en', 'es-MX')" - ) + common: ( + Annotated[ + dict[ + # The key type + Annotated[ + str, + Field( + pattern=r"^[a-z]{2,3}(-[A-Z]{2})?$", + description="Language tag (e.g., 'en', 'es-MX')", + ), + ], + str, # The value type ], - str, # The value type - ], - Field(json_schema_extra={"additionalProperties": False}), - ] | None = None + Field(json_schema_extra={"additionalProperties": False}), + ] + | None + ) = None ``` **Example data:** @@ -918,22 +968,24 @@ The `additionalProperties: False` ensures only keys matching the pattern are all from typing import Annotated from pydantic import Field + # Each item has its own field validation @no_extra_fields class HierarchyItem(BaseModel): division_id: str name: str + class Division(OvertureFeature): # Nested list validation: outer list AND inner lists both have length constraints hierarchies: Annotated[ list[ # Outer list Annotated[ list[HierarchyItem], # Inner list - Field(min_length=1) # Inner list must have at least 1 item + Field(min_length=1), # Inner list must have at least 1 item ] ], - Field(min_length=1) # Outer list must have at least 1 hierarchy + Field(min_length=1), # Outer list must have at least 1 hierarchy ] ``` @@ -969,10 +1021,11 @@ SegmentId = NewType("SegmentId", str) # IDs are strings, but distinct CountryCode = NewType("CountryCode", str) # Country codes are strings, but distinct # Create aliases for complex field patterns -EmailList = NewType("EmailList", Annotated[ - list[str], - Field(min_length=1, description="List of email addresses") -]) +EmailList = NewType( + "EmailList", + Annotated[list[str], Field(min_length=1, description="List of email addresses")], +) + @no_extra_fields class Contact(BaseModel): @@ -997,44 +1050,75 @@ class Contact(BaseModel): #### File Organization -Organize code by scope and avoid circular imports: +Organize code by scope, and avoid circular imports. -**Cross-theme shared**: `overture-schema-common` package +**Cross-theme shared**: the `overture-schema-common` package. Definitions more than +one theme needs -- `OvertureFeature`, `Names`, `Sources`, the scoping framework. -- Used by multiple themes (e.g., `OvertureFeature`, `Names`, `Sources`, `Scope`) +**One module per feature type**: at the theme package root, named after the type in +snake_case. + +```text +packages/overture-schema-theme-buildings/src/overture/schema/buildings/ + __init__.py # re-exports the public names, declares __all__ + _common.py # shared by Building and BuildingPart + building.py # Building, BuildingSubtype, BuildingClass + building_part.py # BuildingPart +``` -**Theme-level shared**: Theme package root (e.g., `overture-schema-theme-transportation/src/overture/schema/transportation/`) +The module owns everything specific to its type: the `Feature` subclass, its enums, +its NewTypes, and its supporting models. `building.py` defines `BuildingSubtype` and +`BuildingClass` next to `Building`, because nothing else uses them. -- Used by multiple types within a theme (e.g., `AccessRules`, `RoadSurface`) +**Theme-level shared**: `_common.py` at the theme package root, for definitions two +or more types in the theme need. `buildings/_common.py` holds `Appearance`, +`RoofShape`, and the material enums that both `Building` and `BuildingPart` use. The +leading underscore marks the module private -- the theme's `__init__.py` re-exports +the public names from it. -**Type-specific**: Type subdirectory (e.g., `overture-schema-theme-transportation/src/overture/schema/transportation/segment/`) +**A type large enough to split**: a subpackage named after the type, applying the +same rules one level down. -- Only used by one specific type (e.g., `SegmentType`, `LaneConfiguration`) +```text +packages/overture-schema-theme-transportation/src/overture/schema/transportation/ + __init__.py # re-exports from connector and segment + connector.py # Connector + segment/ + __init__.py # assembles the Segment discriminated union, re-exports + _common.py # TransportationSegment and what the arms share + road.py # RoadSegment and its supporting types + rail.py # RailSegment and its supporting types + water.py # WaterSegment +``` -**File type rules:** +Every `__init__.py` re-exports its public names and declares `__all__`, so consumers +import from the package rather than reaching into the defining module: +`from overture.schema.buildings import Building`, not +`from overture.schema.buildings.building import Building`. Entry points name the +package root for the same reason -- `building = "overture.schema.buildings:Building"`. -- **`models.py`**: Pydantic model classes (and type aliases that reference models) -- **`enums.py`**: Enum classes only, no project imports -- **`types.py`**: Type aliases that don't reference models, no project imports +Modules are named for the thing they define, not the kind of thing: an enum lives in +the module whose type uses it, and moves up to `_common.py` once a second type needs it. #### Import Organization ```python # Standard library imports first -from typing import Annotated, Literal, NewType from enum import Enum +from typing import Annotated, Literal, NewType # Third-party imports from pydantic import BaseModel, ConfigDict, Field -# Cross-theme imports -from overture.schema.common import OvertureFeature +# Cross-theme and system imports +from overture.schema.common.scoping import Heading, Scope, scoped +from overture.schema.system.doc import DocumentedEnum from overture.schema.system.field_constraint import UniqueItemsConstraint from overture.schema.system.model_constraint import no_extra_fields -# Local imports last -from .enums import SegmentType -from .types import SegmentId, LaneWidth # Only non-model type aliases +# Local imports last -- siblings in the theme, then the module's own package +from ..connector import Connector +from ._common import SegmentSubtype, TransportationSegment ``` `uv run ruff format ` will sort your imports in this order automatically. @@ -1045,20 +1129,22 @@ This project uses a custom validation system that generates better JSON Schema o ```python # Don't do this -@field_validator('categories') +@field_validator("categories") def validate_categories_unique(cls, v): if v and len(v) != len(set(v)): - raise ValueError('Categories must be unique') + raise ValueError("Categories must be unique") return v + # Do this instead from overture.schema.system.field_constraint import UniqueItemsConstraint + class Building(OvertureFeature): categories: Annotated[ list[str] | None, Field(min_length=1, description="Building categories"), - UniqueItemsConstraint() + UniqueItemsConstraint(), ] = None ``` @@ -1088,16 +1174,18 @@ properties: **Pydantic approach:** ```python -# In overture-schema-theme-addresses/src/overture/schema/addresses/address.py +# In overture-schema-theme-places/src/overture/schema/places/place.py @no_extra_fields class Address(BaseModel): """A postal address.""" + freeform: str | None = None locality: str | None = None -# In overture-schema-theme-buildings/src/overture/schema/buildings/building.py -class Building(OvertureFeature): - address: Address | None = None + +# Same module -- Place is the type that carries one. +class Place(OvertureFeature): + addresses: list[Address] | None = None ``` **Primary differences:** @@ -1131,20 +1219,31 @@ allOf: **Pydantic equivalent** uses **mixin classes**: ```python -# In common/names.py +# namesContainer -> Named, in +# overture-schema-common/src/overture/schema/common/names.py class Named(BaseModel): """Properties defining the names of a feature.""" + names: Names | None = None -# In buildings/models.py -class Shape(BaseModel): - """Properties of the building's shape.""" + +# shapeContainer -> Appearance, in buildings/_common.py, +# shared by Building and BuildingPart +class Appearance(BaseModel): + """Physical and visual properties of a building.""" + height: float64 | None = None num_floors: int32 | None = None + # ... roof and facade fields + -# Usage with multiple inheritance -class Building(Feature, Named, Shape): - pass # "pass" means "do nothing" - Building inherits names, height, num_floors, etc. from its parents +# Usage with multiple inheritance -- this is how Building is actually declared +class Building( + OvertureFeature[Literal["buildings"], Literal["building"]], + Named, + Stacked, + Appearance, +): ... # inherits names, level, height, num_floors, and the rest from its parents ``` JSON Schema containers become **mixin classes** in Pydantic that you inherit from. @@ -1166,147 +1265,25 @@ JSON Schema containers become **mixin classes** in Pydantic that you inherit fro ## Reference -### Complete Templates - -#### Basic Model Template - -```python models.py -from typing import Annotated -from pydantic import BaseModel, Field -from overture.schema.system.model_constraint import no_extra_fields -from overture.schema.system.numeric import int8, float64 - -@no_extra_fields -class MyCustomType(BaseModel): - """Brief description of what this represents.""" - - # Required fields (no default value) - name: str - category: str - - # Optional fields (with default values) - description: str | None = None - - # Field with constraints and description - priority: Annotated[ - int8 | None, - Field( - ge=1, - le=10, - description="Priority level from 1 (lowest) to 10 (highest)" - ) - ] = None -``` - -#### Feature Template - -```python models.py -from typing import Annotated, Literal -from pydantic import Field -from overture.schema.common import OvertureFeature -from overture.schema.system.geometric import Geometry, GeometryType, GeometryTypeConstraint - -class MyFeature(OvertureFeature[Literal["my_theme"], Literal["my_type"]]): - """Description of what this feature represents.""" - - # Geometry with constraints - geometry: Annotated[ - Geometry, - GeometryTypeConstraint(GeometryType.POINT), - Field(description="Location of this feature"), - ] - - # Custom fields - my_field: str | None = None -``` - -#### Enum Template - -```python enums.py -from enum import Enum - -class MyEnum(str, Enum): - """Description of what this enum represents.""" - - VALUE_ONE = "value_one" - VALUE_TWO = "value_two" - VALUE_THREE = "value_three" -``` - -#### Model with Validation Constraints - -```python models.py -from typing import Annotated -from pydantic import BaseModel, Field -from overture.schema.system.field_constraint import UniqueItemsConstraint -from overture.schema.system.model_constraint import no_extra_fields - -@no_extra_fields -class Contact(BaseModel): - """Contact information with validation constraints.""" - - name: str - email: str | None = None - phone: str | None = None - - # List with constraints - tags: Annotated[ - list[str] | None, - Field(min_length=1, description="Contact tags"), - UniqueItemsConstraint() # No duplicate tags - ] = None -``` - -#### Association Feature Template - -```python models.py -from typing import Annotated, Literal -from pydantic import Field -from overture.schema.common import OvertureFeature -from overture.schema.system.numeric import float64 -from overture.schema.system.ref import Id, Reference, Relationship - -class MyAssociation(OvertureFeature[Literal["associations"], Literal["my_association"]]): - """Represents a relationship between two features with metadata.""" - - # References to the associated features - feature_a_id: Annotated[ - Id, - Reference(Relationship.CONNECTS_TO, FeatureA), - Field(description="First feature in the relationship") - ] - - feature_b_id: Annotated[ - Id, - Reference(Relationship.CONNECTS_TO, FeatureB), - Field(description="Second feature in the relationship") - ] - - # Relationship metadata - relationship_type: Literal["primary", "secondary"] = "primary" - confidence: Annotated[float64 | None, Field(ge=0.0, le=1.0)] = None - - # Optional contextual information - notes: str | None = None -``` - ### Quick Reference #### Essential Patterns (Most Common) ```python # Basic field types -name: str # Required string -name: str | None = None # Optional string -count: int32 # Required integer +name: str # Required string +name: str | None = None # Optional string +count: int32 # Required integer priority: Literal["high", "medium", "low"] | None = None # Constrained values # Validated fields height: Annotated[float64 | None, Field(ge=0, description="Height in meters")] = None tags: Annotated[list[str] | None, Field(min_length=1), UniqueItemsConstraint()] = None -# Association patterns -parent_id: Annotated[Id | None, Reference(Relationship.BELONGS_TO, ParentModel)] = None +# Association patterns -- Relationship is the kind, role is the meaning +parent_id: Annotated[ + Id | None, Reference(Relationship.HIERARCHY, ParentModel, role="child_of") +] = None connector_ids: list[Id] # References to multiple related features ``` @@ -1319,11 +1296,13 @@ class Address(BaseModel): street: str city: str | None = None + # Feature model class Building(OvertureFeature[Literal["buildings"], Literal["building"]]): geometry: Geometry height: float64 | None = None + # Enum class Status(str, Enum): ACTIVE = "active" diff --git a/README.pydantic.md b/README.pydantic.md index 866e39670..1b0376f75 100644 --- a/README.pydantic.md +++ b/README.pydantic.md @@ -116,21 +116,28 @@ Install the main package using `pip` (or your package manager of choice): pip install overture-schema ``` +Overture publishes data in one shape: flat and tabular, the column layout of the +Parquet release, which Pydantic's Python mode reads. The models also accept and +emit GeoJSON, through JSON mode, so the schema works with tools that expect +features rather than rows -- and it is the representation the generated JSON +Schema describes. The modes are not interchangeable: a GeoJSON dict passed to +`model_validate` reports `theme` and `version` missing and `type` set to +`'Feature'`. + ```python -from overture.schema.buildings.building import Building -from overture.schema.places.place import Place -import json +from overture.schema.buildings import Building +from overture.schema.places import Place -# Validate data - supports both flat/tabular- (Parquet-style) and GeoJSON-formatted -# dicts -building = Building.model_validate(feature_data) -building_geojson = Building.model_validate(geojson_feature) +# Flat / tabular dict -- Python mode +building = Building.model_validate(feature_row) -# Parse and validate JSON strings -building_from_json = Building.model_validate_json(json_string) +# GeoJSON, as a string or bytes -- JSON mode +building = Building.model_validate_json(geojson_text) -# Convert to GeoJSON format -geojson_output = building.model_dump(mode="json") +# Serialize back to GeoJSON. by_alias=True is required: without it, aliased +# fields serialize under their Python names (`class_`, not `class`) and the +# output will not re-validate. +geojson_output = building.model_dump(mode="json", by_alias=True, exclude_none=True) ``` ## Schema Extension @@ -187,9 +194,9 @@ from overture.schema.system.discovery import ( models = discover_models() # { # ModelKey(name="building", entry_point="overture.schema.buildings:Building", -# tags=frozenset({"feature", "overture", "overture:theme=buildings"})): BuildingModel, +# tags=frozenset({"feature", "overture", "overture:theme=buildings"})): Building, # ModelKey(name="place", entry_point="overture.schema.places:Place", -# tags=frozenset({"feature", "overture", "overture:theme=places"})): PlaceModel, +# tags=frozenset({"feature", "overture", "overture:theme=places"})): Place, # ... # } diff --git a/packages/overture-schema-cli/pyproject.toml b/packages/overture-schema-cli/pyproject.toml index e5642bc96..783c5d75c 100644 --- a/packages/overture-schema-cli/pyproject.toml +++ b/packages/overture-schema-cli/pyproject.toml @@ -34,7 +34,7 @@ build-backend = "uv_build" [dependency-groups] dev = [ "pytest>=9.0.0", - "ruff>=0.13.0", + "ruff>=0.16.0", "mypy>=1.17.0", ] diff --git a/packages/overture-schema-codegen/README.md b/packages/overture-schema-codegen/README.md index 3252dfffd..22384e38a 100644 --- a/packages/overture-schema-codegen/README.md +++ b/packages/overture-schema-codegen/README.md @@ -10,10 +10,10 @@ structure collapses into `anyOf` arrays with duplicated fields. Navigating Python's type annotation machinery -- NewType chains, nested `Annotated` wrappers, union filtering, generic resolution -- is complex. The codegen does it once. -`analyze_type()` unwraps annotations into `TypeInfo`, a flat target-independent -representation. Extractors build specs from `TypeInfo`. Renderers consume specs without -touching the type system. New output targets (Arrow schemas, PySpark expressions) add -renderers, not extraction logic. +`analyze_type()` unwraps an annotation into a `FieldShape`, a tree-shaped +target-independent representation. Extractors build specs from `FieldShape`. Renderers +consume specs without touching the type system. New output targets (Arrow schemas, +PySpark expressions) add renderers, not extraction logic. ## Usage @@ -41,9 +41,9 @@ Rendering Output formatting, all presentation decisions ^ Output Layout What to generate, where it goes, how outputs link ^ -Extraction TypeInfo, FieldSpec, RecordSpec, UnionSpec +Extraction FieldShape, FieldSpec, RecordSpec, UnionSpec ^ -Discovery discover_models() from overture-schema-common +Discovery discover_models() from overture-schema-system ``` **Discovery** loads registered Pydantic models via entry points. The return dict @@ -72,16 +72,38 @@ examples), enum pages, NewType pages, and aggregate numeric/geometry reference p ## Programmatic use +`analyze_type()` returns a 3-tuple: the annotation's `FieldShape`, whether the +field accepts `None`, and the first `Field(description=...)` encountered while +unwrapping. + +```python +from overture.schema.buildings import Building +from overture.schema.codegen.extraction.type_analyzer import analyze_type + +annotation = Building.model_fields["version"].rebuild_annotation() +shape, nullable, description = analyze_type(annotation, owner=Building) + +# NewTypeShape(name='FeatureVersion', ref=..., inner=Primitive(base_type='int32', ...)) +assert shape.name == "FeatureVersion" +assert shape.inner.base_type == "int32" +assert nullable is False +``` + +`FieldShape` is a tree, not a flat record: `NewTypeShape`, `ArrayOf`, and `MapOf` +wrap an inner shape, and the three terminals (`Primitive`, `LiteralScalar`, +`AnyScalar`) sit at the leaves. Nesting order is meaningful -- +`NewTypeShape(inner=ArrayOf(...))` is a NewType over `list[X]`, while +`ArrayOf(element=NewTypeShape(...))` is a list of NewType-wrapped values. + +Constraints attach to the layer they target and carry the NewType that +contributed them: + ```python -from overture.schema.codegen.extraction.type_analyzer import analyze_type, TypeKind - -info = analyze_type(some_annotation) -assert info.kind == TypeKind.PRIMITIVE -assert info.base_type == "int32" -assert info.newtype_name == "FeatureVersion" -# Constraints carry provenance: -for cs in info.constraints: - print(f"{cs.constraint} from {cs.source}") +for source in shape.inner.constraints: + print(f"{source.constraint} from {source.source_name}") +# Ge(ge=0) from FeatureVersion +# Ge(ge=-2147483648) from int32 +# Le(le=2147483647) from int32 ``` ## Fetching sample data diff --git a/packages/overture-schema-codegen/changelog.d/668.docs.md b/packages/overture-schema-codegen/changelog.d/668.docs.md new file mode 100644 index 000000000..88e7d94f0 --- /dev/null +++ b/packages/overture-schema-codegen/changelog.d/668.docs.md @@ -0,0 +1 @@ +Rewrote the README's programmatic-use section against the current `analyze_type()` signature, which returns a `FieldShape` tuple rather than the removed `TypeInfo`/`TypeKind`. diff --git a/packages/overture-schema-pyspark/README.md b/packages/overture-schema-pyspark/README.md index 13981ffba..85c85b24a 100644 --- a/packages/overture-schema-pyspark/README.md +++ b/packages/overture-schema-pyspark/README.md @@ -65,7 +65,7 @@ overture-validate segment samples/segment.parquet \ --conf spark.master=local[4] # Continue past schema mismatches (e.g. Float vs Double on bbox) -overture-validate place s3a://overturemaps-us-west-2/release/2026-06-17.0 \ +overture-validate place s3a://overturemaps-us-west-2/release/2026-07-22.0 \ --skip-schema-check # Skip checks for a column absent from the data @@ -106,15 +106,15 @@ structure: | --- | --- | --- | | Hive partition path (contains `/theme=`) | `.../theme=transportation/type=segment/` | Reads directly; derives `basePath` so Spark discovers partition columns. | | Individual file | `segment.parquet` | Reads directly; data already contains `theme`/`type` columns. | -| Release root | `s3a://overturemaps-us-west-2/release/2026-06-17.0` | Appends `theme={theme}/type={type}` using the schema's theme mapping; sets `basePath` to the original path. | +| Release root | `s3a://overturemaps-us-west-2/release/2026-07-22.0` | Appends `theme={theme}/type={type}` using the schema's theme mapping; sets `basePath` to the original path. | This means you can point the CLI at a release root and it constructs the full Hive path automatically: ```bash # These are equivalent: -overture-validate segment s3a://overturemaps-us-west-2/release/2026-06-17.0 -overture-validate segment s3a://overturemaps-us-west-2/release/2026-06-17.0/theme=transportation/type=segment/ +overture-validate segment s3a://overturemaps-us-west-2/release/2026-07-22.0 +overture-validate segment s3a://overturemaps-us-west-2/release/2026-07-22.0/theme=transportation/type=segment/ ``` ### Reading from S3 @@ -126,14 +126,22 @@ the Overture release bucket: ```bash overture-validate segment \ - s3a://overturemaps-us-west-2/release/2026-06-17.0/theme=transportation/type=segment/ + s3a://overturemaps-us-west-2/release/2026-07-22.0/theme=transportation/type=segment/ +``` + +The bucket retains only recent releases -- older identifiers, including ones +named in earlier revisions of this file, have been removed. Read the current +one from the STAC catalog's `latest` field: + +```bash +curl -s https://stac.overturemaps.org/catalog.json | jq -r .latest ``` To use named AWS credentials instead of anonymous access: ```bash overture-validate segment \ - s3a://overturemaps-us-west-2/release/2026-06-17.0/theme=transportation/type=segment/ \ + s3a://overturemaps-us-west-2/release/2026-07-22.0/theme=transportation/type=segment/ \ --conf spark.hadoop.fs.s3a.aws.credentials.provider=software.amazon.awssdk.auth.credentials.ProfileCredentialsProvider ``` diff --git a/packages/overture-schema-pyspark/changelog.d/668.docs.md b/packages/overture-schema-pyspark/changelog.d/668.docs.md new file mode 100644 index 000000000..ce95a6981 --- /dev/null +++ b/packages/overture-schema-pyspark/changelog.d/668.docs.md @@ -0,0 +1 @@ +Updated the README's S3 examples to a release the bucket still retains, and documented how to read the current release identifier from the STAC catalog. diff --git a/packages/overture-schema-system/pyproject.toml b/packages/overture-schema-system/pyproject.toml index 613349cef..952d299ec 100644 --- a/packages/overture-schema-system/pyproject.toml +++ b/packages/overture-schema-system/pyproject.toml @@ -26,7 +26,7 @@ Issues = "https://github.com/OvertureMaps/schema/issues" [dependency-groups] dev = [ "pytest>=9.0.0", - "ruff>=0.13.0", + "ruff>=0.16.0", "mypy>=1.17.0", ] diff --git a/packages/overture-schema-validation/README.md b/packages/overture-schema-validation/README.md index 41cda9f8b..f6fdb7220 100644 --- a/packages/overture-schema-validation/README.md +++ b/packages/overture-schema-validation/README.md @@ -15,11 +15,13 @@ pip install overture-schema-validation ```python from overture.schema.validation import validate, validate_json -# Validate a Python object (a dict or a model instance) -feature = validate({"type": "segment", "id": "...", "geometry": "..."}) +# A Python object -- the flat, tabular (Parquet-style) shape +feature = validate(feature_row) -# Validate a JSON document -feature = validate_json('{"type": "segment", "id": "...", "geometry": "..."}') +# A JSON document -- GeoJSON +feature = validate_json(geojson_text) ``` +The two entry points are not interchangeable. `validate` runs Pydantic's Python mode, which reads the flat column layout of the Parquet release -- the shape Overture publishes. `validate_json` runs JSON mode, which reads the GeoJSON representation the models support for compatibility with tools that expect features rather than rows. Handing a GeoJSON dict to `validate` reports `theme` and `version` missing and `type` set to `'Feature'`. + Both raise `pydantic.ValidationError` when the input matches no model. Which models participate is resolved at runtime by entry-point discovery, so installing additional Overture theme packages widens what these functions accept. diff --git a/packages/overture-schema-validation/changelog.d/668.docs.md b/packages/overture-schema-validation/changelog.d/668.docs.md new file mode 100644 index 000000000..014c31693 --- /dev/null +++ b/packages/overture-schema-validation/changelog.d/668.docs.md @@ -0,0 +1 @@ +Corrected the README's usage examples, which showed `validate_json` accepting the flat tabular shape rather than GeoJSON. diff --git a/packages/overture-schema/README.md b/packages/overture-schema/README.md index 4a725eda8..d3d55c6b7 100644 --- a/packages/overture-schema/README.md +++ b/packages/overture-schema/README.md @@ -10,76 +10,118 @@ This package provides Pydantic models for validating and working with Overture M pip install overture-schema ``` +`overture-schema` is a metapackage: it pulls in every theme package plus the +validation library and the CLI, and ships no code of its own. `overture.schema` +is a namespace root, so import from the theme and system packages rather than +from `overture.schema` directly. + ## Usage -Import and use schemas: +Import models from the theme package that defines them: ```python -from overture.schema import Building, Place -import json +from overture.schema.buildings import Building +from overture.schema.places import Place +``` + +### Tabular data, and GeoJSON for compatibility + +Overture publishes data in one shape: flat and tabular -- the column layout of the +Parquet release, with `theme`, `type`, and `version` as top-level columns and +geometry as WKT. That is what **Python mode** (`model_validate`) reads. + +The models also accept and emit GeoJSON, through **JSON mode** +(`model_validate_json`), so the schema works with tools that expect features +rather than rows. The generated JSON Schema describes that representation. + +The modes are not interchangeable. Passing a GeoJSON dict to `model_validate` +reports `theme`/`version` missing and `type` set to `'Feature'`, because it is +reading GeoJSON keys as flat columns. -# Validate Overture Maps data (supports both flat/tabular and GeoJSON formats) -building = Building.model_validate(feature_data) -place = Place.model_validate(geojson_feature) +```python +# Flat / tabular (Parquet-shaped) dict +building = Building.model_validate(feature_row) -# Parse and validate JSON strings -building_from_json = Building.model_validate_json(json_string) +# GeoJSON -- JSON mode, from a string or bytes +building = Building.model_validate_json(geojson_text) -# Convert to GeoJSON format for output -geojson_output = building.model_dump(mode="json") +# Serialize back to GeoJSON. by_alias=True is required: without it, +# aliased fields serialize under their Python names (`class_`, not +# `class`) and the output will not re-validate. +geojson_output = building.model_dump(mode="json", by_alias=True, exclude_none=True) ``` ### Available Models +Each model lives in its theme package. The metapackage installs all of them: + ```python -# All models are re-exported from their respective theme packages for convenience -from overture.schema import ( - # Addresses theme - Address, - # Base theme +from overture.schema.addresses import Address +from overture.schema.base import ( Bathymetry, Infrastructure, Land, LandCover, LandUse, Water, - # Buildings theme - Building, - BuildingPart, - # Divisions theme - Division, - DivisionArea, - DivisionBoundary, - # Places theme - Place, - # Transportation theme - Connector, - Segment, ) +from overture.schema.buildings import Building, BuildingPart +from overture.schema.divisions import Division, DivisionArea, DivisionBoundary +from overture.schema.places import Place +from overture.schema.transportation import Connector, Segment +``` + +`Segment` is a discriminated union alias rather than a class, so it validates +through a `TypeAdapter`: + +```python +from pydantic import TypeAdapter + +segments = TypeAdapter(Segment) +segment = segments.validate_json(geojson_text) ``` -### Utility Functions +### Validating without knowing the type -The package also exports several utility functions: +`overture-schema-validation` validates against the union of every installed +model, picking the right one from the data: ```python -from overture.schema import parse, discover_models, json_schema -from overture.schema import Building +from overture.schema.validation import validate, validate_json + +feature = validate(feature_row) # flat / tabular dict +feature = validate_json(geojson_text) # GeoJSON +``` -# Parse any Overture feature (auto-discovers all registered models) -validated_feature = parse(feature_data, mode="json") # Parses GeoJSON format -validated_feature = parse(feature_data, mode="python") # Parses flat format +### Discovering models programmatically + +Discovery lives in `overture-schema-system`. `discover_models()` returns a dict +keyed by `ModelKey` -- entry point `name`, its `entry_point` value, and the set +of tags attached during discovery: + +```python +from overture.schema.system.discovery import discover_models, get_registered_model -# Discover all registered models programmatically all_models = discover_models() -# Returns: # { -# ("buildings", "building"): BuildingModel, -# ("places", "place"): PlaceModel, -# ... +# ModelKey(name="building", entry_point="overture.schema.buildings:Building", +# tags=frozenset({"feature", "overture", "overture:theme=buildings"})): Building, +# ModelKey(name="place", entry_point="overture.schema.places:Place", +# tags=frozenset({"feature", "overture", "overture:theme=places"})): Place, +# ... # } -# Generate JSON Schema for models or unions +building_model = get_registered_model("building") # None if not installed +``` + +### Generating JSON Schema + +```python +from overture.schema.system.json_schema import json_schema + schema = json_schema(Building) -union_schema = json_schema(Building | Place) # Works with unions too +union_schema = json_schema(Building | Place) # emits an anyOf ``` + +See the [`overture-schema-system` README](../overture-schema-system/README.md) +for tag format, tag providers, and the discovery API in full. diff --git a/packages/overture-schema/changelog.d/668.docs.md b/packages/overture-schema/changelog.d/668.docs.md new file mode 100644 index 000000000..d2c7340f6 --- /dev/null +++ b/packages/overture-schema/changelog.d/668.docs.md @@ -0,0 +1 @@ +Rewrote the README against the real import surface: `overture.schema` is a namespace root, so models import from their theme packages and the utility functions from `overture.schema.validation` and `overture.schema.system`. diff --git a/pyproject.toml b/pyproject.toml index bbfe8e01e..b9d6d08bf 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -64,7 +64,7 @@ dev = [ "pytest>=9.0.0", "pytest-cov>=7.0.0", "pytest-testmon>=2.2.0", - "ruff>=0.13.0", + "ruff>=0.16.0", "towncrier>=25.8.0", ] diff --git a/uv.lock b/uv.lock index 0b918f474..3ae73f7f8 100644 --- a/uv.lock +++ b/uv.lock @@ -963,7 +963,7 @@ requires-dist = [ dev = [ { name = "mypy", specifier = ">=1.17.0" }, { name = "pytest", specifier = ">=9.0.0" }, - { name = "ruff", specifier = ">=0.13.0" }, + { name = "ruff", specifier = ">=0.16.0" }, ] [[package]] @@ -1098,7 +1098,7 @@ requires-dist = [ dev = [ { name = "mypy", specifier = ">=1.17.0" }, { name = "pytest", specifier = ">=9.0.0" }, - { name = "ruff", specifier = ">=0.13.0" }, + { name = "ruff", specifier = ">=0.16.0" }, ] [[package]] @@ -1259,7 +1259,7 @@ dev = [ { name = "pytest", specifier = ">=9.0.0" }, { name = "pytest-cov", specifier = ">=7.0.0" }, { name = "pytest-testmon", specifier = ">=2.2.0" }, - { name = "ruff", specifier = ">=0.13.0" }, + { name = "ruff", specifier = ">=0.16.0" }, { name = "towncrier", specifier = ">=25.8.0" }, ] From e7325f96d5dfc473c1f6c7bfcd780095232374a7 Mon Sep 17 00:00:00 2001 From: Seth Fitzsimmons Date: Wed, 19 Aug 2026 11:55:38 -0700 Subject: [PATCH 5/5] test: check the imports and enum members the docs name Documentation drifts silently: a module moves, a helper is renamed, and the README keeps confidently describing the old surface. #489, #604 and #668 are all instances. This makes the class fail the suite instead of accumulating. `tests/test_documented_imports.py` parses every fenced Python block in every tracked Markdown file and checks two things: that each `overture.*` import resolves, and that each `Enum.MEMBER` named exists on the enum. The second is not redundant -- `from ... import Relationship` resolves whether or not `Relationship.CONNECTS_TO` does. Executing the blocks outright is not viable: most fail on undefined names, because a documentation fragment legitimately omits its imports and uses placeholder names the prose supplies. Imports and enum members are the granularity that separates a defect from a fragment. Block extraction and import parsing are separate, unit-tested functions. The corpus cannot test either: a form the matcher misses simply yields nothing, and a sweep that collects less still passes -- so the fixtures are inline rather than drawn from repo content. Blocks indented inside list items are dedented before parsing, since an `IndentationError` is a `SyntaxError` and would otherwise be filed as "expected unparseable". Blocks that are deliberately not valid Python are pinned by a digest of their body rather than a count, which would stay put when one breaks as another is fixed; and an excused block may not contain an `overture` import, so the waiver cannot swallow the API citation it excuses. The tests live outside `packages/` because their subject is the repo's Markdown, including root-level files belonging to no package, and they import across every package -- `overture.schema.codegen` among them, which no single distribution depends on. `tests/README.md` records the admission rule. `make check` reaches the tree through ruff, `ruff format`, mypy and pytest; a bare `pytest packages/` does not. `lint-only` also checks Python embedded in Markdown, with `make format` as the counterpart. At the old ruff floor that check silently finds no files, and CI resolves one job at `lowest-direct`. Signed-off-by: Seth Fitzsimmons --- Makefile | 23 +- pyproject.toml | 2 +- tests/README.md | 23 ++ tests/test_documented_imports.py | 368 +++++++++++++++++++++++++++++++ 4 files changed, 408 insertions(+), 8 deletions(-) create mode 100644 tests/README.md create mode 100644 tests/test_documented_imports.py diff --git a/Makefile b/Makefile index 6a0d92a48..8a50ff6d1 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: default uv-sync clean-pyspark generate-pyspark check check-namespace test-all test test-only docformat docformat-only doctest doctest-only mypy mypy-only lint-only update-baselines +.PHONY: default uv-sync clean-pyspark generate-pyspark check check-namespace test-all test test-only docformat docformat-only doctest doctest-only mypy mypy-only lint-only format update-baselines TESTMON ?= --testmon @@ -47,13 +47,13 @@ check-namespace: # the PySpark output first: that tree is no longer tracked in git, so the # generated conformance tests only exist once generation has run. test-all: uv-sync generate-pyspark - @uv run pytest -W error packages/ + @uv run pytest -W error packages/ tests/ test: uv-sync - @uv run pytest -W error $(TESTMON) packages/ -x -q --tb=short + @uv run pytest -W error $(TESTMON) packages/ tests/ -x -q --tb=short test-only: - @uv run pytest -W error $(TESTMON) packages/ -x -q --tb=short + @uv run pytest -W error $(TESTMON) packages/ tests/ -x -q --tb=short coverage: uv-sync @uv run pytest packages/ --cov overture.schema --cov-report=term --cov-report=html && open htmlcov/index.html @@ -90,11 +90,20 @@ mypy-only: | tr - . \ | sed 's|^packages/|-p |' \ | xargs uv run mypy --no-error-summary - @for d in packages/*/tests; do find "$$d" -name "*.py" | sort | xargs uv run mypy --no-error-summary || exit 1; done + @for d in packages/*/tests tests; do find "$$d" -name "*.py" | sort | xargs uv run mypy --no-error-summary || exit 1; done lint-only: - @uv run ruff check -q packages/ - @uv run ruff format --check packages/ + @uv run ruff check -q packages/ tests/ + @uv run ruff format --check packages/ tests/ + @# Python embedded in Markdown -- keeps documented snippets in the + @# same style as the code they describe. + @git ls-files -z '*.md' | xargs -0 --no-run-if-empty uv run ruff format -q --check + +# The fixing counterpart to lint-only: what to run when that gate fires. +format: + @uv run ruff check -q --fix packages/ tests/ + @uv run ruff format -q packages/ tests/ + @git ls-files -z '*.md' | xargs -0 --no-run-if-empty uv run ruff format -q update-baselines: @uv run pytest --update-baselines -m baseline -q packages/ diff --git a/pyproject.toml b/pyproject.toml index b9d6d08bf..b825787d9 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -45,7 +45,7 @@ select = [ [tool.mypy] disallow_untyped_defs = true explicit_package_bases = true -files = "packages/**/*.py" +files = ["packages/**/*.py", "tests/**/*.py"] # The pyspark test tree is PEP 420 and its tests import `_support` as a # top-level package (mirroring pytest's pythonpath). Put that tests dir on # the mypy path so `_support` resolves the same way for the type checker. diff --git a/tests/README.md b/tests/README.md new file mode 100644 index 000000000..bfc4caf6a --- /dev/null +++ b/tests/README.md @@ -0,0 +1,23 @@ +# Workspace tests + +Checks whose subject is the repository or the workspace as a whole, and which +therefore cannot correctly live in any one package. + +A test belongs here only if both hold: + +- **Its subject spans packages, or sits outside them.** `test_documented_imports` + reads every tracked Markdown file, including root-level ones that ship in no + package, and imports across every package's namespace — `overture.schema.codegen` + among them, which no single distribution depends on. +- **It cannot run from one package's install.** Placing such a test under + `packages//tests/` would make that package's suite depend on packages its + `pyproject.toml` does not declare, and it would pass only because the workspace + happens to install everything. + +Anything that can be scoped to a package belongs in that package's `tests/` +instead. This directory is not a home for tests that are merely inconvenient to +place. + +`make check` covers this tree the same way it covers `packages/` — ruff, `ruff +format`, mypy, and pytest. Note that a bare `pytest packages/` does not; use +`make test` or include `tests/` explicitly. diff --git a/tests/test_documented_imports.py b/tests/test_documented_imports.py new file mode 100644 index 000000000..783e6af44 --- /dev/null +++ b/tests/test_documented_imports.py @@ -0,0 +1,368 @@ +"""Every `overture.*` import written in repo Markdown must resolve. + +Documentation drifts silently: a module moves, a helper is renamed, and the +README keeps confidently describing the old surface. This parses every fenced +Python block in every tracked Markdown file -- including blocks indented inside +a list item -- and imports each `overture.*` statement it finds. + +The subject is the repo's Markdown rather than any one package, so this lives +outside `packages/`. It reaches across the whole workspace -- a README may cite +`overture.schema.codegen`, which no single distribution depends on. +""" + +from __future__ import annotations + +import ast +import hashlib +import importlib +import re +import subprocess +import textwrap +from enum import Enum +from pathlib import Path + +import pytest + +# A fenced ```python block, capturing its body. The indent group matches a +# block nested in a list item; the body is dedented before parsing, since an +# `IndentationError` is a `SyntaxError` and would file the block under +# "expected unparseable" rather than reporting it. +_PYTHON_BLOCK = re.compile( + r"^(?P *)```python[^\n]*\n(?P.*?)^(?P=indent)```", + re.MULTILINE | re.DOTALL, +) + +# Blocks that are deliberately not valid Python: illustrations whose `...` +# placeholders stand in for elided content. Pinned by identity, not count -- a +# bare total stays put when one block breaks as another is fixed. The identity +# is a digest of the block body, so reordering a document does not trip it but +# editing one of these blocks does. +_EXPECTED_UNPARSEABLE = { + "PYDANTIC_GUIDE.md:43488ef4", +} + +# An import statement, matched textually -- used to police the excuse list, +# where by definition `ast` cannot be applied. +_OVERTURE_IMPORT = re.compile(r"^\s*(?:from|import)\s+overture\b", re.MULTILINE) + +# Golden files are generated fixtures, not documentation. +_EXCLUDED = "tests/golden/" + + +def _repo_root() -> Path: + """Locate the checkout root, or skip the module when there isn't one.""" + for candidate in Path(__file__).resolve().parents: + if (candidate / ".git").exists(): + return candidate + pytest.skip("not running from a git checkout", allow_module_level=True) + + +def _markdown_files() -> list[Path]: + root = _ROOT + try: + listed = subprocess.run( + ["git", "ls-files", "-z", "*.md"], + cwd=root, + capture_output=True, + text=True, + check=True, + ).stdout + except (OSError, subprocess.CalledProcessError) as exc: # pragma: no cover + pytest.skip(f"cannot list tracked files: {exc}", allow_module_level=True) + return [root / n for n in listed.split("\0") if n and _EXCLUDED not in n] + + +def python_blocks(text: str) -> list[str]: + """Yield the dedented body of every fenced Python block in `text`. + + Handles a block indented inside a list item. The dedent is load-bearing: + without it `ast.parse` raises `IndentationError`, a `SyntaxError`, and the + block would be filed as "expected unparseable" rather than reported. + """ + return [textwrap.dedent(match["body"]) for match in _PYTHON_BLOCK.finditer(text)] + + +def _documented_enum_references( + paths: list[Path], +) -> list[tuple[Path, str, str, str]]: + """Collect `(file, module, enum, member)` quads across the corpus.""" + found: list[tuple[Path, str, str, str]] = [] + for path in paths: + for body in python_blocks(path.read_text(encoding="utf-8")): + try: + refs = enum_references(body) + except SyntaxError: + continue + found += [(path, module, name, member) for module, name, member in refs] + return found + + +def parse_imports(source: str) -> list[tuple[str, str | None]]: + """Extract `overture.*` imports from one block of Python source. + + Returns `(module, attribute)` pairs; `attribute` is None for a plain + `import overture.x`. Raises `SyntaxError` if the block is not valid Python. + """ + found: list[tuple[str, str | None]] = [] + for node in ast.walk(ast.parse(source)): + if isinstance(node, ast.ImportFrom): + # A relative import (`from . import x`) has no absolute module. + if node.level or not node.module or not _is_overture(node.module): + continue + found += [ + (node.module, alias.name) for alias in node.names if alias.name != "*" + ] + elif isinstance(node, ast.Import): + found += [ + (alias.name, None) for alias in node.names if _is_overture(alias.name) + ] + return found + + +def _is_overture(module: str) -> bool: + return module == "overture" or module.startswith("overture.") + + +def enum_references(source: str) -> list[tuple[str, str, str]]: + """Find `Enum.MEMBER` uses of enums imported in the same block. + + Returns `(module, enum_name, member)`. Restricted to enums on purpose: + their members really are class attributes, whereas a Pydantic model's + fields are not, so `Place.addresses` would read as missing. + """ + tree = ast.parse(source) + imported = { + alias.asname or alias.name: node.module + for node in ast.walk(tree) + if isinstance(node, ast.ImportFrom) + and node.module + and not node.level + and _is_overture(node.module) + for alias in node.names + } + return [ + (imported[node.value.id], node.value.id, node.attr) + for node in ast.walk(tree) + if isinstance(node, ast.Attribute) + and isinstance(node.value, ast.Name) + and node.value.id in imported + and node.attr.isupper() # a member, not a method + ] + + +def _documented_imports( + paths: list[Path], +) -> tuple[list[tuple[Path, str, str | None]], dict[str, str]]: + """Collect `(file, module, attribute)` triples, and identify unparseable blocks.""" + found: list[tuple[Path, str, str | None]] = [] + unparseable: dict[str, str] = {} + for path in paths: + for body in python_blocks(path.read_text(encoding="utf-8")): + try: + imports = parse_imports(body) + except SyntaxError: + digest = hashlib.sha256(body.encode()).hexdigest()[:8] + unparseable[f"{path.relative_to(_ROOT)}:{digest}"] = body + continue + found += [(path, module, attr) for module, attr in imports] + return found, unparseable + + +_ROOT = _repo_root() +_MARKDOWN = _markdown_files() +_DOCUMENTED, _UNPARSEABLE = _documented_imports(_MARKDOWN) +_ENUM_REFERENCES = _documented_enum_references(_MARKDOWN) + + +class TestBlocks: + """`python_blocks` against inline fixtures. + + Same argument as `TestParser` below, one layer up: a fence form the + matcher misses yields no block, and a sweep that collects less still + passes. The corpus cannot pin this. + """ + + def test_flush_block(self) -> None: + text = "text\n\n```python\nx = 1\n```\n\nmore\n" + assert python_blocks(text) == ["x = 1\n"] + + def test_list_item_block(self) -> None: + """A block indented under a list item, dedented to parse.""" + text = "1. Example:\n\n ```python\n x = 1\n ```\n" + assert python_blocks(text) == ["x = 1\n"] + + def test_deeply_indented_block(self) -> None: + text = "- a\n - b\n\n ```python\n x = 1\n ```\n" + assert python_blocks(text) == ["x = 1\n"] + + def test_indented_block_parses_only_after_dedent(self) -> None: + """Without the dedent this is an IndentationError, i.e. excused.""" + text = ( + "- item\n\n ```python\n from overture.schema.places import Place\n ```\n" + ) + (body,) = python_blocks(text) + assert parse_imports(body) == [("overture.schema.places", "Place")] + + def test_relative_indent_inside_a_block_is_preserved(self) -> None: + text = "- item\n\n ```python\n def f():\n return 1\n ```\n" + assert python_blocks(text) == ["def f():\n return 1\n"] + + def test_non_python_fences_are_ignored(self) -> None: + text = "```toml\nx = 1\n```\n\n```\nplain\n```\n" + assert python_blocks(text) == [] + + def test_several_blocks(self) -> None: + text = "```python\na = 1\n```\n\ntext\n\n```python\nb = 2\n```\n" + assert python_blocks(text) == ["a = 1\n", "b = 2\n"] + + +class TestOvertureImportMatcher: + """`_OVERTURE_IMPORT` against fixtures. + + It polices the excuse list, where `ast` cannot be applied by definition. + No excused block imports `overture` today, so the corpus never exercises + it -- the matcher would go blind without anyone noticing. + """ + + def test_matches_from_import(self) -> None: + assert _OVERTURE_IMPORT.search("x = 1\nfrom overture.schema import y\n") + + def test_matches_plain_import(self) -> None: + assert _OVERTURE_IMPORT.search(" import overture.schema.places\n") + + def test_ignores_other_packages(self) -> None: + assert not _OVERTURE_IMPORT.search("import json\nfrom pydantic import X\n") + + def test_ignores_a_longer_name(self) -> None: + assert not _OVERTURE_IMPORT.search("import overtures\n") + + def test_ignores_a_mention_in_prose(self) -> None: + assert not _OVERTURE_IMPORT.search("# import overture is what you'd write\n") + + +class TestParser: + """`parse_imports` against inline fixtures. + + The corpus cannot test the parser: any form it fails to handle simply + yields nothing, and a sweep that collects less still passes. + """ + + def test_single_line(self) -> None: + assert parse_imports("from overture.schema.buildings import Building") == [ + ("overture.schema.buildings", "Building") + ] + + def test_wrapped(self) -> None: + """The form `ruff format` produces once the names outgrow a line.""" + source = ( + "from overture.schema.system.numeric import (\n" + " int8, # signed\n" + " float64,\n" + ")\n" + ) + assert parse_imports(source) == [ + ("overture.schema.system.numeric", "int8"), + ("overture.schema.system.numeric", "float64"), + ] + + def test_alias(self) -> None: + assert parse_imports("from overture.schema.places import Place as P") == [ + ("overture.schema.places", "Place") + ] + + def test_plain_import_with_alias(self) -> None: + assert parse_imports("import overture.schema.buildings as b") == [ + ("overture.schema.buildings", None) + ] + + def test_star_and_relative_are_skipped(self) -> None: + assert parse_imports("from overture.schema.buildings import *") == [] + assert parse_imports("from . import buildings") == [] + + def test_non_overture_is_skipped(self) -> None: + assert parse_imports("import json\nfrom pydantic import BaseModel") == [] + + def test_indented_import(self) -> None: + source = "def f():\n from overture.schema.places import Place\n" + assert parse_imports(source) == [("overture.schema.places", "Place")] + + def test_invalid_source_raises(self) -> None: + with pytest.raises(SyntaxError): + parse_imports('{"a": 1, ...}\nthis is not python') + + +def test_corpus_is_non_empty() -> None: + """Guard against a silently empty sweep reporting success. + + The thresholds only have to be low enough to survive ordinary doc churn + and high enough that an empty or near-empty parse fails loudly. + """ + assert len(_MARKDOWN) > 10 + assert len(_DOCUMENTED) > 20 + + +def test_unparseable_blocks_are_the_expected_ones() -> None: + """A block that stops parsing drops out of the sweep silently.""" + assert set(_UNPARSEABLE) == _EXPECTED_UNPARSEABLE + + +def test_no_excused_block_hides_an_import() -> None: + """An excused block may illustrate, but may not cite the API. + + A content-addressed waiver is durable, so a block that both fails to + parse and imports `overture.*` would be exempt from the sweep forever -- + the one outcome this module exists to prevent. Making the block parse is + always available; every case so far took a one-token edit. + """ + for name, body in _UNPARSEABLE.items(): + assert not _OVERTURE_IMPORT.search(body), ( + f"{name} is excused from parsing but imports overture.*; " + "make the block parse instead of excusing it" + ) + + +@pytest.mark.parametrize( + ("path", "module", "enum_name", "member"), + _ENUM_REFERENCES, + ids=[ + f"{path.relative_to(_ROOT)}:{name}.{member}" + for path, _, name, member in _ENUM_REFERENCES + ], +) +def test_documented_enum_member_exists( + path: Path, module: str, enum_name: str, member: str +) -> None: + """An enum member named in the docs exists on the enum. + + The import sweep cannot see this: `from ... import Relationship` + resolves whether or not `Relationship.CONNECTS_TO` does. + """ + enum_class = getattr(importlib.import_module(module), enum_name) + if not (isinstance(enum_class, type) and issubclass(enum_class, Enum)): + return + names = [m.name for m in enum_class] + assert member in names, ( + f"{path}: `{enum_name}` has no member `{member}`; it has {names}" + ) + + +@pytest.mark.parametrize( + ("path", "module", "attribute"), + _DOCUMENTED, + ids=[ + f"{path.relative_to(_ROOT)}:{module}" + (f".{attr}" if attr else "") + for path, module, attr in _DOCUMENTED + ], +) +def test_documented_import_resolves( + path: Path, module: str, attribute: str | None +) -> None: + """An import written in the docs imports cleanly.""" + imported = importlib.import_module(module) + if attribute is None or hasattr(imported, attribute): + return + # `from pkg import submodule` binds only once the submodule is imported. + try: + importlib.import_module(f"{module}.{attribute}") + except ImportError as exc: + pytest.fail(f"{path}: `{module}` has no attribute `{attribute}` ({exc})")