From 2688ec22a22fa11d8ea42aacc033cfbc568c2dd0 Mon Sep 17 00:00:00 2001 From: Seth Fitzsimmons Date: Fri, 4 Sep 2026 10:11:33 -0700 Subject: [PATCH 1/5] codegen: carry defaults and deprecation through extraction FieldSpec had no slot for either, so a renderer that needed to know a field's default -- or that it was deprecated -- had to re-walk Pydantic and re-derive the unwrapping the extraction layer already does. Two sentinels mean "no default", not one: PydanticUndefined, and the MISSING that Omitable[T] installs to get JSON Schema omissibility rather than nullability. Extracting MISSING as a value would make every Omitable field claim a default it does not have. NO_DEFAULT keeps the absence distinct from a declared default of None, which is legal and different. default_factory deliberately does not land here. A factory is behavior, and calling it at extraction time would freeze one sample of a value whose whole point is to be produced per instance. Signed-off-by: Seth Fitzsimmons --- .../codegen/extraction/model_extraction.py | 45 ++++++++- .../schema/codegen/extraction/specs.py | 32 +++++++ .../tests/test_model_extraction.py | 91 +++++++++++++++++++ 3 files changed, 167 insertions(+), 1 deletion(-) diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/model_extraction.py b/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/model_extraction.py index 7371db208..1bb902e92 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/model_extraction.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/model_extraction.py @@ -5,6 +5,7 @@ from collections.abc import Mapping from pydantic import BaseModel +from pydantic.experimental.missing_sentinel import MISSING from pydantic.fields import FieldInfo from pydantic_core import PydanticUndefined @@ -15,7 +16,7 @@ ModelRef, UnionRef, ) -from .specs import FieldSpec, RecordSpec, is_model_class +from .specs import NO_DEFAULT, FieldSpec, RecordSpec, is_model_class from .type_analyzer import ( ModelResolver, UnionResolver, @@ -46,6 +47,44 @@ def resolve_field_alias(field_name: str, field_info: FieldInfo) -> str: return field_name +def _field_default(field_info: FieldInfo) -> object: + """Return the field's declared default, or `NO_DEFAULT`. + + Typed `object` rather than `Any`: a default really is an arbitrary value, but + `object` says so without switching off type checking at every call site. + + `default_factory` is deliberately not consulted: a factory is behavior, and + calling it here would capture one sample of a value meant to be produced per + instance. A field with only a factory extracts as `NO_DEFAULT` -- which is + accurate about what a static schema can carry, not a silent drop. + + Two sentinels mean "no default", not one. `PydanticUndefined` is Pydantic's, + and `MISSING` is the one `Omitable[T]` installs (`Field(default=MISSING)`) to + get JSON Schema omissibility instead of Pydantic nullability -- see + `overture.schema.system.optionality`. `MISSING` is machinery for "this key may + be absent", never a value anyone declared, so extracting it as a default makes + every `Omitable` field claim a default it does not have. + """ + if field_info.default is PydanticUndefined or field_info.default is MISSING: + return NO_DEFAULT + return field_info.default + + +def _field_deprecation(field_info: FieldInfo) -> tuple[bool, str | None]: + """Return `(is_deprecated, message)` from Pydantic's `deprecated`. + + Pydantic admits `True`, a string message, or a `warnings.deprecated` + instance. All three mean deprecated; only the string carries prose, and it is + kept beside the flag rather than folded into it. + """ + deprecated = field_info.deprecated + if deprecated is None or deprecated is False: + return False, None + if isinstance(deprecated, str): + return True, deprecated + return True, None + + def _is_field_required(field_info: FieldInfo, is_optional: bool) -> bool: """Determine whether a field is required (no default and not Optional).""" has_default = ( @@ -178,6 +217,7 @@ def _extract_model_recursive( # misses those constraints. Reattach them at the topmost # constraint-bearing layer. shape = attach_field_metadata(shape, field_info) + is_deprecated, deprecation_message = _field_deprecation(field_info) fields.append( FieldSpec( name=resolve_field_alias(field_name, field_info), @@ -185,6 +225,9 @@ def _extract_model_recursive( description=field_info.description or ti_description, is_required=_is_field_required(field_info, is_optional), is_optional=is_optional, + default=_field_default(field_info), + is_deprecated=is_deprecated, + deprecation_message=deprecation_message, ) ) diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/specs.py b/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/specs.py index 6b5259da1..3fc4ed340 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/specs.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/specs.py @@ -120,6 +120,23 @@ class EnumSpec(_SourceTypeIdentityMixin): source_type: type | None = None +class _NoDefault: + """The absence of a default, distinct from a default of `None`. + + `None` is a legal default value, so `default=None` cannot mean "no default" + without collapsing the two. A dedicated sentinel keeps them apart, and reads + at a use site (`spec.default is NO_DEFAULT`) as the question being asked. + """ + + __slots__ = () + + def __repr__(self) -> str: # pragma: no cover - debugging aid + return "NO_DEFAULT" + + +NO_DEFAULT = _NoDefault() + + @dataclass class FieldSpec: """Specification for a model field: header metadata plus structural shape. @@ -134,6 +151,21 @@ class FieldSpec: description: str | None = None is_required: bool = True is_optional: bool = False + # The field's declared default, or `NO_DEFAULT` when it has none. A + # `default_factory` does NOT land here: a factory is behavior, and calling it + # at extraction time would freeze one sample of a value whose whole point is + # to be produced per instance. + default: Any = NO_DEFAULT + # Pydantic's `deprecated`, normalized to a flag. Pydantic admits `True` or a + # deprecation *message*; a message narrows to `True` here, because every + # target this IR feeds has a boolean keyword and none has anywhere to put the + # prose. The renderer logs that narrowing where it applies -- the IR's job is + # to carry the fact, not to decide what a target does about it. + is_deprecated: bool = False + # The message form of `deprecated`, kept beside the flag so a target that + # grows somewhere to put it does not have to re-extract. `None` when the + # field declared a bare `True`, or nothing at all. + deprecation_message: str | None = None @dataclass diff --git a/packages/overture-schema-codegen/tests/test_model_extraction.py b/packages/overture-schema-codegen/tests/test_model_extraction.py index e8b3fd2c7..80c2eac29 100644 --- a/packages/overture-schema-codegen/tests/test_model_extraction.py +++ b/packages/overture-schema-codegen/tests/test_model_extraction.py @@ -15,6 +15,7 @@ from overture.schema.codegen.extraction.field_walk import terminal_of from overture.schema.codegen.extraction.length_constraints import ArrayMinLen from overture.schema.codegen.extraction.model_extraction import extract_model +from overture.schema.codegen.extraction.specs import NO_DEFAULT from overture.schema.common.scoping.vehicle import VehicleSelector @@ -179,3 +180,93 @@ class M(BaseModel): assert isinstance(items_field.shape, ArrayOf) constraints = [cs.constraint for cs in items_field.shape.constraints] assert ArrayMinLen(min_length=2) in constraints + + +def test_declared_default_reaches_field_spec() -> None: + """A field's declared default lands on `FieldSpec.default`.""" + + class M(BaseModel): + level: int = 3 + + spec = extract_model(M) + field = next(f for f in spec.fields if f.name == "level") + + assert field.default == 3 + + +def test_field_without_default_is_no_default_not_none() -> None: + """No default extracts as `NO_DEFAULT`, which `None` cannot stand in for. + + `default=None` is a legal declaration, so collapsing the two would make a + field that defaults to null indistinguishable from one that has no default. + """ + + class M(BaseModel): + required: int + nullable_default: int | None = None + + spec = extract_model(M) + fields = {f.name: f for f in spec.fields} + + assert fields["required"].default is NO_DEFAULT + assert fields["nullable_default"].default is None + + +def test_default_factory_does_not_become_a_default() -> None: + """A `default_factory` extracts as `NO_DEFAULT`. + + Calling it here would freeze one sample of a value whose whole point is to + be produced per instance, so extraction records that there is no static + default rather than inventing one. + """ + + class M(BaseModel): + tags: list[str] = Field(default_factory=list) + + spec = extract_model(M) + field = next(f for f in spec.fields if f.name == "tags") + + assert field.default is NO_DEFAULT + + +def test_deprecated_message_sets_flag_and_keeps_prose() -> None: + """A string `deprecated` sets the flag and preserves the message beside it.""" + + class M(BaseModel): + old: Annotated[str, Field(deprecated="use `new`")] = "x" + + spec = extract_model(M) + field = next(f for f in spec.fields if f.name == "old") + + assert field.is_deprecated is True + assert field.deprecation_message == "use `new`" + + +def test_bare_deprecated_true_carries_no_message() -> None: + """A bare `deprecated=True` sets the flag with no prose to carry.""" + + class M(BaseModel): + old: Annotated[str, Field(deprecated=True)] = "x" + + spec = extract_model(M) + field = next(f for f in spec.fields if f.name == "old") + + assert field.is_deprecated is True + assert field.deprecation_message is None + + +def test_undeclared_deprecation_is_false() -> None: + """A field that says nothing about deprecation is not deprecated. + + The did-happen half of the pair above: without this, a reader cannot tell a + working flag from one that is wired to a slot nothing ever sets. + """ + + class M(BaseModel): + current: str + + spec = extract_model(M) + field = next(f for f in spec.fields if f.name == "current") + + assert field.is_deprecated is False + assert field.deprecation_message is None From fa33fd3edcbefc94749c991c361c5b7144a9ae34 Mon Sep 17 00:00:00 2001 From: Seth Fitzsimmons Date: Fri, 4 Sep 2026 10:12:44 -0700 Subject: [PATCH 2/5] codegen: render STAC table:columns from the models A new --format stac-table-columns emitting one STAC properties fragment per feature type. Closes the gap in stac.overturemaps.org, whose table:columns today carries names and neither type nor description. The column list is the easy half. A Column Object is name/type/ description and nothing else, so the renderer's second output is a gap log: 3058 entries over 246 columns -- 1528 constraints, 628 statements of optionality, 395 defaults, 353 nested field descriptions, 103 enum sites over 41 vocabularies. None of them are IR gaps. The models can supply everything a Column Object asks for, so this is a strict, lossy projection with no authoring step. Three decisions are stated in the code, because nothing downstream can falsify any of them. The type dialect is lowercase DuckDB. The extension declares type as a bare JSON Schema string, so there is no vocabulary and no validator -- an absurd type string passes. Published catalogs carry two dialects in mixed case across 526 columns, and the spec repo's own generated reference catalog emits DuckDB. A discriminated union is taken from the merged field list, and every arm-only loss is logged. Walking only the arms misses fields that stay un-narrowed in the merged list; walking only the merged list drops each non-first arm at a duplicated name. Both losses are real and neither can raise, because the arms stringify identically -- three names over seven sites on the current models, all in segment. Ambiguity is refused rather than guessed. table:primary_geometry is emitted only when exactly one geometry column exists; table:row_count never, since a row count is a property of data and this path has only a schema. A recursive model raises, because a type string cannot name a type and so a cycle cannot terminate. Fixes #723. Signed-off-by: Seth Fitzsimmons --- .../changelog.d/723.feature.md | 1 + .../src/overture/schema/codegen/cli.py | 21 +- .../codegen/stac_table_columns/__init__.py | 6 + .../codegen/stac_table_columns/exceptions.py | 52 ++ .../codegen/stac_table_columns/pipeline.py | 57 ++ .../codegen/stac_table_columns/renderer.py | 619 ++++++++++++++++++ .../tests/test_stac_table_columns.py | 164 +++++ 7 files changed, 919 insertions(+), 1 deletion(-) create mode 100644 packages/overture-schema-codegen/changelog.d/723.feature.md create mode 100644 packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/__init__.py create mode 100644 packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/exceptions.py create mode 100644 packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py create mode 100644 packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/renderer.py create mode 100644 packages/overture-schema-codegen/tests/test_stac_table_columns.py diff --git a/packages/overture-schema-codegen/changelog.d/723.feature.md b/packages/overture-schema-codegen/changelog.d/723.feature.md new file mode 100644 index 000000000..f7c71e90d --- /dev/null +++ b/packages/overture-schema-codegen/changelog.d/723.feature.md @@ -0,0 +1 @@ +Added a `stac-table-columns` codegen format rendering the STAC Table extension's `table:columns` from the models, with a gap log recording what a flat column list cannot hold. Field defaults and deprecation now carry through extraction so the renderer can report both. diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py b/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py index a5687d817..5a20509c5 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py @@ -23,12 +23,13 @@ from .markdown.pipeline import generate_markdown_pages from .pyspark.pipeline import generate_pyspark_modules from .spec_discovery import extract_alias_spec, extract_model_spec +from .stac_table_columns.pipeline import generate_table_columns_documents log = logging.getLogger(__name__) __all__ = ["cli"] -_OUTPUT_FORMATS = ("markdown", "pyspark") +_OUTPUT_FORMATS = ("markdown", "pyspark", "stac-table-columns") _FEATURE_FRONTMATTER = "---\nsidebar_position: 1\n---\n\n" @@ -121,6 +122,8 @@ def generate( if output_format == "pyspark": _generate_pyspark(model_specs, output_dir, test_output_dir) + elif output_format == "stac-table-columns": + _generate_table_columns(model_specs, output_dir) else: # RootModel entry points yield no ModelSpec, so they document as # named aliases -- reachable no other way, since a RootModel field @@ -176,6 +179,22 @@ def _generate_pyspark( _write_output(mod.content, test_output_dir, mod.path) +def _generate_table_columns( + model_specs: list[ModelSpec], + output_dir: Path | None, +) -> None: + """Generate one STAC `table:` properties fragment per model. + + Every model emits, including the discriminated-union root: a flat column + list is what a columnar sink stores for a union. The gap count is logged + per model because it, not the fragment, is what a flattening target has to + be judged on. + """ + for doc in generate_table_columns_documents(model_specs): + _write_output(doc.stac, output_dir, doc.stac_path) + log.info("%s: %d gaps", doc.model, len(doc.gaps)) + + def _ancestor_dirs(paths: set[PurePosixPath]) -> set[PurePosixPath]: """Collect all ancestor directories for a set of file paths.""" dirs: set[PurePosixPath] = set() diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/__init__.py b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/__init__.py new file mode 100644 index 000000000..8dedce803 --- /dev/null +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/__init__.py @@ -0,0 +1,6 @@ +"""STAC Table extension (`table:columns`) rendering. + +A flattening target: a Column Object is `name`, `type`, `description` and +nothing else, so this renderer's output is as much a record of what a flat +column list cannot hold as it is the list itself. +""" diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/exceptions.py b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/exceptions.py new file mode 100644 index 000000000..1911ea3ad --- /dev/null +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/exceptions.py @@ -0,0 +1,52 @@ +"""Typed record of everything the `table:columns` emit could not carry. + +`flatten-collision` is the kind only a flattening target produces: two union +arms contributing the same column name, where a columnar sink keeps one and the +other's meaning is gone with no trace in the output. `unrepresentable-in-pydantic` +runs the other way -- the target asks for something Pydantic has no way to +declare, so the gap is a possible Overture Schema feature rather than a defect. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import Literal + +__all__ = ["Kind", "TableColumnsGap", "TableColumnsUnrepresentable"] + +Kind = Literal[ + "ir-gap", + "target-gap", + "target-dialect", + "flatten-collision", + "unrepresentable-in-pydantic", + "renderer-gap", +] + + +@dataclass(frozen=True, slots=True) +class TableColumnsGap: + """One capability that did not survive the emit. + + `path` locates it in the emitted document, `capability` names what was lost + in the vocabulary of whichever side owns the loss, and `detail` carries the + mechanism. Classification is by mechanism, not by keyword. + """ + + model: str + path: str + kind: Kind + capability: str + detail: str + + +class TableColumnsUnrepresentable(Exception): + """Raised in strict mode, or when the renderer cannot proceed at all.""" + + def __init__(self, gaps: tuple[TableColumnsGap, ...]) -> None: + self.gaps = gaps + head = gaps[0] + super().__init__( + f"{head.model}{head.path}: {head.capability} ({head.kind}) -- {head.detail}" + + (f" [+{len(gaps) - 1} more]" if len(gaps) > 1 else "") + ) diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py new file mode 100644 index 000000000..2b3203233 --- /dev/null +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py @@ -0,0 +1,57 @@ +"""STAC `table:columns` generation pipeline: render documents without I/O. + +One artifact per model -- a STAC Item properties fragment carrying the `table:` +fields -- and the gap log beside it, which for a target this thin is as much the +deliverable as the fragment. +""" + +from __future__ import annotations + +import json +from collections.abc import Sequence +from dataclasses import dataclass +from pathlib import PurePosixPath + +from overture.schema.system.case import to_snake_case + +from ..extraction.specs import ModelSpec +from .exceptions import TableColumnsGap +from .renderer import TABLE_EXTENSION_URI, render_table_columns + +__all__ = ["TableColumnsOutput", "generate_table_columns_documents"] + + +@dataclass(frozen=True, slots=True) +class TableColumnsOutput: + """A rendered STAC fragment and everything it could not hold.""" + + model: str + stac: str + stac_path: PurePosixPath + gaps: tuple[TableColumnsGap, ...] + + +def generate_table_columns_documents( + model_specs: Sequence[ModelSpec], +) -> list[TableColumnsOutput]: + """Render one STAC fragment per spec, plus the gap log.""" + outputs: list[TableColumnsOutput] = [] + for spec in model_specs: + rendered = render_table_columns(spec) + stem = to_snake_case(spec.name) + # A bare properties fragment rather than a whole Item: the extension + # fields are what this target emits, and an Item would need id, geometry, + # bbox, datetime and links, none of which come from a schema. + fragment = { + "stac_extensions": [TABLE_EXTENSION_URI], + "properties": rendered.stac_fields(), + } + outputs.append( + TableColumnsOutput( + model=spec.name, + stac=json.dumps(fragment, indent=2) + "\n", + stac_path=PurePosixPath(f"{stem}.json"), + gaps=rendered.gaps, + ) + ) + return outputs diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/renderer.py b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/renderer.py new file mode 100644 index 000000000..82fcca90e --- /dev/null +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/renderer.py @@ -0,0 +1,619 @@ +"""Render a model spec to the STAC Table extension's `table:columns`. + +The Column Object is `name` (required), `type`, `description`. That is the whole +target. Everything else the IR carries -- constraints, enum vocabularies, named +types, defaults, optionality, nested field descriptions -- has nowhere to go, so +this renderer's real output is the gap log beside the columns. + +Two decisions are stated here rather than defaulted into, because nothing +downstream can falsify either: + +1. **The type dialect.** `type` is an unconstrained string, and real catalogs + carry whatever their generating engine's stringifier produced -- two dialects + in mixed case across the 526 columns of published catalogs surveyed. This renderer + emits lowercase DuckDB SQL, matching the spec repo's own generated reference + catalog. No validator can tell a right dialect from a wrong one. + +2. **Which walk supplies the union.** Walking only concrete arms misses fields + that stay un-narrowed in the merged list; walking only the merged list drops + every non-first arm's contribution at a duplicated name. Both losses are + real and neither can raise, because the arms stringify identically. This + renderer walks both, emits the merged list (that is what a columnar sink + stores), and logs a `flatten-collision` for every name where the two + disagree. +""" + +from __future__ import annotations + +import builtins +import datetime as _dt +from dataclasses import dataclass, field +from enum import Enum +from typing import Any + +from ..extraction.enum_extraction import extract_enum +from ..extraction.field import ( + AnyScalar, + ArrayOf, + FieldShape, + LiteralScalar, + MapOf, + ModelRef, + NewTypeShape, + Primitive, + UnionRef, +) +from ..extraction.field_walk import enum_source +from ..extraction.specs import NO_DEFAULT, FieldSpec, ModelSpec, UnionSpec +from .exceptions import Kind, TableColumnsGap, TableColumnsUnrepresentable + +__all__ = [ + "TABLE_EXTENSION_URI", + "TableColumnsDocument", + "render_table_columns", +] + +TABLE_EXTENSION_URI = "https://stac-extensions.github.io/table/v1.2.0/schema.json" + +# Lowercase DuckDB SQL, the dialect the Portolan spec repo's own generated +# reference catalog emits. See the module docstring: this is a choice. +_DUCKDB_TYPES: dict[str, str] = { + "int8": "tinyint", + "int16": "smallint", + "int32": "integer", + "int64": "bigint", + "uint8": "utinyint", + "uint16": "usmallint", + "uint32": "uinteger", + "uint64": "ubigint", + "float32": "float", + "float64": "double", + "str": "varchar", + "bool": "boolean", + "bytes": "blob", + "datetime": "timestamp with time zone", + "date": "date", + "Geometry": "geometry", +} + +# BBox is a plain class the codegen cannot walk, so the physical layout keeps a +# shared struct constant for it. Its members are fixed by the GeoParquet +# covering convention, and all three struct grammars in the wild describe the +# same four doubles. +_BBOX_TYPE = "struct(xmin double, ymin double, xmax double, ymax double)" + +# Pydantic types with no SQL type of their own. Keyed by NAME rather than by +# `issubclass`: in Pydantic v2 neither `HttpUrl` nor `EmailStr` is a `str` +# subclass, so the builtin fallback below never reaches them -- the same trap +# the Vecorel renderer records, found here the same way, by running it. +_PYDANTIC_AS_VARCHAR: frozenset[str] = frozenset( + {"HttpUrl", "AnyUrl", "AnyHttpUrl", "EmailStr", "UUID"} +) + +# Fallback keyed on the underlying Python type, for constrained-string NewTypes +# the registry has no name entry for. `bool` precedes `int` because it is a +# subclass. +_BUILTIN_TYPES: tuple[tuple[type, str], ...] = ( + (bool, "boolean"), + (_dt.datetime, "timestamp with time zone"), + (_dt.date, "date"), + (str, "varchar"), + (bytes, "blob"), + (int, "bigint"), + (float, "double"), +) + + +@dataclass +class Column: + """One emitted Column Object, plus what the emitter knows and cannot say.""" + + name: str + type: str + description: str | None + # Retained for the gap log; never emitted into `table:columns`, which + # has no field for any of it. + base_type: str | None + # The underlying Python type, kept so a downstream target can resolve + # through the same fallback chain the SQL type does. + python_type: builtins.type | None + enum_name: str | None + is_literal: bool + enum_members: tuple[tuple[str, str | None], ...] + has_default: bool + is_required: bool + constraint_names: tuple[str, ...] + + def as_column_object(self) -> dict[str, Any]: + """Return the Column Object exactly: name, type, description when present. + + An absent description is an absent KEY, not an empty string. A silent + empty string would make an undescribed column indistinguishable from a + described one in every downstream count. + """ + obj: dict[str, Any] = {"name": self.name, "type": self.type} + if self.description is not None: + obj["description"] = self.description + return obj + + +@dataclass +class TableColumnsDocument: + """The emitted STAC fields and the gap log.""" + + model: str + columns: list[Column] + primary_geometry: str | None + gaps: tuple[TableColumnsGap, ...] + + def stac_fields(self) -> dict[str, Any]: + """Return the `table:` fields for a STAC object. + + `table:row_count` is deliberately absent and cannot be otherwise: a row + count is a property of data, and this path has only a schema. Emitting a + placeholder would be a fabricated measurement; leaving it out silently + would be indistinguishable from forgetting it, which is what the gap log + is for. + """ + fields: dict[str, Any] = { + "table:columns": [c.as_column_object() for c in self.columns] + } + if self.primary_geometry is not None: + fields["table:primary_geometry"] = self.primary_geometry + return fields + + +@dataclass +class _Ctx: + """Render state: the model name, the gap log, and the recursion guard.""" + + model: str + gaps: list[TableColumnsGap] = field(default_factory=list) + stack: list[str] = field(default_factory=list) + # Named-type sites seen anywhere under this model, for the gap log. + named_sites: list[str] = field(default_factory=list) + + def gap(self, path: str, kind: Kind, capability: str, detail: str) -> None: + self.gaps.append( + TableColumnsGap(self.model, path or "/", kind, capability, detail) + ) + + +def _dedupe(fields: list[FieldSpec], ctx: _Ctx, path: str) -> list[FieldSpec]: + """Keep one FieldSpec per name, logging every name where arms disagree. + + The physical layout keeps the first and says nothing. Keeping the first is + right -- one column stores one type -- but the silence is not: the dropped + arm's enum vocabulary and description are gone with no trace in the output. + """ + seen: dict[str, FieldSpec] = {} + for f in fields: + prior = seen.get(f.name) + if prior is None: + seen[f.name] = f + continue + kept, dropped = _vocabulary_of(prior.shape), _vocabulary_of(f.shape) + ctx.gap( + f"{path}/{f.name}", + "flatten-collision", + "union arm merge", + f"two union arms contribute a column named {f.name!r}; the first is " + f"kept ({kept or 'no vocabulary'}) and the other's meaning " + f"({dropped or 'no vocabulary'}) is discarded. Both stringify to the " + "same physical type, so the layout's compatibility check cannot " + "refuse it and nothing downstream records the loss.", + ) + return list(seen.values()) + + +def _vocabulary_of(shape: FieldShape) -> str | None: + """Name the enum backing a shape, for collision reporting.""" + inner = _unwrap(shape) + if isinstance(inner, Primitive): + src = enum_source(inner) + if src is not None: + return src.__name__ + return inner.base_type + return type(inner).__name__ + + +def _unwrap(shape: FieldShape) -> FieldShape: + """Strip NewType wrappers, returning the structural shape beneath.""" + while isinstance(shape, NewTypeShape): + shape = shape.inner + return shape + + +def _top_fields(spec: ModelSpec, ctx: _Ctx) -> list[FieldSpec]: + """Top-level columns, the way a columnar sink sees them.""" + if isinstance(spec, UnionSpec): + merged = _dedupe(spec.fields, ctx, "") + _report_arm_only_types(spec, merged, ctx) + return merged + return spec.fields + + +def _report_arm_only_types( + union: UnionSpec, merged: list[FieldSpec], ctx: _Ctx +) -> None: + """Log named types reachable from the arms but not from the merged list. + + The mirror of `_dedupe`'s loss, and the one the doc pipeline and both prior + renderers see while this walk does not. Reported rather than merged in: the + merged list is what the column layout is, so adding arm-only types to it + would misdescribe the emitted table. + """ + merged_names: set[str] = set() + for f in merged: + _collect_type_names(f.shape, merged_names) + arm_names: set[str] = set() + for member in union.member_specs: + for f in member.spec.fields: + _collect_type_names(f.shape, arm_names) + only_arm = sorted(arm_names - merged_names) + if only_arm: + ctx.gap( + "", + "flatten-collision", + "arm-only named types", + f"reachable from the union's concrete arms but not from the merged " + f"field list this table is built from: {', '.join(only_arm)}. A walk " + "over either side alone under-counts, in opposite directions.", + ) + + +def _collect_type_names(shape: FieldShape, acc: set[str], depth: int = 0) -> None: + """Every named type reachable from *shape*: NewTypes, records, unions, enums.""" + if depth > 40: + return + match shape: + case NewTypeShape(name=name, inner=inner): + if name: + acc.add(name) + _collect_type_names(inner, acc, depth + 1) + case ArrayOf(element=element): + _collect_type_names(element, acc, depth + 1) + case MapOf(key=key, value=value): + _collect_type_names(key, acc, depth + 1) + _collect_type_names(value, acc, depth + 1) + case ModelRef(model=model, starts_cycle=starts_cycle): + acc.add(model.name) + if not starts_cycle: + for f in model.fields: + _collect_type_names(f.shape, acc, depth + 1) + case UnionRef(union=union): + acc.add(union.name) + for f in union.fields: + _collect_type_names(f.shape, acc, depth + 1) + case Primitive() as prim: + src = enum_source(prim) + if src is not None: + acc.add(src.__name__) + + +def _scalar_type( + scalar: Primitive | LiteralScalar | AnyScalar, ctx: _Ctx, path: str +) -> str: + """Map a terminal shape to a DuckDB type string, logging what that costs.""" + if isinstance(scalar, LiteralScalar): + ctx.gap( + path, + "target-gap", + "literal alternatives", + f"Literal{list(scalar.values)!r} narrows the column to " + f"{len(scalar.values)} value(s); a Column Object has no enum, const " + "or pattern keyword, so the column is a bare string.", + ) + return "varchar" + if isinstance(scalar, AnyScalar): + ctx.gap( + path, + "target-gap", + "any", + "typing.Any has no column type. Emitted as varchar to match what the " + "physical layout stores, which is a stringification, not a type.", + ) + return "varchar" + src = enum_source(scalar) + if src is not None: + spec = extract_enum(src) + described = sum(1 for m in spec.members if m.description is not None) + ctx.gap( + path, + "target-gap", + "enum vocabulary", + f"{src.__name__}: {len(spec.members)} members ({described} described) " + "collapse to a bare string; table:columns has no enum keyword.", + ) + return "varchar" + if scalar.base_type == "BBox": + return _BBOX_TYPE + mapped = _DUCKDB_TYPES.get(scalar.base_type) + if mapped is not None: + return mapped + source_type = scalar.source_type + if scalar.base_type in _PYDANTIC_AS_VARCHAR or ( + source_type is not None and source_type.__name__ in _PYDANTIC_AS_VARCHAR + ): + ctx.gap( + path, + "target-dialect", + "semantic type erased", + f"{scalar.base_type} has no SQL type of its own and stringifies to " + "varchar; the semantics live only in the name, which the column does " + "not carry.", + ) + return "varchar" + if source_type is not None: + by_name = _DUCKDB_TYPES.get(source_type.__name__) + if by_name is not None: + return by_name + for py_type, name in _BUILTIN_TYPES: + if isinstance(source_type, type) and issubclass(source_type, py_type): + ctx.gap( + path, + "target-dialect", + "semantic type erased", + f"{source_type.__name__} has no SQL type of its own and " + f"stringifies to {name}; the semantics live only in the name, " + "which the column does not carry.", + ) + return name + ctx.gap( + path, + "renderer-gap", + "unmapped base type", + f"no DuckDB type for base_type={scalar.base_type!r} " + f"(source_type={getattr(source_type, '__name__', None)!r}).", + ) + return "varchar" + + +def _type_string(shape: FieldShape, ctx: _Ctx, path: str) -> str: + """Build the DuckDB type string for a column, recursing into nesting. + + Nesting lives here, not in the column name: of the 526 columns across the + published catalogs surveyed, zero contain a dot, so a flattener that invents + dotted paths would describe a table nobody publishes. + """ + match shape: + case NewTypeShape(name=name, inner=inner): + if name: + ctx.named_sites.append(f"{path}:{name}") + return _type_string(inner, ctx, path) + case ArrayOf(element=element): + return f"{_type_string(element, ctx, path + '[]')}[]" + case MapOf(key=key, value=value): + key_type = _type_string(key, ctx, path + "/key") + value_type = _type_string(value, ctx, path + "/value") + return f"map({key_type}, {value_type})" + case ModelRef(model=model, starts_cycle=starts_cycle): + ctx.named_sites.append(f"{path}:{model.name}") + if starts_cycle or model.name in ctx.stack: + raise TableColumnsUnrepresentable( + ( + TableColumnsGap( + ctx.model, + path, + "target-gap", + "recursion", + f"{model.name} is recursive; a type string has no " + "way to name a type, so a cycle cannot terminate.", + ), + ) + ) + ctx.stack.append(model.name) + try: + return _struct_string(model.fields, ctx, path, model.name) + finally: + ctx.stack.pop() + case UnionRef(union=union): + ctx.named_sites.append(f"{path}:{union.name}") + if union.name in ctx.stack: + raise TableColumnsUnrepresentable( + ( + TableColumnsGap( + ctx.model, + path, + "target-gap", + "recursion", + f"{union.name} is recursive; see above.", + ), + ) + ) + ctx.stack.append(union.name) + try: + merged = _dedupe(union.fields, ctx, path) + _report_arm_only_types(union, merged, ctx) + ctx.gap( + path, + "target-gap", + "discriminated union", + f"{union.name} has {len(union.members)} arms; the column is a " + "single struct of their merged fields, and the discriminator " + "survives only as a string. Which arm a row belongs to is not " + "recoverable from the schema.", + ) + return _struct_string(merged, ctx, path, union.name) + finally: + ctx.stack.pop() + case Primitive() | LiteralScalar() | AnyScalar() as scalar: + return _scalar_type(scalar, ctx, path) + raise TypeError(f"Unhandled FieldShape: {shape!r}") + + +def _struct_string(fields: list[FieldSpec], ctx: _Ctx, path: str, owner: str) -> str: + """`struct(name type, ...)`, logging the descriptions that die inside it.""" + if not fields: + ctx.gap( + path, + "renderer-gap", + "empty struct", + f"{owner} has no fields; an empty struct column cannot carry data.", + ) + return "struct()" + parts = [] + for f in fields: + member_path = f"{path}/{f.name}" + if f.description is not None: + ctx.gap( + member_path, + "target-gap", + "nested description", + f"{owner}.{f.name} is described, and a Column Object describes " + "only top-level columns. The NAME survives inside the type " + "string; the description does not.", + ) + _log_field_surplus(f, ctx, member_path, nested=True) + parts.append(f"{f.name} {_type_string(f.shape, ctx, member_path)}") + return f"struct({', '.join(parts)})" + + +def _constraint_names(shape: FieldShape, acc: list[str], depth: int = 0) -> None: + """Collect constraint class names at every layer of a shape.""" + if depth > 40: + return + for source in getattr(shape, "constraints", ()) or (): + acc.append(type(source.constraint).__name__) + match shape: + case NewTypeShape(inner=inner): + _constraint_names(inner, acc, depth + 1) + case ArrayOf(element=element): + _constraint_names(element, acc, depth + 1) + case MapOf(key=key, value=value): + _constraint_names(key, acc, depth + 1) + _constraint_names(value, acc, depth + 1) + case ModelRef(model=model, starts_cycle=starts_cycle): + if not starts_cycle: + for f in model.fields: + _constraint_names(f.shape, acc, depth + 1) + case UnionRef(union=union): + for f in union.fields: + _constraint_names(f.shape, acc, depth + 1) + + +def _log_field_surplus(f: FieldSpec, ctx: _Ctx, path: str, *, nested: bool) -> None: + """Log the per-field capabilities a Column Object has no field for.""" + where = "nested field" if nested else "column" + if f.default is not NO_DEFAULT: + ctx.gap( + path, + "target-gap", + "default", + f"{where} carries default={f.default!r}. The IR carries defaults; a " + "Column Object is name/type/description and has nowhere to put one. " + "The gap is the target's, not the IR's.", + ) + if f.is_deprecated: + ctx.gap( + path, + "target-gap", + "deprecated", + f"{where} is deprecated; there is no keyword for it.", + ) + ctx.gap( + path, + "target-gap", + "optionality", + f"{where} is {'required' if f.is_required else 'optional'}" + f"{' and nullable' if f.is_optional else ''}; a Column Object has no " + "required, nullable or optional keyword, so this is not expressible at " + "all -- strictly less than Vecorel, which at least had a root required " + "list.", + ) + names: list[str] = [] + _constraint_names(f.shape, names) + for name in names: + ctx.gap( + path, + "target-gap", + "constraint", + f"{name} has no representation: the Column Object vocabulary is " + "name, type, description.", + ) + + +def _enum_of(shape: FieldShape) -> type[Enum] | None: + inner = _unwrap(shape) + return enum_source(inner) if isinstance(inner, Primitive) else None + + +def render_table_columns( + spec: ModelSpec, *, strict: bool = False +) -> TableColumnsDocument: + """Render *spec* to STAC `table:` fields, with a gap log beside them. + + `strict=True` raises rather than returning a non-empty log. A `UnionSpec` + root emits normally: a flat column list is exactly + what a columnar sink stores for a union, and the IR already carries the + merged field list that sink needs. + """ + ctx = _Ctx(model=spec.name) + columns: list[Column] = [] + for f in _top_fields(spec, ctx): + path = f"/{f.name}" + _log_field_surplus(f, ctx, path, nested=False) + inner = _unwrap(f.shape) + enum_cls = _enum_of(f.shape) + members: tuple[tuple[str, str | None], ...] = () + if enum_cls is not None: + members = tuple( + (m.value, m.description) for m in extract_enum(enum_cls).members + ) + constraints: list[str] = [] + _constraint_names(f.shape, constraints) + columns.append( + Column( + name=f.name, + type=_type_string(f.shape, ctx, path), + description=f.description, + base_type=inner.base_type if isinstance(inner, Primitive) else None, + python_type=( + inner.source_type + if isinstance(inner, Primitive) + and isinstance(inner.source_type, type) + else None + ), + enum_name=enum_cls.__name__ if enum_cls is not None else None, + is_literal=isinstance(inner, LiteralScalar), + enum_members=members, + has_default=f.default is not NO_DEFAULT, + is_required=f.is_required, + constraint_names=tuple(constraints), + ) + ) + + primary = _primary_geometry(columns, ctx) + gaps = tuple(ctx.gaps) + if strict and gaps: + raise TableColumnsUnrepresentable(gaps) + return TableColumnsDocument( + model=spec.name, + columns=columns, + primary_geometry=primary, + gaps=gaps, + ) + + +def _primary_geometry(columns: list[Column], ctx: _Ctx) -> str | None: + """Derive `table:primary_geometry`, refusing to guess when it is ambiguous.""" + geometry = [c.name for c in columns if c.base_type == "Geometry"] + if len(geometry) == 1: + return geometry[0] + if not geometry: + ctx.gap( + "", + "target-gap", + "primary geometry", + "no geometry column; table:primary_geometry is omitted.", + ) + return None + ctx.gap( + "", + "ir-gap", + "primary geometry", + f"{len(geometry)} geometry columns ({', '.join(geometry)}) and nothing in " + "the IR says which is primary. Omitted rather than guessed -- picking the " + "first would invent a fact about the data.", + ) + return None diff --git a/packages/overture-schema-codegen/tests/test_stac_table_columns.py b/packages/overture-schema-codegen/tests/test_stac_table_columns.py new file mode 100644 index 000000000..4148e7e89 --- /dev/null +++ b/packages/overture-schema-codegen/tests/test_stac_table_columns.py @@ -0,0 +1,164 @@ +"""Tests for the STAC `table:columns` renderer. + +A Column Object is `name`, `type`, `description`, so most of what this renderer +knows it cannot say. These cover the decisions that survive that flattening: how +an absent description is spelled, what happens at a union collision, and the two +cases where guessing is worse than refusing. +""" + +from typing import Annotated, Literal + +import pytest +from pydantic import BaseModel, Field + +from overture.schema.codegen.extraction.model_extraction import extract_model +from overture.schema.codegen.stac_table_columns.exceptions import ( + TableColumnsUnrepresentable, +) +from overture.schema.codegen.stac_table_columns.renderer import render_table_columns +from overture.schema.system.geometric import Geometry + + +def _capabilities(doc: object) -> set[str]: + return {g.capability for g in doc.gaps} # type: ignore[attr-defined] + + +def test_undescribed_column_omits_the_key_entirely() -> None: + """A column with no description omits `description`, never emits `""`. + + The two are not interchangeable downstream: an omitted key leaves the + undescribed set as something a consumer can query for, while an empty string + is a description that happens to say nothing, and no count can tell it from + one that was authored badly. + """ + + class M(BaseModel): + described: Annotated[str, Field(description="what it is")] + bare: str + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert columns["described"]["description"] == "what it is" + assert "description" not in columns["bare"] + + +def test_primary_geometry_is_emitted_when_exactly_one_geometry_exists() -> None: + """One geometry column resolves `table:primary_geometry` unambiguously.""" + + class M(BaseModel): + geometry: Annotated[Geometry, Field(description="the shape")] + name: str + + doc = render_table_columns(extract_model(M)) + + assert doc.stac_fields()["table:primary_geometry"] == "geometry" + + +def test_two_geometries_omit_primary_rather_than_guessing() -> None: + """Nothing in the IR says which of two geometry columns is primary. + + Picking the first would invent a fact about the data, so the key is omitted + and the ambiguity is logged as an IR gap. This is the direction that + discriminates: the shipped models all carry exactly one geometry column, so + without a synthetic pair the branch is never exercised. + """ + + class M(BaseModel): + footprint: Annotated[Geometry, Field(description="one")] + centroid: Annotated[Geometry, Field(description="another")] + + doc = render_table_columns(extract_model(M)) + + assert "table:primary_geometry" not in doc.stac_fields() + assert any( + g.kind == "ir-gap" and g.capability == "primary geometry" for g in doc.gaps + ) + + +def test_row_count_is_never_emitted() -> None: + """`table:row_count` is a property of data; this path has only a schema. + + Emitting a placeholder would be a fabricated measurement, so the key is + absent by construction rather than by omission. + """ + + class M(BaseModel): + id: str + + assert "table:row_count" not in render_table_columns(extract_model(M)).stac_fields() + + +def test_union_name_collision_is_logged_not_silently_dropped() -> None: + """Two arms contributing one column name emit a `flatten-collision`. + + Both arms stringify to the same physical type, so nothing downstream can + refuse the merge and nothing else records that an arm's meaning was lost. + This gap is the only trace. + """ + + class Base(BaseModel): + pass + + class Left(Base): + kind: Literal["left"] + value: Annotated[str, Field(description="a left value")] + + class Right(Base): + kind: Literal["right"] + value: Annotated[str, Field(description="a right value")] + + class Holder(BaseModel): + item: Annotated[Left | Right, Field(discriminator="kind")] + + doc = render_table_columns(extract_model(Holder)) + + assert any(g.kind == "flatten-collision" for g in doc.gaps) + + +def test_optionality_is_reported_as_lost_for_every_column() -> None: + """A Column Object has no required, nullable or optional keyword. + + Unlike a constraint, this one cannot be expressed at all, so it is logged + for every column rather than only for the ones that declare something. + """ + + class M(BaseModel): + required: str + optional: str | None = None + + doc = render_table_columns(extract_model(M)) + + assert "optionality" in _capabilities(doc) + + +def test_strict_mode_raises_rather_than_returning_a_gap_log() -> None: + """`strict=True` turns the log into a refusal. + + The default is to emit and report, because a flattening target that refused + on every loss would never emit at all; strict is for a caller that wants the + losses to be fatal. + """ + + class M(BaseModel): + anything: str | None = None + + with pytest.raises(TableColumnsUnrepresentable): + render_table_columns(extract_model(M), strict=True) + + +def test_gap_log_is_non_empty_for_a_real_model() -> None: + """The did-happen control for every assertion above. + + Each test that asserts a *particular* capability was logged would pass just + as well against a renderer whose log was accidentally empty of everything + else. This one fails if the log stops being populated at all. + """ + + class M(BaseModel): + bounded: Annotated[int, Field(ge=0, le=10)] + + doc = render_table_columns(extract_model(M)) + + assert doc.gaps + assert "constraint" in _capabilities(doc) From accab7e6e2fed74178ea50954339f56f0d0036b0 Mon Sep 17 00:00:00 2001 From: Seth Fitzsimmons Date: Thu, 10 Sep 2026 11:19:25 -0700 Subject: [PATCH 3/5] feat(codegen): stac-table-columns tracks table v1.3.0 Tracks the table extension's v1.3.0, released 2026-09-07 after this PR was written against v1.2.0. Four changes: - Bump the extension URL. Done. - Add data_type (STAC common metadata) alongside type. Done: an allowlist against Overture's own numeric base-type names, which already match the vocabulary verbatim; everything else -- structs, arrays, enums, unions, strings -- gets "other", itself a vocabulary member, not an omission. - Add geometry_type for the geometry column. The released v1.3.0 README defines no such field on the Column Object -- checked at https://github.com/stac-extensions/table/blob/v1.3.0/README.md. What it actually recommends is vector:geometry_types (plural) from the separate Vector extension, a list of GeoJSON type names. Implemented that instead, reading GeometryTypeConstraint.allowed_types off the geometry field: an exhaustive schema bound validated on every row, not a guess about what a dataset happens to contain. stac_extensions now also declares the Vector extension whenever a column emits the field, or the reference would not validate against either schema. A geometry column with no such constraint gets neither the field nor a claim about it, logged as a gap instead -- none of the shipped models hit that branch. - Add unit for columns with a real unit, e.g. speed limits. Not implemented: no Overture field carries a structured, column-wide fixed unit anywhere in the IR. Speed limits carry theirs as a per-row SpeedUnit enum nested inside Speed, not a schema-fixed unit, so a unit on the speed_limits column would misdescribe data that can legitimately mix mph and km/h row to row. The fields that really do have one fixed unit (Height, Elevation, Depth, roof bearing) carry it only in a free-text description ("... in meters"), which this renderer will not parse to fabricate a structured fact the schema never asserted. Separately, `type` now carries Arrow type names rather than lowercase DuckDB SQL. Arrow is what Overture actually distributes -- the release is GeoParquet -- so the names a reader of the published data reports are the ones a column list describing it should carry. Every type name and the composite grammar came from pyarrow's own stringifier, and match published catalogs, which name their columns int64, string, double, binary and struct. - Scalars: double and float rather than float64/float32 (Arrow's own names), string not varchar, bool not boolean, binary not blob, timestamp[us, tz=UTC] not "timestamp with time zone", date32[day] not date. The sized integers need no translation -- Overture's base-type names already are Arrow's, the same way they already are STAC's. - Composites: struct, list, map. A round-trip through Parquet renames a list's child "element" and appends the Parquet group name to a map's stringification; both are artifacts of the file rather than type names, so what is emitted is pyarrow's canonical constructor form. - Geometry is binary. Arrow has no geometry type: GeoParquet stores WKB in a binary column and keeps the semantics in file-level metadata, and binary is exactly what an Arrow reader reports for Overture's own released geometry column, as it is in both cited catalogs. The erasure is logged as a gap rather than absorbed; vector:geometry_types still carries the part of the semantic the schema can assert. type and data_type therefore disagree on floats -- Arrow says float/double where STAC common metadata says float32/float64. That is correct rather than a bug: each field is bound to its own vocabulary and harmonising them would make one of them lie. The module docstring and a test both say so, so the next reader does not "fix" it. Signed-off-by: Seth Fitzsimmons --- .../changelog.d/723.feature.md | 2 +- .../codegen/stac_table_columns/pipeline.py | 13 +- .../codegen/stac_table_columns/renderer.py | 332 +++++++++++++----- .../tests/test_stac_table_columns.py | 249 ++++++++++++- 4 files changed, 504 insertions(+), 92 deletions(-) diff --git a/packages/overture-schema-codegen/changelog.d/723.feature.md b/packages/overture-schema-codegen/changelog.d/723.feature.md index f7c71e90d..3b82423ca 100644 --- a/packages/overture-schema-codegen/changelog.d/723.feature.md +++ b/packages/overture-schema-codegen/changelog.d/723.feature.md @@ -1 +1 @@ -Added a `stac-table-columns` codegen format rendering the STAC Table extension's `table:columns` from the models, with a gap log recording what a flat column list cannot hold. Field defaults and deprecation now carry through extraction so the renderer can report both. +Added a `stac-table-columns` codegen format rendering the STAC Table extension's (v1.3.0) `table:columns` from the models, with a gap log recording what a flat column list cannot hold. Column types use Arrow's names, the vocabulary a reader of Overture's released GeoParquet reports. Each Column Object also carries STAC common metadata's `data_type` and, for the geometry column, the Vector extension's `vector:geometry_types` (declared in `stac_extensions` alongside the Table extension when emitted). Field defaults and deprecation now carry through extraction so the renderer can report both. diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py index 2b3203233..d0d7b8aeb 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py @@ -16,7 +16,7 @@ from ..extraction.specs import ModelSpec from .exceptions import TableColumnsGap -from .renderer import TABLE_EXTENSION_URI, render_table_columns +from .renderer import TABLE_EXTENSION_URI, VECTOR_EXTENSION_URI, render_table_columns __all__ = ["TableColumnsOutput", "generate_table_columns_documents"] @@ -42,8 +42,17 @@ def generate_table_columns_documents( # A bare properties fragment rather than a whole Item: the extension # fields are what this target emits, and an Item would need id, geometry, # bbox, datetime and links, none of which come from a schema. + extensions = [TABLE_EXTENSION_URI] + if rendered.uses_vector_extension: + # A `vector:` field without the Vector extension declared here + # would not validate against either extension's schema. Every + # feature type currently emits one, so this is unconditional in + # practice; it is a condition because a model with no geometry + # column, or a geometry column carrying no `GeometryTypeConstraint`, + # must not declare an extension it never uses. + extensions.append(VECTOR_EXTENSION_URI) fragment = { - "stac_extensions": [TABLE_EXTENSION_URI], + "stac_extensions": extensions, "properties": rendered.stac_fields(), } outputs.append( diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/renderer.py b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/renderer.py index 82fcca90e..b9af8e906 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/renderer.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/renderer.py @@ -1,26 +1,58 @@ """Render a model spec to the STAC Table extension's `table:columns`. -The Column Object is `name` (required), `type`, `description`. That is the whole -target. Everything else the IR carries -- constraints, enum vocabularies, named -types, defaults, optionality, nested field descriptions -- has nowhere to go, so -this renderer's real output is the gap log beside the columns. +The Column Object is `name` (required), `type`, `description`, plus what v1.3.0 +permits from STAC common metadata (`data_type`, `unit`) and from the Vector +extension (`vector:geometry_types`). Everything else the IR carries -- +constraints, enum vocabularies, named types, defaults, optionality, nested +field descriptions -- has nowhere to go, so this renderer's real output is the +gap log beside the columns. -Two decisions are stated here rather than defaulted into, because nothing -downstream can falsify either: +Four decisions, none of which anything downstream can check: 1. **The type dialect.** `type` is an unconstrained string, and real catalogs - carry whatever their generating engine's stringifier produced -- two dialects - in mixed case across the 526 columns of published catalogs surveyed. This renderer - emits lowercase DuckDB SQL, matching the spec repo's own generated reference - catalog. No validator can tell a right dialect from a wrong one. + carry whatever their generating engine's stringifier produced, in more than + one dialect and inconsistent case. This renderer emits Arrow, because Arrow + is what Overture distributes: the release is GeoParquet, and these are the + names an Arrow reader reports opening it. Both the vocabulary and the + composite grammar come from `pyarrow`'s own stringifier, and they match + published catalogs, which name columns `int64`, `string`, `double`, + `binary` and `struct<...>`. Struct member order comes from the released + files rather than from a model's declaration order, since the type + describes the physical column. + + `type` and `data_type` therefore disagree on floats. `data_type` (decision + 3) is bound to STAC common metadata's fixed vocabulary, which names them + `float32`/`float64`; `type` is bound to Arrow, which names the same two + `float`/`double`. Harmonising them would make one field lie about the + vocabulary it claims to speak. 2. **Which walk supplies the union.** Walking only concrete arms misses fields that stay un-narrowed in the merged list; walking only the merged list drops - every non-first arm's contribution at a duplicated name. Both losses are - real and neither can raise, because the arms stringify identically. This - renderer walks both, emits the merged list (that is what a columnar sink - stores), and logs a `flatten-collision` for every name where the two - disagree. + every non-first arm's contribution at a duplicated name. Neither loss can + raise, because the arms stringify identically. This renderer walks both, + emits the merged list -- what a columnar sink stores -- and logs a + `flatten-collision` for every name where the two disagree. + +3. **`data_type` is Overture's own base-type name, not a translation.** STAC + common metadata's data-type vocabulary (int8/16/32/64, uint8/16/32/64, + float32/64, `other`) already matches Overture's numeric base-type names, so + this is an allowlist against that vocabulary rather than a lookup table. + Every column gets one -- `other` is a member of the vocabulary, not an + omission -- so most columns resolve to it and only the numerics carry + signal. + +4. **`vector:geometry_types` reads the schema's `GeometryTypeConstraint`, not + data.** The Vector extension defines the field as the geometry types present + in the dataset, which by the `table:row_count` precedent this renderer would + refuse. `GeometryTypeConstraint` is validated on every row, so it bounds + what a column can contain rather than reporting what it does. A column with + no such constraint gets neither the field nor a claim about it, and is + logged as a gap. + + `unit` is not emitted, because Overture Schema has no consistent way to + encode one. Where a unit varies per row it is a field (`Speed` carries a + `SpeedUnit`); where it is fixed it appears only in the description prose. + Neither is a column-level fact, and this renderer does not parse prose. """ from __future__ import annotations @@ -31,6 +63,8 @@ from enum import Enum from typing import Any +from overture.schema.system.geometric import GeometryTypeConstraint + from ..extraction.enum_extraction import extract_enum from ..extraction.field import ( AnyScalar, @@ -49,57 +83,101 @@ __all__ = [ "TABLE_EXTENSION_URI", + "VECTOR_EXTENSION_URI", "TableColumnsDocument", "render_table_columns", ] -TABLE_EXTENSION_URI = "https://stac-extensions.github.io/table/v1.2.0/schema.json" - -# Lowercase DuckDB SQL, the dialect the Portolan spec repo's own generated -# reference catalog emits. See the module docstring: this is a choice. -_DUCKDB_TYPES: dict[str, str] = { - "int8": "tinyint", - "int16": "smallint", - "int32": "integer", - "int64": "bigint", - "uint8": "utinyint", - "uint16": "usmallint", - "uint32": "uinteger", - "uint64": "ubigint", +TABLE_EXTENSION_URI = "https://stac-extensions.github.io/table/v1.3.0/schema.json" + +# Recommended (not required) by the Table extension's README for the column +# `vector:geometry_types` is emitted on -- see decision 4 in the module +# docstring. Declared here, not just used, because a field from this +# extension in `table:columns` without the extension in `stac_extensions` +# would not validate; `pipeline.py` adds it to that list exactly when this +# renderer emits the field. +VECTOR_EXTENSION_URI = "https://stac-extensions.github.io/vector/v0.1.0/schema.json" + +# Arrow type names, exactly as `pyarrow`'s own stringifier emits them +# (`str(pa.float64())` is `"double"`, not `"float64"`). See decision 1 in the +# module docstring. The sized integers look like a no-op mapping because they +# are one: Overture's base-type names already ARE Arrow's for every sized +# integer, exactly as they already are STAC's in `_STAC_DATA_TYPES` below. +# Only the floats, strings, bytes and time types need translating. +_ARROW_TYPES: dict[str, str] = { + "int8": "int8", + "int16": "int16", + "int32": "int32", + "int64": "int64", + "uint8": "uint8", + "uint16": "uint16", + "uint32": "uint32", + "uint64": "uint64", "float32": "float", "float64": "double", - "str": "varchar", - "bool": "boolean", - "bytes": "blob", - "datetime": "timestamp with time zone", - "date": "date", - "Geometry": "geometry", + "str": "string", + "bool": "bool", + "bytes": "binary", + "datetime": "timestamp[us, tz=UTC]", + "date": "date32[day]", } -# BBox is a plain class the codegen cannot walk, so the physical layout keeps a -# shared struct constant for it. Its members are fixed by the GeoParquet -# covering convention, and all three struct grammars in the wild describe the -# same four doubles. -_BBOX_TYPE = "struct(xmin double, ymin double, xmax double, ymax double)" +# The type string a shape with no Arrow type of its own collapses to. +_STRING_TYPE = "string" + +# Arrow has no geometry type. GeoParquet stores geometry as WKB in a plain +# binary column and carries the semantics in file-level metadata, so `binary` +# is what an Arrow reader reports for a geometry column. The erasure is logged +# rather than absorbed; see `_scalar_type`. +_GEOMETRY_TYPE = "binary" + +# STAC common metadata's `data_type` vocabulary. See decision 3 in the module +# docstring: Overture's own base-type names already ARE this vocabulary for +# every numeric primitive, so this is an allowlist, not a translation table. +_STAC_DATA_TYPES: dict[str, str] = { + "int8": "int8", + "int16": "int16", + "int32": "int32", + "int64": "int64", + "uint8": "uint8", + "uint16": "uint16", + "uint32": "uint32", + "uint64": "uint64", + "float32": "float32", + "float64": "float64", +} +_OTHER_DATA_TYPE = "other" -# Pydantic types with no SQL type of their own. Keyed by NAME rather than by +# BBox is a plain class the codegen cannot walk, so the physical layout keeps a +# shared struct constant for it. Member order is the `BBox` class's own +# declaration order. +# +# THAT DOES NOT MATCH THE ORDER OVERTURE SHIPS, which is `xmin, xmax, ymin, +# ymax` -- what this repo's PySpark `BBOX_STRUCT` declares and what a reader +# reports for a released file. The discrepancy is accepted rather than encoded: +# this renderer derives types from models, and a constant matching one +# publisher's file layout would be a fact about that publisher rather than +# about the schema. A consumer reading member order off this field will be +# wrong about it. +_BBOX_TYPE = "struct" + +# Pydantic types with no Arrow type of their own. Keyed by NAME rather than by # `issubclass`: in Pydantic v2 neither `HttpUrl` nor `EmailStr` is a `str` -# subclass, so the builtin fallback below never reaches them -- the same trap -# the Vecorel renderer records, found here the same way, by running it. -_PYDANTIC_AS_VARCHAR: frozenset[str] = frozenset( +# subclass, so the builtin fallback below never reaches them. +_PYDANTIC_AS_STRING: frozenset[str] = frozenset( {"HttpUrl", "AnyUrl", "AnyHttpUrl", "EmailStr", "UUID"} ) # Fallback keyed on the underlying Python type, for constrained-string NewTypes # the registry has no name entry for. `bool` precedes `int` because it is a -# subclass. +# subclass. A bare Python `int` widens to `int64`, Arrow's own default for one. _BUILTIN_TYPES: tuple[tuple[type, str], ...] = ( - (bool, "boolean"), - (_dt.datetime, "timestamp with time zone"), - (_dt.date, "date"), - (str, "varchar"), - (bytes, "blob"), - (int, "bigint"), + (bool, "bool"), + (_dt.datetime, "timestamp[us, tz=UTC]"), + (_dt.date, "date32[day]"), + (str, _STRING_TYPE), + (bytes, "binary"), + (int, "int64"), (float, "double"), ) @@ -115,7 +193,7 @@ class Column: # has no field for any of it. base_type: str | None # The underlying Python type, kept so a downstream target can resolve - # through the same fallback chain the SQL type does. + # through the same fallback chain the Arrow type does. python_type: builtins.type | None enum_name: str | None is_literal: bool @@ -123,9 +201,17 @@ class Column: has_default: bool is_required: bool constraint_names: tuple[str, ...] + # STAC common metadata's data-type vocabulary name for `base_type`. Always + # set -- `other` is a member of the vocabulary, not an omission. + data_type: str + # GeoJSON geometry type names from `GeometryTypeConstraint`, for a + # geometry column that carries one. `None` for every other column, and + # for a geometry column with no such constraint (see decision 4). + geometry_types: tuple[str, ...] | None = None def as_column_object(self) -> dict[str, Any]: - """Return the Column Object exactly: name, type, description when present. + """Return the Column Object: name, type, description when present, + `data_type` always, and `vector:geometry_types` when known. An absent description is an absent KEY, not an empty string. A silent empty string would make an undescribed column indistinguishable from a @@ -134,6 +220,9 @@ def as_column_object(self) -> dict[str, Any]: obj: dict[str, Any] = {"name": self.name, "type": self.type} if self.description is not None: obj["description"] = self.description + obj["data_type"] = self.data_type + if self.geometry_types is not None: + obj["vector:geometry_types"] = list(self.geometry_types) return obj @@ -149,11 +238,9 @@ class TableColumnsDocument: def stac_fields(self) -> dict[str, Any]: """Return the `table:` fields for a STAC object. - `table:row_count` is deliberately absent and cannot be otherwise: a row - count is a property of data, and this path has only a schema. Emitting a - placeholder would be a fabricated measurement; leaving it out silently - would be indistinguishable from forgetting it, which is what the gap log - is for. + `table:row_count` is absent: a row count is a property of data, and + this path has only a schema. Its omission is logged as a gap, so that + it is distinguishable from having been forgotten. """ fields: dict[str, Any] = { "table:columns": [c.as_column_object() for c in self.columns] @@ -162,6 +249,16 @@ def stac_fields(self) -> dict[str, Any]: fields["table:primary_geometry"] = self.primary_geometry return fields + @property + def uses_vector_extension(self) -> bool: + """Whether any column carries `vector:geometry_types`. + + The pipeline needs this to decide whether `stac_extensions` must also + declare the Vector extension -- a field from that extension without + its URI declared would not validate. + """ + return any(c.geometry_types is not None for c in self.columns) + @dataclass class _Ctx: @@ -224,6 +321,38 @@ def _unwrap(shape: FieldShape) -> FieldShape: return shape +def _data_type(base_type: str | None) -> str: + """Map a column's base type to the STAC common-metadata `data_type` vocabulary. + + Total, unlike `type`: `other` is itself a vocabulary member, so every + column gets a value. `base_type` is `None` for anything that is not a + terminal `Primitive` (a struct, array, map, enum, or union column), which + resolves to `other` exactly like a primitive with no numeric identity. + """ + if base_type is None: + return _OTHER_DATA_TYPE + return _STAC_DATA_TYPES.get(base_type, _OTHER_DATA_TYPE) + + +def _geometry_types_of(shape: FieldShape) -> tuple[str, ...] | None: + """Return the GeoJSON geometry type names a geometry column is constrained to. + + Reads `GeometryTypeConstraint.allowed_types` -- see decision 4 in the + module docstring for why this schema constraint is a legitimate source for + `vector:geometry_types`. Returns `None` when the shape isn't a geometry + column, or is one with no such constraint attached. + """ + inner = _unwrap(shape) + if not isinstance(inner, Primitive): + return None + for source in inner.constraints: + if isinstance(source.constraint, GeometryTypeConstraint): + return tuple( + sorted(t.geo_json_type for t in source.constraint.allowed_types) + ) + return None + + def _top_fields(spec: ModelSpec, ctx: _Ctx) -> list[FieldSpec]: """Top-level columns, the way a columnar sink sees them.""" if isinstance(spec, UnionSpec): @@ -294,7 +423,7 @@ def _collect_type_names(shape: FieldShape, acc: set[str], depth: int = 0) -> Non def _scalar_type( scalar: Primitive | LiteralScalar | AnyScalar, ctx: _Ctx, path: str ) -> str: - """Map a terminal shape to a DuckDB type string, logging what that costs.""" + """Map a terminal shape to an Arrow type string, logging what that costs.""" if isinstance(scalar, LiteralScalar): ctx.gap( path, @@ -304,16 +433,16 @@ def _scalar_type( f"{len(scalar.values)} value(s); a Column Object has no enum, const " "or pattern keyword, so the column is a bare string.", ) - return "varchar" + return _STRING_TYPE if isinstance(scalar, AnyScalar): ctx.gap( path, "target-gap", "any", - "typing.Any has no column type. Emitted as varchar to match what the " + "typing.Any has no column type. Emitted as string to match what the " "physical layout stores, which is a stringification, not a type.", ) - return "varchar" + return _STRING_TYPE src = enum_source(scalar) if src is not None: spec = extract_enum(src) @@ -325,27 +454,39 @@ def _scalar_type( f"{src.__name__}: {len(spec.members)} members ({described} described) " "collapse to a bare string; table:columns has no enum keyword.", ) - return "varchar" + return _STRING_TYPE if scalar.base_type == "BBox": return _BBOX_TYPE - mapped = _DUCKDB_TYPES.get(scalar.base_type) + if scalar.base_type == "Geometry": + ctx.gap( + path, + "target-dialect", + "semantic type erased", + "Arrow has no geometry type: GeoParquet stores geometry as WKB in a " + "binary column and keeps the semantics in file-level metadata, which " + "a type string cannot reach. vector:geometry_types carries the part " + "of the semantic the schema can assert; the type string carries none " + "of it.", + ) + return _GEOMETRY_TYPE + mapped = _ARROW_TYPES.get(scalar.base_type) if mapped is not None: return mapped source_type = scalar.source_type - if scalar.base_type in _PYDANTIC_AS_VARCHAR or ( - source_type is not None and source_type.__name__ in _PYDANTIC_AS_VARCHAR + if scalar.base_type in _PYDANTIC_AS_STRING or ( + source_type is not None and source_type.__name__ in _PYDANTIC_AS_STRING ): ctx.gap( path, "target-dialect", "semantic type erased", - f"{scalar.base_type} has no SQL type of its own and stringifies to " - "varchar; the semantics live only in the name, which the column does " + f"{scalar.base_type} has no Arrow type of its own and stringifies to " + "string; the semantics live only in the name, which the column does " "not carry.", ) - return "varchar" + return _STRING_TYPE if source_type is not None: - by_name = _DUCKDB_TYPES.get(source_type.__name__) + by_name = _ARROW_TYPES.get(source_type.__name__) if by_name is not None: return by_name for py_type, name in _BUILTIN_TYPES: @@ -354,7 +495,7 @@ def _scalar_type( path, "target-dialect", "semantic type erased", - f"{source_type.__name__} has no SQL type of its own and " + f"{source_type.__name__} has no Arrow type of its own and " f"stringifies to {name}; the semantics live only in the name, " "which the column does not carry.", ) @@ -363,18 +504,25 @@ def _scalar_type( path, "renderer-gap", "unmapped base type", - f"no DuckDB type for base_type={scalar.base_type!r} " + f"no Arrow type for base_type={scalar.base_type!r} " f"(source_type={getattr(source_type, '__name__', None)!r}).", ) - return "varchar" + return _STRING_TYPE def _type_string(shape: FieldShape, ctx: _Ctx, path: str) -> str: - """Build the DuckDB type string for a column, recursing into nesting. - - Nesting lives here, not in the column name: of the 526 columns across the - published catalogs surveyed, zero contain a dot, so a flattener that invents - dotted paths would describe a table nobody publishes. + """Build the Arrow type string for a column, recursing into nesting. + + Nesting lives in the type string, not in the column name: published + catalogs do not put dots in column names. + + Composites use Arrow's angle-bracket grammar -- `struct`, + `list`, `map` -- taken from `pyarrow`'s + stringifier. A list's child is named `item`, pyarrow's constructor form. A + round-trip through a Parquet file renames it `element`, since the Parquet + LIST logical type fixes that name, and the same round-trip stringifies a + map as `map`, appending the Parquet group name. + Those are artifacts of the file rather than type names. """ match shape: case NewTypeShape(name=name, inner=inner): @@ -382,11 +530,11 @@ def _type_string(shape: FieldShape, ctx: _Ctx, path: str) -> str: ctx.named_sites.append(f"{path}:{name}") return _type_string(inner, ctx, path) case ArrayOf(element=element): - return f"{_type_string(element, ctx, path + '[]')}[]" + return f"list" case MapOf(key=key, value=value): key_type = _type_string(key, ctx, path + "/key") value_type = _type_string(value, ctx, path + "/value") - return f"map({key_type}, {value_type})" + return f"map<{key_type}, {value_type}>" case ModelRef(model=model, starts_cycle=starts_cycle): ctx.named_sites.append(f"{path}:{model.name}") if starts_cycle or model.name in ctx.stack: @@ -443,7 +591,7 @@ def _type_string(shape: FieldShape, ctx: _Ctx, path: str) -> str: def _struct_string(fields: list[FieldSpec], ctx: _Ctx, path: str, owner: str) -> str: - """`struct(name type, ...)`, logging the descriptions that die inside it.""" + """`struct`, logging the descriptions that die inside it.""" if not fields: ctx.gap( path, @@ -451,7 +599,7 @@ def _struct_string(fields: list[FieldSpec], ctx: _Ctx, path: str, owner: str) -> "empty struct", f"{owner} has no fields; an empty struct column cannot carry data.", ) - return "struct()" + return "struct<>" parts = [] for f in fields: member_path = f"{path}/{f.name}" @@ -465,8 +613,8 @@ def _struct_string(fields: list[FieldSpec], ctx: _Ctx, path: str, owner: str) -> "string; the description does not.", ) _log_field_surplus(f, ctx, member_path, nested=True) - parts.append(f"{f.name} {_type_string(f.shape, ctx, member_path)}") - return f"struct({', '.join(parts)})" + parts.append(f"{f.name}: {_type_string(f.shape, ctx, member_path)}") + return f"struct<{', '.join(parts)}>" def _constraint_names(shape: FieldShape, acc: list[str], depth: int = 0) -> None: @@ -518,8 +666,7 @@ def _log_field_surplus(f: FieldSpec, ctx: _Ctx, path: str, *, nested: bool) -> N f"{where} is {'required' if f.is_required else 'optional'}" f"{' and nullable' if f.is_optional else ''}; a Column Object has no " "required, nullable or optional keyword, so this is not expressible at " - "all -- strictly less than Vecorel, which at least had a root required " - "list.", + "all.", ) names: list[str] = [] _constraint_names(f.shape, names) @@ -562,12 +709,23 @@ def render_table_columns( ) constraints: list[str] = [] _constraint_names(f.shape, constraints) + base_type = inner.base_type if isinstance(inner, Primitive) else None + geometry_types = _geometry_types_of(f.shape) + if base_type == "Geometry" and geometry_types is None: + ctx.gap( + path, + "ir-gap", + "geometry types", + "geometry column has no GeometryTypeConstraint; " + "vector:geometry_types would have to guess which types can " + "appear, so it is omitted rather than claiming all seven.", + ) columns.append( Column( name=f.name, type=_type_string(f.shape, ctx, path), description=f.description, - base_type=inner.base_type if isinstance(inner, Primitive) else None, + base_type=base_type, python_type=( inner.source_type if isinstance(inner, Primitive) @@ -580,6 +738,8 @@ def render_table_columns( has_default=f.default is not NO_DEFAULT, is_required=f.is_required, constraint_names=tuple(constraints), + data_type=_data_type(base_type), + geometry_types=geometry_types, ) ) diff --git a/packages/overture-schema-codegen/tests/test_stac_table_columns.py b/packages/overture-schema-codegen/tests/test_stac_table_columns.py index 4148e7e89..10e6920ba 100644 --- a/packages/overture-schema-codegen/tests/test_stac_table_columns.py +++ b/packages/overture-schema-codegen/tests/test_stac_table_columns.py @@ -2,10 +2,11 @@ A Column Object is `name`, `type`, `description`, so most of what this renderer knows it cannot say. These cover the decisions that survive that flattening: how -an absent description is spelled, what happens at a union collision, and the two +an absent description is rendered, what happens at a union collision, and the two cases where guessing is worse than refusing. """ +import json from typing import Annotated, Literal import pytest @@ -15,14 +16,31 @@ from overture.schema.codegen.stac_table_columns.exceptions import ( TableColumnsUnrepresentable, ) -from overture.schema.codegen.stac_table_columns.renderer import render_table_columns -from overture.schema.system.geometric import Geometry +from overture.schema.codegen.stac_table_columns.pipeline import ( + generate_table_columns_documents, +) +from overture.schema.codegen.stac_table_columns.renderer import ( + VECTOR_EXTENSION_URI, + render_table_columns, +) +from overture.schema.system.geometric import ( + BBox, + Geometry, + GeometryType, + GeometryTypeConstraint, +) +from overture.schema.system.numeric import float32, float64, int32, int64 def _capabilities(doc: object) -> set[str]: return {g.capability for g in doc.gaps} # type: ignore[attr-defined] +def _columns(model: type[BaseModel]) -> dict[str, dict[str, object]]: + doc = render_table_columns(extract_model(model)) + return {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + def test_undescribed_column_omits_the_key_entirely() -> None: """A column with no description omits `description`, never emits `""`. @@ -147,6 +165,231 @@ class M(BaseModel): render_table_columns(extract_model(M), strict=True) +def test_scalar_columns_use_arrow_names() -> None: + """`type` uses Arrow's names, the ones a reader of the released GeoParquet + reports -- `double` and `float`, not `float64`/`float32`, and `string`, + `bool`, `binary` rather than DuckDB's `varchar`, `boolean`, `blob`. Every + name here came from `pyarrow`'s own stringifier.""" + + class M(BaseModel): + big: float64 + small: float32 + label: str + flag: bool + raw: bytes + count: int64 + + columns = _columns(M) + + assert columns["big"]["type"] == "double" + assert columns["small"]["type"] == "float" + assert columns["label"]["type"] == "string" + assert columns["flag"]["type"] == "bool" + assert columns["raw"]["type"] == "binary" + assert columns["count"]["type"] == "int64" + + +def test_composite_columns_use_arrow_angle_bracket_grammar() -> None: + """Structs, lists and maps are the part most likely to drift. + + DuckDB wrote these `struct(name type, ...)`, `type[]` and + `map(varchar, varchar)`; Arrow uses angle brackets and names a + list's child `item`. Pinning the whole nested string is the point -- a + per-scalar check would pass against either grammar. + """ + + class Inner(BaseModel): + depth: float64 + tag: str + + class M(BaseModel): + nested: Inner + labels: list[str] + matrix: list[list[int32]] + lookup: dict[str, str] + + columns = _columns(M) + + assert columns["nested"]["type"] == "struct" + assert columns["labels"]["type"] == "list" + assert columns["matrix"]["type"] == "list>" + assert columns["lookup"]["type"] == "map" + + +def test_geometry_is_binary_and_the_erasure_is_logged() -> None: + """Arrow has no geometry type; GeoParquet stores WKB in a binary column. + + `binary` is what an Arrow reader reports for Overture's own released + geometry column, so it is what this renderer emits. The semantic the + DuckDB name carried in the type string is gone, and the gap log is + what records that -- `vector:geometry_types` carries only the part the + schema can assert. + """ + + class M(BaseModel): + geometry: Annotated[Geometry, GeometryTypeConstraint(GeometryType.POINT)] + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert columns["geometry"]["type"] == "binary" + assert columns["geometry"]["vector:geometry_types"] == ["Point"] + assert any(g.kind == "target-dialect" and "geometry" in g.detail for g in doc.gaps) + + +def test_type_and_data_type_differ_for_floats_on_purpose() -> None: + """The two fields speak different vocabularies, and neither is bent. + + `data_type` is bound to STAC common metadata, which names floats + `float32`/`float64`; `type` is bound to Arrow, which names the same two + `float`/`double`. + """ + + class M(BaseModel): + small: float32 + big: float64 + + columns = _columns(M) + + assert columns["small"]["type"] == "float" + assert columns["small"]["data_type"] == "float32" + assert columns["big"]["type"] == "double" + assert columns["big"]["data_type"] == "float64" + + +def test_bbox_struct_members_follow_the_model_not_a_publisher() -> None: + """The bbox struct's member order is the `BBox` class's declaration order. + + It does not match the order Overture ships (`xmin, xmax, ymin, ymax`, what + the PySpark `BBOX_STRUCT` declares). That discrepancy is accepted: this + renderer derives types from models, and matching one publisher's file + layout would encode a fact about that publisher rather than the schema. + """ + + class M(BaseModel): + bbox: BBox + + assert ( + _columns(M)["bbox"]["type"] + == "struct" + ) + + +def test_data_type_maps_a_numeric_column_to_the_stac_vocabulary() -> None: + """A numeric base type resolves to STAC common metadata's own name. + + Overture's `int32`/`float64` names already ARE the vocabulary's names, so + this is a pass-through, not a translation. + """ + + class M(BaseModel): + count: int32 + ratio: float64 + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert columns["count"]["data_type"] == "int32" + assert columns["ratio"]["data_type"] == "float64" + + +def test_data_type_falls_back_to_other_for_a_non_numeric_column() -> None: + """A string, a struct, and everything else with no numeric identity get + `other` -- a member of the STAC vocabulary, not an omission.""" + + class Nested(BaseModel): + value: str + + class M(BaseModel): + label: str + nested: Nested + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert columns["label"]["data_type"] == "other" + assert columns["nested"]["data_type"] == "other" + + +def test_geometry_type_constraint_becomes_vector_geometry_types() -> None: + """A single allowed geometry type becomes a one-element `vector:geometry_types`. + + v1.3.0's README defines no `geometry_type` on the Column Object at all, + only `vector:geometry_types` from the separate Vector extension (see the + renderer module docstring). + """ + + class M(BaseModel): + geometry: Annotated[ + Geometry, + GeometryTypeConstraint(GeometryType.POINT), + Field(description="the shape"), + ] + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert columns["geometry"]["vector:geometry_types"] == ["Point"] + + +def test_multiple_allowed_geometry_types_are_sorted_geojson_names() -> None: + """Several allowed types become a sorted list of GeoJSON (not snake_case) names.""" + + class M(BaseModel): + geometry: Annotated[ + Geometry, + GeometryTypeConstraint(GeometryType.MULTI_POLYGON, GeometryType.POLYGON), + ] + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert columns["geometry"]["vector:geometry_types"] == ["MultiPolygon", "Polygon"] + + +def test_unconstrained_geometry_omits_vector_geometry_types_and_logs_a_gap() -> None: + """No `GeometryTypeConstraint` means no claim about which types can appear. + + Emitting all seven would assert something the schema never said; omitting + silently would look identical to forgetting the field. The gap log is the + only record. + """ + + class M(BaseModel): + geometry: Geometry + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert "vector:geometry_types" not in columns["geometry"] + assert any( + g.kind == "ir-gap" and g.capability == "geometry types" for g in doc.gaps + ) + + +def test_pipeline_declares_the_vector_extension_only_when_used() -> None: + """`stac_extensions` gains the Vector extension exactly when a column needs it. + + A `vector:` field with its extension undeclared would not validate; a + model with no geometry constraint should not carry a URI for a field it + never emits. + """ + + class Constrained(BaseModel): + geometry: Annotated[Geometry, GeometryTypeConstraint(GeometryType.POINT)] + + class Unconstrained(BaseModel): + geometry: Geometry + + [with_vector] = generate_table_columns_documents([extract_model(Constrained)]) + [without_vector] = generate_table_columns_documents([extract_model(Unconstrained)]) + + assert VECTOR_EXTENSION_URI in json.loads(with_vector.stac)["stac_extensions"] + assert ( + VECTOR_EXTENSION_URI not in json.loads(without_vector.stac)["stac_extensions"] + ) + + def test_gap_log_is_non_empty_for_a_real_model() -> None: """The did-happen control for every assertion above. From f92c0897f4fa91346684d9811d30615f826063b1 Mon Sep 17 00:00:00 2001 From: Seth Fitzsimmons Date: Wed, 16 Sep 2026 21:55:30 -0700 Subject: [PATCH 4/5] fix(codegen): serialize STAC fragments as UTF-8 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit json.dumps defaults to ensure_ascii=True, so every non-ASCII character in a model description left the renderer as a \uXXXX escape -- `if—and-only-if` in the one place the id column explains GERS. Both spellings parse to the same string, but RFC 8259 mandates UTF-8 for interchange and these fragments are read by people as often as by parsers. Signed-off-by: Seth Fitzsimmons --- .../codegen/stac_table_columns/pipeline.py | 5 +++- .../tests/test_stac_table_columns.py | 24 +++++++++++++++++++ 2 files changed, 28 insertions(+), 1 deletion(-) diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py index d0d7b8aeb..bef4e3dbb 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py @@ -58,7 +58,10 @@ def generate_table_columns_documents( outputs.append( TableColumnsOutput( model=spec.name, - stac=json.dumps(fragment, indent=2) + "\n", + # ensure_ascii=False: RFC 8259 mandates UTF-8 for interchange, + # and these fragments carry model prose -- an escaped em-dash + # parses the same and reads as a defect. + stac=json.dumps(fragment, indent=2, ensure_ascii=False) + "\n", stac_path=PurePosixPath(f"{stem}.json"), gaps=rendered.gaps, ) diff --git a/packages/overture-schema-codegen/tests/test_stac_table_columns.py b/packages/overture-schema-codegen/tests/test_stac_table_columns.py index 10e6920ba..bc8fbf4fd 100644 --- a/packages/overture-schema-codegen/tests/test_stac_table_columns.py +++ b/packages/overture-schema-codegen/tests/test_stac_table_columns.py @@ -405,3 +405,27 @@ class M(BaseModel): assert doc.gaps assert "constraint" in _capabilities(doc) + + +def test_fragment_carries_non_ascii_text_literally() -> None: + """Descriptions serialize as UTF-8, not as `\\uXXXX` escapes. + + Both spellings parse to the same string, so nothing downstream depends on + this -- but the fragment is read by people as often as by parsers, and an + escaped em-dash reads as a defect in a document whose whole content is + prose lifted from the models. + """ + + class M(BaseModel): + field: str = Field(description="An em-dash — and a café.") + + [doc] = generate_table_columns_documents([extract_model(M)]) + [column] = [ + c + for c in json.loads(doc.stac)["properties"]["table:columns"] + if c["name"] == "field" + ] + + assert "An em-dash — and a café." in doc.stac + assert "\\u" not in doc.stac + assert column["description"] == "An em-dash — and a café." From 1422ae58fbc67009e6c220afdf1eeb393f41512d Mon Sep 17 00:00:00 2001 From: Seth Fitzsimmons Date: Wed, 16 Sep 2026 22:01:15 -0700 Subject: [PATCH 5/5] fix(codegen): write generated files as UTF-8 `Path.write_text` with no `encoding` uses the locale's, which through Python 3.14 is whatever the environment says. On an ASCII locale the writer did not produce mojibake -- it raised UnicodeEncodeError on the first em-dash in a model description and wrote nothing, which the markdown format has been reachable by all along. Generated Python is decoded as UTF-8 by PEP 3120 and JSON is UTF-8 by RFC 8259, so the locale is never the right answer here. The test runs the writer in a subprocess under LC_ALL=C: the encoding is bound at interpreter start, so an in-process check would read this machine's UTF-8 and pass either way. Signed-off-by: Seth Fitzsimmons --- .../changelog.d/724.bugfix.md | 1 + .../src/overture/schema/codegen/cli.py | 7 ++- .../overture-schema-codegen/tests/test_cli.py | 53 +++++++++++++++++++ 3 files changed, 59 insertions(+), 2 deletions(-) create mode 100644 packages/overture-schema-codegen/changelog.d/724.bugfix.md diff --git a/packages/overture-schema-codegen/changelog.d/724.bugfix.md b/packages/overture-schema-codegen/changelog.d/724.bugfix.md new file mode 100644 index 000000000..90b749c93 --- /dev/null +++ b/packages/overture-schema-codegen/changelog.d/724.bugfix.md @@ -0,0 +1 @@ +Fixed `generate --output-dir` raising `UnicodeEncodeError` on a non-UTF-8 locale: generated files are now written as UTF-8 rather than in the locale's encoding, which through Python 3.14 could be ASCII and refused the first em-dash in a model description. diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py b/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py index 5a20509c5..cf4ff8e97 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py @@ -43,7 +43,10 @@ def _write_output( if output_dir: file_path = output_dir / output_path file_path.parent.mkdir(parents=True, exist_ok=True) - file_path.write_text(content) + # UTF-8, not the locale's encoding: generated Python is decoded as + # UTF-8 (PEP 3120) and JSON is UTF-8 by RFC 8259, and on an ASCII + # locale the default raises on the first em-dash in a description. + file_path.write_text(content, encoding="utf-8") else: click.echo(content) click.echo() # separate entries with a blank line in stdout mode @@ -235,7 +238,7 @@ def _write_category_files( file_path = output_dir / dir_path / "_category_.json" file_path.parent.mkdir(parents=True, exist_ok=True) - file_path.write_text(json.dumps(category, indent=2) + "\n") + file_path.write_text(json.dumps(category, indent=2) + "\n", encoding="utf-8") def main() -> None: diff --git a/packages/overture-schema-codegen/tests/test_cli.py b/packages/overture-schema-codegen/tests/test_cli.py index dd90c6880..fee01391e 100644 --- a/packages/overture-schema-codegen/tests/test_cli.py +++ b/packages/overture-schema-codegen/tests/test_cli.py @@ -1,7 +1,10 @@ """Tests for CLI entrypoint.""" import json +import os import re +import subprocess +import sys from pathlib import Path import pytest @@ -541,3 +544,53 @@ def test_used_by_sections_appear_in_markdown( break assert has_used_by, "No 'Used By' sections found in any generated markdown" + + +_NON_ASCII = "em-dash — café" + +_WRITE_PROBE = """ +import json, locale, sys, tempfile +from pathlib import Path, PurePosixPath +from overture.schema.codegen.cli import _write_output + +out = Path(tempfile.mkdtemp()) +_write_output({text!r}, out, PurePosixPath("probe.txt")) +print(json.dumps({{ + "encoding": locale.getpreferredencoding(False), + "text": (out / "probe.txt").read_bytes().decode("utf-8"), +}})) +""" + + +def test_generated_files_are_utf8_under_a_non_utf8_locale() -> None: + """Written artifacts are UTF-8 regardless of the machine's locale. + + `Path.write_text` with no `encoding` uses the locale's, which through + Python 3.14 is whatever the environment says -- so on an ASCII locale + every description carrying an em-dash raised `UnicodeEncodeError` and no + file was written at all. Generated Python is decoded as UTF-8 by PEP + 3120 and JSON is UTF-8 by RFC 8259, so the locale is never the right + answer here. + + The subprocess is the point: the encoding is bound at interpreter start, + so an in-process check would read this machine's UTF-8 and pass either + way. + """ + proc = subprocess.run( + [sys.executable, "-c", _WRITE_PROBE.format(text=_NON_ASCII)], + capture_output=True, + text=True, + env={ + **os.environ, + "LC_ALL": "C", + "LANG": "C", + "PYTHONUTF8": "0", + "PYTHONCOERCECLOCALE": "0", + }, + ) + + assert proc.returncode == 0, proc.stderr + payload = json.loads(proc.stdout) + if payload["encoding"].lower().replace("-", "") in {"utf8", "utf"}: + pytest.skip(f"locale could not be forced off UTF-8: {payload['encoding']}") + assert payload["text"] == _NON_ASCII