diff --git a/packages/overture-schema-codegen/changelog.d/723.feature.md b/packages/overture-schema-codegen/changelog.d/723.feature.md new file mode 100644 index 000000000..3b82423ca --- /dev/null +++ b/packages/overture-schema-codegen/changelog.d/723.feature.md @@ -0,0 +1 @@ +Added a `stac-table-columns` codegen format rendering the STAC Table extension's (v1.3.0) `table:columns` from the models, with a gap log recording what a flat column list cannot hold. Column types use Arrow's names, the vocabulary a reader of Overture's released GeoParquet reports. Each Column Object also carries STAC common metadata's `data_type` and, for the geometry column, the Vector extension's `vector:geometry_types` (declared in `stac_extensions` alongside the Table extension when emitted). Field defaults and deprecation now carry through extraction so the renderer can report both. diff --git a/packages/overture-schema-codegen/changelog.d/724.bugfix.md b/packages/overture-schema-codegen/changelog.d/724.bugfix.md new file mode 100644 index 000000000..90b749c93 --- /dev/null +++ b/packages/overture-schema-codegen/changelog.d/724.bugfix.md @@ -0,0 +1 @@ +Fixed `generate --output-dir` raising `UnicodeEncodeError` on a non-UTF-8 locale: generated files are now written as UTF-8 rather than in the locale's encoding, which through Python 3.14 could be ASCII and refused the first em-dash in a model description. diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py b/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py index a5687d817..cf4ff8e97 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/cli.py @@ -23,12 +23,13 @@ from .markdown.pipeline import generate_markdown_pages from .pyspark.pipeline import generate_pyspark_modules from .spec_discovery import extract_alias_spec, extract_model_spec +from .stac_table_columns.pipeline import generate_table_columns_documents log = logging.getLogger(__name__) __all__ = ["cli"] -_OUTPUT_FORMATS = ("markdown", "pyspark") +_OUTPUT_FORMATS = ("markdown", "pyspark", "stac-table-columns") _FEATURE_FRONTMATTER = "---\nsidebar_position: 1\n---\n\n" @@ -42,7 +43,10 @@ def _write_output( if output_dir: file_path = output_dir / output_path file_path.parent.mkdir(parents=True, exist_ok=True) - file_path.write_text(content) + # UTF-8, not the locale's encoding: generated Python is decoded as + # UTF-8 (PEP 3120) and JSON is UTF-8 by RFC 8259, and on an ASCII + # locale the default raises on the first em-dash in a description. + file_path.write_text(content, encoding="utf-8") else: click.echo(content) click.echo() # separate entries with a blank line in stdout mode @@ -121,6 +125,8 @@ def generate( if output_format == "pyspark": _generate_pyspark(model_specs, output_dir, test_output_dir) + elif output_format == "stac-table-columns": + _generate_table_columns(model_specs, output_dir) else: # RootModel entry points yield no ModelSpec, so they document as # named aliases -- reachable no other way, since a RootModel field @@ -176,6 +182,22 @@ def _generate_pyspark( _write_output(mod.content, test_output_dir, mod.path) +def _generate_table_columns( + model_specs: list[ModelSpec], + output_dir: Path | None, +) -> None: + """Generate one STAC `table:` properties fragment per model. + + Every model emits, including the discriminated-union root: a flat column + list is what a columnar sink stores for a union. The gap count is logged + per model because it, not the fragment, is what a flattening target has to + be judged on. + """ + for doc in generate_table_columns_documents(model_specs): + _write_output(doc.stac, output_dir, doc.stac_path) + log.info("%s: %d gaps", doc.model, len(doc.gaps)) + + def _ancestor_dirs(paths: set[PurePosixPath]) -> set[PurePosixPath]: """Collect all ancestor directories for a set of file paths.""" dirs: set[PurePosixPath] = set() @@ -216,7 +238,7 @@ def _write_category_files( file_path = output_dir / dir_path / "_category_.json" file_path.parent.mkdir(parents=True, exist_ok=True) - file_path.write_text(json.dumps(category, indent=2) + "\n") + file_path.write_text(json.dumps(category, indent=2) + "\n", encoding="utf-8") def main() -> None: diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/model_extraction.py b/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/model_extraction.py index 7371db208..1bb902e92 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/model_extraction.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/model_extraction.py @@ -5,6 +5,7 @@ from collections.abc import Mapping from pydantic import BaseModel +from pydantic.experimental.missing_sentinel import MISSING from pydantic.fields import FieldInfo from pydantic_core import PydanticUndefined @@ -15,7 +16,7 @@ ModelRef, UnionRef, ) -from .specs import FieldSpec, RecordSpec, is_model_class +from .specs import NO_DEFAULT, FieldSpec, RecordSpec, is_model_class from .type_analyzer import ( ModelResolver, UnionResolver, @@ -46,6 +47,44 @@ def resolve_field_alias(field_name: str, field_info: FieldInfo) -> str: return field_name +def _field_default(field_info: FieldInfo) -> object: + """Return the field's declared default, or `NO_DEFAULT`. + + Typed `object` rather than `Any`: a default really is an arbitrary value, but + `object` says so without switching off type checking at every call site. + + `default_factory` is deliberately not consulted: a factory is behavior, and + calling it here would capture one sample of a value meant to be produced per + instance. A field with only a factory extracts as `NO_DEFAULT` -- which is + accurate about what a static schema can carry, not a silent drop. + + Two sentinels mean "no default", not one. `PydanticUndefined` is Pydantic's, + and `MISSING` is the one `Omitable[T]` installs (`Field(default=MISSING)`) to + get JSON Schema omissibility instead of Pydantic nullability -- see + `overture.schema.system.optionality`. `MISSING` is machinery for "this key may + be absent", never a value anyone declared, so extracting it as a default makes + every `Omitable` field claim a default it does not have. + """ + if field_info.default is PydanticUndefined or field_info.default is MISSING: + return NO_DEFAULT + return field_info.default + + +def _field_deprecation(field_info: FieldInfo) -> tuple[bool, str | None]: + """Return `(is_deprecated, message)` from Pydantic's `deprecated`. + + Pydantic admits `True`, a string message, or a `warnings.deprecated` + instance. All three mean deprecated; only the string carries prose, and it is + kept beside the flag rather than folded into it. + """ + deprecated = field_info.deprecated + if deprecated is None or deprecated is False: + return False, None + if isinstance(deprecated, str): + return True, deprecated + return True, None + + def _is_field_required(field_info: FieldInfo, is_optional: bool) -> bool: """Determine whether a field is required (no default and not Optional).""" has_default = ( @@ -178,6 +217,7 @@ def _extract_model_recursive( # misses those constraints. Reattach them at the topmost # constraint-bearing layer. shape = attach_field_metadata(shape, field_info) + is_deprecated, deprecation_message = _field_deprecation(field_info) fields.append( FieldSpec( name=resolve_field_alias(field_name, field_info), @@ -185,6 +225,9 @@ def _extract_model_recursive( description=field_info.description or ti_description, is_required=_is_field_required(field_info, is_optional), is_optional=is_optional, + default=_field_default(field_info), + is_deprecated=is_deprecated, + deprecation_message=deprecation_message, ) ) diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/specs.py b/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/specs.py index 6b5259da1..3fc4ed340 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/specs.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/extraction/specs.py @@ -120,6 +120,23 @@ class EnumSpec(_SourceTypeIdentityMixin): source_type: type | None = None +class _NoDefault: + """The absence of a default, distinct from a default of `None`. + + `None` is a legal default value, so `default=None` cannot mean "no default" + without collapsing the two. A dedicated sentinel keeps them apart, and reads + at a use site (`spec.default is NO_DEFAULT`) as the question being asked. + """ + + __slots__ = () + + def __repr__(self) -> str: # pragma: no cover - debugging aid + return "NO_DEFAULT" + + +NO_DEFAULT = _NoDefault() + + @dataclass class FieldSpec: """Specification for a model field: header metadata plus structural shape. @@ -134,6 +151,21 @@ class FieldSpec: description: str | None = None is_required: bool = True is_optional: bool = False + # The field's declared default, or `NO_DEFAULT` when it has none. A + # `default_factory` does NOT land here: a factory is behavior, and calling it + # at extraction time would freeze one sample of a value whose whole point is + # to be produced per instance. + default: Any = NO_DEFAULT + # Pydantic's `deprecated`, normalized to a flag. Pydantic admits `True` or a + # deprecation *message*; a message narrows to `True` here, because every + # target this IR feeds has a boolean keyword and none has anywhere to put the + # prose. The renderer logs that narrowing where it applies -- the IR's job is + # to carry the fact, not to decide what a target does about it. + is_deprecated: bool = False + # The message form of `deprecated`, kept beside the flag so a target that + # grows somewhere to put it does not have to re-extract. `None` when the + # field declared a bare `True`, or nothing at all. + deprecation_message: str | None = None @dataclass diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/__init__.py b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/__init__.py new file mode 100644 index 000000000..8dedce803 --- /dev/null +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/__init__.py @@ -0,0 +1,6 @@ +"""STAC Table extension (`table:columns`) rendering. + +A flattening target: a Column Object is `name`, `type`, `description` and +nothing else, so this renderer's output is as much a record of what a flat +column list cannot hold as it is the list itself. +""" diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/exceptions.py b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/exceptions.py new file mode 100644 index 000000000..1911ea3ad --- /dev/null +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/exceptions.py @@ -0,0 +1,52 @@ +"""Typed record of everything the `table:columns` emit could not carry. + +`flatten-collision` is the kind only a flattening target produces: two union +arms contributing the same column name, where a columnar sink keeps one and the +other's meaning is gone with no trace in the output. `unrepresentable-in-pydantic` +runs the other way -- the target asks for something Pydantic has no way to +declare, so the gap is a possible Overture Schema feature rather than a defect. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import Literal + +__all__ = ["Kind", "TableColumnsGap", "TableColumnsUnrepresentable"] + +Kind = Literal[ + "ir-gap", + "target-gap", + "target-dialect", + "flatten-collision", + "unrepresentable-in-pydantic", + "renderer-gap", +] + + +@dataclass(frozen=True, slots=True) +class TableColumnsGap: + """One capability that did not survive the emit. + + `path` locates it in the emitted document, `capability` names what was lost + in the vocabulary of whichever side owns the loss, and `detail` carries the + mechanism. Classification is by mechanism, not by keyword. + """ + + model: str + path: str + kind: Kind + capability: str + detail: str + + +class TableColumnsUnrepresentable(Exception): + """Raised in strict mode, or when the renderer cannot proceed at all.""" + + def __init__(self, gaps: tuple[TableColumnsGap, ...]) -> None: + self.gaps = gaps + head = gaps[0] + super().__init__( + f"{head.model}{head.path}: {head.capability} ({head.kind}) -- {head.detail}" + + (f" [+{len(gaps) - 1} more]" if len(gaps) > 1 else "") + ) diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py new file mode 100644 index 000000000..bef4e3dbb --- /dev/null +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/pipeline.py @@ -0,0 +1,69 @@ +"""STAC `table:columns` generation pipeline: render documents without I/O. + +One artifact per model -- a STAC Item properties fragment carrying the `table:` +fields -- and the gap log beside it, which for a target this thin is as much the +deliverable as the fragment. +""" + +from __future__ import annotations + +import json +from collections.abc import Sequence +from dataclasses import dataclass +from pathlib import PurePosixPath + +from overture.schema.system.case import to_snake_case + +from ..extraction.specs import ModelSpec +from .exceptions import TableColumnsGap +from .renderer import TABLE_EXTENSION_URI, VECTOR_EXTENSION_URI, render_table_columns + +__all__ = ["TableColumnsOutput", "generate_table_columns_documents"] + + +@dataclass(frozen=True, slots=True) +class TableColumnsOutput: + """A rendered STAC fragment and everything it could not hold.""" + + model: str + stac: str + stac_path: PurePosixPath + gaps: tuple[TableColumnsGap, ...] + + +def generate_table_columns_documents( + model_specs: Sequence[ModelSpec], +) -> list[TableColumnsOutput]: + """Render one STAC fragment per spec, plus the gap log.""" + outputs: list[TableColumnsOutput] = [] + for spec in model_specs: + rendered = render_table_columns(spec) + stem = to_snake_case(spec.name) + # A bare properties fragment rather than a whole Item: the extension + # fields are what this target emits, and an Item would need id, geometry, + # bbox, datetime and links, none of which come from a schema. + extensions = [TABLE_EXTENSION_URI] + if rendered.uses_vector_extension: + # A `vector:` field without the Vector extension declared here + # would not validate against either extension's schema. Every + # feature type currently emits one, so this is unconditional in + # practice; it is a condition because a model with no geometry + # column, or a geometry column carrying no `GeometryTypeConstraint`, + # must not declare an extension it never uses. + extensions.append(VECTOR_EXTENSION_URI) + fragment = { + "stac_extensions": extensions, + "properties": rendered.stac_fields(), + } + outputs.append( + TableColumnsOutput( + model=spec.name, + # ensure_ascii=False: RFC 8259 mandates UTF-8 for interchange, + # and these fragments carry model prose -- an escaped em-dash + # parses the same and reads as a defect. + stac=json.dumps(fragment, indent=2, ensure_ascii=False) + "\n", + stac_path=PurePosixPath(f"{stem}.json"), + gaps=rendered.gaps, + ) + ) + return outputs diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/renderer.py b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/renderer.py new file mode 100644 index 000000000..b9af8e906 --- /dev/null +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/stac_table_columns/renderer.py @@ -0,0 +1,779 @@ +"""Render a model spec to the STAC Table extension's `table:columns`. + +The Column Object is `name` (required), `type`, `description`, plus what v1.3.0 +permits from STAC common metadata (`data_type`, `unit`) and from the Vector +extension (`vector:geometry_types`). Everything else the IR carries -- +constraints, enum vocabularies, named types, defaults, optionality, nested +field descriptions -- has nowhere to go, so this renderer's real output is the +gap log beside the columns. + +Four decisions, none of which anything downstream can check: + +1. **The type dialect.** `type` is an unconstrained string, and real catalogs + carry whatever their generating engine's stringifier produced, in more than + one dialect and inconsistent case. This renderer emits Arrow, because Arrow + is what Overture distributes: the release is GeoParquet, and these are the + names an Arrow reader reports opening it. Both the vocabulary and the + composite grammar come from `pyarrow`'s own stringifier, and they match + published catalogs, which name columns `int64`, `string`, `double`, + `binary` and `struct<...>`. Struct member order comes from the released + files rather than from a model's declaration order, since the type + describes the physical column. + + `type` and `data_type` therefore disagree on floats. `data_type` (decision + 3) is bound to STAC common metadata's fixed vocabulary, which names them + `float32`/`float64`; `type` is bound to Arrow, which names the same two + `float`/`double`. Harmonising them would make one field lie about the + vocabulary it claims to speak. + +2. **Which walk supplies the union.** Walking only concrete arms misses fields + that stay un-narrowed in the merged list; walking only the merged list drops + every non-first arm's contribution at a duplicated name. Neither loss can + raise, because the arms stringify identically. This renderer walks both, + emits the merged list -- what a columnar sink stores -- and logs a + `flatten-collision` for every name where the two disagree. + +3. **`data_type` is Overture's own base-type name, not a translation.** STAC + common metadata's data-type vocabulary (int8/16/32/64, uint8/16/32/64, + float32/64, `other`) already matches Overture's numeric base-type names, so + this is an allowlist against that vocabulary rather than a lookup table. + Every column gets one -- `other` is a member of the vocabulary, not an + omission -- so most columns resolve to it and only the numerics carry + signal. + +4. **`vector:geometry_types` reads the schema's `GeometryTypeConstraint`, not + data.** The Vector extension defines the field as the geometry types present + in the dataset, which by the `table:row_count` precedent this renderer would + refuse. `GeometryTypeConstraint` is validated on every row, so it bounds + what a column can contain rather than reporting what it does. A column with + no such constraint gets neither the field nor a claim about it, and is + logged as a gap. + + `unit` is not emitted, because Overture Schema has no consistent way to + encode one. Where a unit varies per row it is a field (`Speed` carries a + `SpeedUnit`); where it is fixed it appears only in the description prose. + Neither is a column-level fact, and this renderer does not parse prose. +""" + +from __future__ import annotations + +import builtins +import datetime as _dt +from dataclasses import dataclass, field +from enum import Enum +from typing import Any + +from overture.schema.system.geometric import GeometryTypeConstraint + +from ..extraction.enum_extraction import extract_enum +from ..extraction.field import ( + AnyScalar, + ArrayOf, + FieldShape, + LiteralScalar, + MapOf, + ModelRef, + NewTypeShape, + Primitive, + UnionRef, +) +from ..extraction.field_walk import enum_source +from ..extraction.specs import NO_DEFAULT, FieldSpec, ModelSpec, UnionSpec +from .exceptions import Kind, TableColumnsGap, TableColumnsUnrepresentable + +__all__ = [ + "TABLE_EXTENSION_URI", + "VECTOR_EXTENSION_URI", + "TableColumnsDocument", + "render_table_columns", +] + +TABLE_EXTENSION_URI = "https://stac-extensions.github.io/table/v1.3.0/schema.json" + +# Recommended (not required) by the Table extension's README for the column +# `vector:geometry_types` is emitted on -- see decision 4 in the module +# docstring. Declared here, not just used, because a field from this +# extension in `table:columns` without the extension in `stac_extensions` +# would not validate; `pipeline.py` adds it to that list exactly when this +# renderer emits the field. +VECTOR_EXTENSION_URI = "https://stac-extensions.github.io/vector/v0.1.0/schema.json" + +# Arrow type names, exactly as `pyarrow`'s own stringifier emits them +# (`str(pa.float64())` is `"double"`, not `"float64"`). See decision 1 in the +# module docstring. The sized integers look like a no-op mapping because they +# are one: Overture's base-type names already ARE Arrow's for every sized +# integer, exactly as they already are STAC's in `_STAC_DATA_TYPES` below. +# Only the floats, strings, bytes and time types need translating. +_ARROW_TYPES: dict[str, str] = { + "int8": "int8", + "int16": "int16", + "int32": "int32", + "int64": "int64", + "uint8": "uint8", + "uint16": "uint16", + "uint32": "uint32", + "uint64": "uint64", + "float32": "float", + "float64": "double", + "str": "string", + "bool": "bool", + "bytes": "binary", + "datetime": "timestamp[us, tz=UTC]", + "date": "date32[day]", +} + +# The type string a shape with no Arrow type of its own collapses to. +_STRING_TYPE = "string" + +# Arrow has no geometry type. GeoParquet stores geometry as WKB in a plain +# binary column and carries the semantics in file-level metadata, so `binary` +# is what an Arrow reader reports for a geometry column. The erasure is logged +# rather than absorbed; see `_scalar_type`. +_GEOMETRY_TYPE = "binary" + +# STAC common metadata's `data_type` vocabulary. See decision 3 in the module +# docstring: Overture's own base-type names already ARE this vocabulary for +# every numeric primitive, so this is an allowlist, not a translation table. +_STAC_DATA_TYPES: dict[str, str] = { + "int8": "int8", + "int16": "int16", + "int32": "int32", + "int64": "int64", + "uint8": "uint8", + "uint16": "uint16", + "uint32": "uint32", + "uint64": "uint64", + "float32": "float32", + "float64": "float64", +} +_OTHER_DATA_TYPE = "other" + +# BBox is a plain class the codegen cannot walk, so the physical layout keeps a +# shared struct constant for it. Member order is the `BBox` class's own +# declaration order. +# +# THAT DOES NOT MATCH THE ORDER OVERTURE SHIPS, which is `xmin, xmax, ymin, +# ymax` -- what this repo's PySpark `BBOX_STRUCT` declares and what a reader +# reports for a released file. The discrepancy is accepted rather than encoded: +# this renderer derives types from models, and a constant matching one +# publisher's file layout would be a fact about that publisher rather than +# about the schema. A consumer reading member order off this field will be +# wrong about it. +_BBOX_TYPE = "struct" + +# Pydantic types with no Arrow type of their own. Keyed by NAME rather than by +# `issubclass`: in Pydantic v2 neither `HttpUrl` nor `EmailStr` is a `str` +# subclass, so the builtin fallback below never reaches them. +_PYDANTIC_AS_STRING: frozenset[str] = frozenset( + {"HttpUrl", "AnyUrl", "AnyHttpUrl", "EmailStr", "UUID"} +) + +# Fallback keyed on the underlying Python type, for constrained-string NewTypes +# the registry has no name entry for. `bool` precedes `int` because it is a +# subclass. A bare Python `int` widens to `int64`, Arrow's own default for one. +_BUILTIN_TYPES: tuple[tuple[type, str], ...] = ( + (bool, "bool"), + (_dt.datetime, "timestamp[us, tz=UTC]"), + (_dt.date, "date32[day]"), + (str, _STRING_TYPE), + (bytes, "binary"), + (int, "int64"), + (float, "double"), +) + + +@dataclass +class Column: + """One emitted Column Object, plus what the emitter knows and cannot say.""" + + name: str + type: str + description: str | None + # Retained for the gap log; never emitted into `table:columns`, which + # has no field for any of it. + base_type: str | None + # The underlying Python type, kept so a downstream target can resolve + # through the same fallback chain the Arrow type does. + python_type: builtins.type | None + enum_name: str | None + is_literal: bool + enum_members: tuple[tuple[str, str | None], ...] + has_default: bool + is_required: bool + constraint_names: tuple[str, ...] + # STAC common metadata's data-type vocabulary name for `base_type`. Always + # set -- `other` is a member of the vocabulary, not an omission. + data_type: str + # GeoJSON geometry type names from `GeometryTypeConstraint`, for a + # geometry column that carries one. `None` for every other column, and + # for a geometry column with no such constraint (see decision 4). + geometry_types: tuple[str, ...] | None = None + + def as_column_object(self) -> dict[str, Any]: + """Return the Column Object: name, type, description when present, + `data_type` always, and `vector:geometry_types` when known. + + An absent description is an absent KEY, not an empty string. A silent + empty string would make an undescribed column indistinguishable from a + described one in every downstream count. + """ + obj: dict[str, Any] = {"name": self.name, "type": self.type} + if self.description is not None: + obj["description"] = self.description + obj["data_type"] = self.data_type + if self.geometry_types is not None: + obj["vector:geometry_types"] = list(self.geometry_types) + return obj + + +@dataclass +class TableColumnsDocument: + """The emitted STAC fields and the gap log.""" + + model: str + columns: list[Column] + primary_geometry: str | None + gaps: tuple[TableColumnsGap, ...] + + def stac_fields(self) -> dict[str, Any]: + """Return the `table:` fields for a STAC object. + + `table:row_count` is absent: a row count is a property of data, and + this path has only a schema. Its omission is logged as a gap, so that + it is distinguishable from having been forgotten. + """ + fields: dict[str, Any] = { + "table:columns": [c.as_column_object() for c in self.columns] + } + if self.primary_geometry is not None: + fields["table:primary_geometry"] = self.primary_geometry + return fields + + @property + def uses_vector_extension(self) -> bool: + """Whether any column carries `vector:geometry_types`. + + The pipeline needs this to decide whether `stac_extensions` must also + declare the Vector extension -- a field from that extension without + its URI declared would not validate. + """ + return any(c.geometry_types is not None for c in self.columns) + + +@dataclass +class _Ctx: + """Render state: the model name, the gap log, and the recursion guard.""" + + model: str + gaps: list[TableColumnsGap] = field(default_factory=list) + stack: list[str] = field(default_factory=list) + # Named-type sites seen anywhere under this model, for the gap log. + named_sites: list[str] = field(default_factory=list) + + def gap(self, path: str, kind: Kind, capability: str, detail: str) -> None: + self.gaps.append( + TableColumnsGap(self.model, path or "/", kind, capability, detail) + ) + + +def _dedupe(fields: list[FieldSpec], ctx: _Ctx, path: str) -> list[FieldSpec]: + """Keep one FieldSpec per name, logging every name where arms disagree. + + The physical layout keeps the first and says nothing. Keeping the first is + right -- one column stores one type -- but the silence is not: the dropped + arm's enum vocabulary and description are gone with no trace in the output. + """ + seen: dict[str, FieldSpec] = {} + for f in fields: + prior = seen.get(f.name) + if prior is None: + seen[f.name] = f + continue + kept, dropped = _vocabulary_of(prior.shape), _vocabulary_of(f.shape) + ctx.gap( + f"{path}/{f.name}", + "flatten-collision", + "union arm merge", + f"two union arms contribute a column named {f.name!r}; the first is " + f"kept ({kept or 'no vocabulary'}) and the other's meaning " + f"({dropped or 'no vocabulary'}) is discarded. Both stringify to the " + "same physical type, so the layout's compatibility check cannot " + "refuse it and nothing downstream records the loss.", + ) + return list(seen.values()) + + +def _vocabulary_of(shape: FieldShape) -> str | None: + """Name the enum backing a shape, for collision reporting.""" + inner = _unwrap(shape) + if isinstance(inner, Primitive): + src = enum_source(inner) + if src is not None: + return src.__name__ + return inner.base_type + return type(inner).__name__ + + +def _unwrap(shape: FieldShape) -> FieldShape: + """Strip NewType wrappers, returning the structural shape beneath.""" + while isinstance(shape, NewTypeShape): + shape = shape.inner + return shape + + +def _data_type(base_type: str | None) -> str: + """Map a column's base type to the STAC common-metadata `data_type` vocabulary. + + Total, unlike `type`: `other` is itself a vocabulary member, so every + column gets a value. `base_type` is `None` for anything that is not a + terminal `Primitive` (a struct, array, map, enum, or union column), which + resolves to `other` exactly like a primitive with no numeric identity. + """ + if base_type is None: + return _OTHER_DATA_TYPE + return _STAC_DATA_TYPES.get(base_type, _OTHER_DATA_TYPE) + + +def _geometry_types_of(shape: FieldShape) -> tuple[str, ...] | None: + """Return the GeoJSON geometry type names a geometry column is constrained to. + + Reads `GeometryTypeConstraint.allowed_types` -- see decision 4 in the + module docstring for why this schema constraint is a legitimate source for + `vector:geometry_types`. Returns `None` when the shape isn't a geometry + column, or is one with no such constraint attached. + """ + inner = _unwrap(shape) + if not isinstance(inner, Primitive): + return None + for source in inner.constraints: + if isinstance(source.constraint, GeometryTypeConstraint): + return tuple( + sorted(t.geo_json_type for t in source.constraint.allowed_types) + ) + return None + + +def _top_fields(spec: ModelSpec, ctx: _Ctx) -> list[FieldSpec]: + """Top-level columns, the way a columnar sink sees them.""" + if isinstance(spec, UnionSpec): + merged = _dedupe(spec.fields, ctx, "") + _report_arm_only_types(spec, merged, ctx) + return merged + return spec.fields + + +def _report_arm_only_types( + union: UnionSpec, merged: list[FieldSpec], ctx: _Ctx +) -> None: + """Log named types reachable from the arms but not from the merged list. + + The mirror of `_dedupe`'s loss, and the one the doc pipeline and both prior + renderers see while this walk does not. Reported rather than merged in: the + merged list is what the column layout is, so adding arm-only types to it + would misdescribe the emitted table. + """ + merged_names: set[str] = set() + for f in merged: + _collect_type_names(f.shape, merged_names) + arm_names: set[str] = set() + for member in union.member_specs: + for f in member.spec.fields: + _collect_type_names(f.shape, arm_names) + only_arm = sorted(arm_names - merged_names) + if only_arm: + ctx.gap( + "", + "flatten-collision", + "arm-only named types", + f"reachable from the union's concrete arms but not from the merged " + f"field list this table is built from: {', '.join(only_arm)}. A walk " + "over either side alone under-counts, in opposite directions.", + ) + + +def _collect_type_names(shape: FieldShape, acc: set[str], depth: int = 0) -> None: + """Every named type reachable from *shape*: NewTypes, records, unions, enums.""" + if depth > 40: + return + match shape: + case NewTypeShape(name=name, inner=inner): + if name: + acc.add(name) + _collect_type_names(inner, acc, depth + 1) + case ArrayOf(element=element): + _collect_type_names(element, acc, depth + 1) + case MapOf(key=key, value=value): + _collect_type_names(key, acc, depth + 1) + _collect_type_names(value, acc, depth + 1) + case ModelRef(model=model, starts_cycle=starts_cycle): + acc.add(model.name) + if not starts_cycle: + for f in model.fields: + _collect_type_names(f.shape, acc, depth + 1) + case UnionRef(union=union): + acc.add(union.name) + for f in union.fields: + _collect_type_names(f.shape, acc, depth + 1) + case Primitive() as prim: + src = enum_source(prim) + if src is not None: + acc.add(src.__name__) + + +def _scalar_type( + scalar: Primitive | LiteralScalar | AnyScalar, ctx: _Ctx, path: str +) -> str: + """Map a terminal shape to an Arrow type string, logging what that costs.""" + if isinstance(scalar, LiteralScalar): + ctx.gap( + path, + "target-gap", + "literal alternatives", + f"Literal{list(scalar.values)!r} narrows the column to " + f"{len(scalar.values)} value(s); a Column Object has no enum, const " + "or pattern keyword, so the column is a bare string.", + ) + return _STRING_TYPE + if isinstance(scalar, AnyScalar): + ctx.gap( + path, + "target-gap", + "any", + "typing.Any has no column type. Emitted as string to match what the " + "physical layout stores, which is a stringification, not a type.", + ) + return _STRING_TYPE + src = enum_source(scalar) + if src is not None: + spec = extract_enum(src) + described = sum(1 for m in spec.members if m.description is not None) + ctx.gap( + path, + "target-gap", + "enum vocabulary", + f"{src.__name__}: {len(spec.members)} members ({described} described) " + "collapse to a bare string; table:columns has no enum keyword.", + ) + return _STRING_TYPE + if scalar.base_type == "BBox": + return _BBOX_TYPE + if scalar.base_type == "Geometry": + ctx.gap( + path, + "target-dialect", + "semantic type erased", + "Arrow has no geometry type: GeoParquet stores geometry as WKB in a " + "binary column and keeps the semantics in file-level metadata, which " + "a type string cannot reach. vector:geometry_types carries the part " + "of the semantic the schema can assert; the type string carries none " + "of it.", + ) + return _GEOMETRY_TYPE + mapped = _ARROW_TYPES.get(scalar.base_type) + if mapped is not None: + return mapped + source_type = scalar.source_type + if scalar.base_type in _PYDANTIC_AS_STRING or ( + source_type is not None and source_type.__name__ in _PYDANTIC_AS_STRING + ): + ctx.gap( + path, + "target-dialect", + "semantic type erased", + f"{scalar.base_type} has no Arrow type of its own and stringifies to " + "string; the semantics live only in the name, which the column does " + "not carry.", + ) + return _STRING_TYPE + if source_type is not None: + by_name = _ARROW_TYPES.get(source_type.__name__) + if by_name is not None: + return by_name + for py_type, name in _BUILTIN_TYPES: + if isinstance(source_type, type) and issubclass(source_type, py_type): + ctx.gap( + path, + "target-dialect", + "semantic type erased", + f"{source_type.__name__} has no Arrow type of its own and " + f"stringifies to {name}; the semantics live only in the name, " + "which the column does not carry.", + ) + return name + ctx.gap( + path, + "renderer-gap", + "unmapped base type", + f"no Arrow type for base_type={scalar.base_type!r} " + f"(source_type={getattr(source_type, '__name__', None)!r}).", + ) + return _STRING_TYPE + + +def _type_string(shape: FieldShape, ctx: _Ctx, path: str) -> str: + """Build the Arrow type string for a column, recursing into nesting. + + Nesting lives in the type string, not in the column name: published + catalogs do not put dots in column names. + + Composites use Arrow's angle-bracket grammar -- `struct`, + `list`, `map` -- taken from `pyarrow`'s + stringifier. A list's child is named `item`, pyarrow's constructor form. A + round-trip through a Parquet file renames it `element`, since the Parquet + LIST logical type fixes that name, and the same round-trip stringifies a + map as `map`, appending the Parquet group name. + Those are artifacts of the file rather than type names. + """ + match shape: + case NewTypeShape(name=name, inner=inner): + if name: + ctx.named_sites.append(f"{path}:{name}") + return _type_string(inner, ctx, path) + case ArrayOf(element=element): + return f"list" + case MapOf(key=key, value=value): + key_type = _type_string(key, ctx, path + "/key") + value_type = _type_string(value, ctx, path + "/value") + return f"map<{key_type}, {value_type}>" + case ModelRef(model=model, starts_cycle=starts_cycle): + ctx.named_sites.append(f"{path}:{model.name}") + if starts_cycle or model.name in ctx.stack: + raise TableColumnsUnrepresentable( + ( + TableColumnsGap( + ctx.model, + path, + "target-gap", + "recursion", + f"{model.name} is recursive; a type string has no " + "way to name a type, so a cycle cannot terminate.", + ), + ) + ) + ctx.stack.append(model.name) + try: + return _struct_string(model.fields, ctx, path, model.name) + finally: + ctx.stack.pop() + case UnionRef(union=union): + ctx.named_sites.append(f"{path}:{union.name}") + if union.name in ctx.stack: + raise TableColumnsUnrepresentable( + ( + TableColumnsGap( + ctx.model, + path, + "target-gap", + "recursion", + f"{union.name} is recursive; see above.", + ), + ) + ) + ctx.stack.append(union.name) + try: + merged = _dedupe(union.fields, ctx, path) + _report_arm_only_types(union, merged, ctx) + ctx.gap( + path, + "target-gap", + "discriminated union", + f"{union.name} has {len(union.members)} arms; the column is a " + "single struct of their merged fields, and the discriminator " + "survives only as a string. Which arm a row belongs to is not " + "recoverable from the schema.", + ) + return _struct_string(merged, ctx, path, union.name) + finally: + ctx.stack.pop() + case Primitive() | LiteralScalar() | AnyScalar() as scalar: + return _scalar_type(scalar, ctx, path) + raise TypeError(f"Unhandled FieldShape: {shape!r}") + + +def _struct_string(fields: list[FieldSpec], ctx: _Ctx, path: str, owner: str) -> str: + """`struct`, logging the descriptions that die inside it.""" + if not fields: + ctx.gap( + path, + "renderer-gap", + "empty struct", + f"{owner} has no fields; an empty struct column cannot carry data.", + ) + return "struct<>" + parts = [] + for f in fields: + member_path = f"{path}/{f.name}" + if f.description is not None: + ctx.gap( + member_path, + "target-gap", + "nested description", + f"{owner}.{f.name} is described, and a Column Object describes " + "only top-level columns. The NAME survives inside the type " + "string; the description does not.", + ) + _log_field_surplus(f, ctx, member_path, nested=True) + parts.append(f"{f.name}: {_type_string(f.shape, ctx, member_path)}") + return f"struct<{', '.join(parts)}>" + + +def _constraint_names(shape: FieldShape, acc: list[str], depth: int = 0) -> None: + """Collect constraint class names at every layer of a shape.""" + if depth > 40: + return + for source in getattr(shape, "constraints", ()) or (): + acc.append(type(source.constraint).__name__) + match shape: + case NewTypeShape(inner=inner): + _constraint_names(inner, acc, depth + 1) + case ArrayOf(element=element): + _constraint_names(element, acc, depth + 1) + case MapOf(key=key, value=value): + _constraint_names(key, acc, depth + 1) + _constraint_names(value, acc, depth + 1) + case ModelRef(model=model, starts_cycle=starts_cycle): + if not starts_cycle: + for f in model.fields: + _constraint_names(f.shape, acc, depth + 1) + case UnionRef(union=union): + for f in union.fields: + _constraint_names(f.shape, acc, depth + 1) + + +def _log_field_surplus(f: FieldSpec, ctx: _Ctx, path: str, *, nested: bool) -> None: + """Log the per-field capabilities a Column Object has no field for.""" + where = "nested field" if nested else "column" + if f.default is not NO_DEFAULT: + ctx.gap( + path, + "target-gap", + "default", + f"{where} carries default={f.default!r}. The IR carries defaults; a " + "Column Object is name/type/description and has nowhere to put one. " + "The gap is the target's, not the IR's.", + ) + if f.is_deprecated: + ctx.gap( + path, + "target-gap", + "deprecated", + f"{where} is deprecated; there is no keyword for it.", + ) + ctx.gap( + path, + "target-gap", + "optionality", + f"{where} is {'required' if f.is_required else 'optional'}" + f"{' and nullable' if f.is_optional else ''}; a Column Object has no " + "required, nullable or optional keyword, so this is not expressible at " + "all.", + ) + names: list[str] = [] + _constraint_names(f.shape, names) + for name in names: + ctx.gap( + path, + "target-gap", + "constraint", + f"{name} has no representation: the Column Object vocabulary is " + "name, type, description.", + ) + + +def _enum_of(shape: FieldShape) -> type[Enum] | None: + inner = _unwrap(shape) + return enum_source(inner) if isinstance(inner, Primitive) else None + + +def render_table_columns( + spec: ModelSpec, *, strict: bool = False +) -> TableColumnsDocument: + """Render *spec* to STAC `table:` fields, with a gap log beside them. + + `strict=True` raises rather than returning a non-empty log. A `UnionSpec` + root emits normally: a flat column list is exactly + what a columnar sink stores for a union, and the IR already carries the + merged field list that sink needs. + """ + ctx = _Ctx(model=spec.name) + columns: list[Column] = [] + for f in _top_fields(spec, ctx): + path = f"/{f.name}" + _log_field_surplus(f, ctx, path, nested=False) + inner = _unwrap(f.shape) + enum_cls = _enum_of(f.shape) + members: tuple[tuple[str, str | None], ...] = () + if enum_cls is not None: + members = tuple( + (m.value, m.description) for m in extract_enum(enum_cls).members + ) + constraints: list[str] = [] + _constraint_names(f.shape, constraints) + base_type = inner.base_type if isinstance(inner, Primitive) else None + geometry_types = _geometry_types_of(f.shape) + if base_type == "Geometry" and geometry_types is None: + ctx.gap( + path, + "ir-gap", + "geometry types", + "geometry column has no GeometryTypeConstraint; " + "vector:geometry_types would have to guess which types can " + "appear, so it is omitted rather than claiming all seven.", + ) + columns.append( + Column( + name=f.name, + type=_type_string(f.shape, ctx, path), + description=f.description, + base_type=base_type, + python_type=( + inner.source_type + if isinstance(inner, Primitive) + and isinstance(inner.source_type, type) + else None + ), + enum_name=enum_cls.__name__ if enum_cls is not None else None, + is_literal=isinstance(inner, LiteralScalar), + enum_members=members, + has_default=f.default is not NO_DEFAULT, + is_required=f.is_required, + constraint_names=tuple(constraints), + data_type=_data_type(base_type), + geometry_types=geometry_types, + ) + ) + + primary = _primary_geometry(columns, ctx) + gaps = tuple(ctx.gaps) + if strict and gaps: + raise TableColumnsUnrepresentable(gaps) + return TableColumnsDocument( + model=spec.name, + columns=columns, + primary_geometry=primary, + gaps=gaps, + ) + + +def _primary_geometry(columns: list[Column], ctx: _Ctx) -> str | None: + """Derive `table:primary_geometry`, refusing to guess when it is ambiguous.""" + geometry = [c.name for c in columns if c.base_type == "Geometry"] + if len(geometry) == 1: + return geometry[0] + if not geometry: + ctx.gap( + "", + "target-gap", + "primary geometry", + "no geometry column; table:primary_geometry is omitted.", + ) + return None + ctx.gap( + "", + "ir-gap", + "primary geometry", + f"{len(geometry)} geometry columns ({', '.join(geometry)}) and nothing in " + "the IR says which is primary. Omitted rather than guessed -- picking the " + "first would invent a fact about the data.", + ) + return None diff --git a/packages/overture-schema-codegen/tests/test_cli.py b/packages/overture-schema-codegen/tests/test_cli.py index dd90c6880..fee01391e 100644 --- a/packages/overture-schema-codegen/tests/test_cli.py +++ b/packages/overture-schema-codegen/tests/test_cli.py @@ -1,7 +1,10 @@ """Tests for CLI entrypoint.""" import json +import os import re +import subprocess +import sys from pathlib import Path import pytest @@ -541,3 +544,53 @@ def test_used_by_sections_appear_in_markdown( break assert has_used_by, "No 'Used By' sections found in any generated markdown" + + +_NON_ASCII = "em-dash — café" + +_WRITE_PROBE = """ +import json, locale, sys, tempfile +from pathlib import Path, PurePosixPath +from overture.schema.codegen.cli import _write_output + +out = Path(tempfile.mkdtemp()) +_write_output({text!r}, out, PurePosixPath("probe.txt")) +print(json.dumps({{ + "encoding": locale.getpreferredencoding(False), + "text": (out / "probe.txt").read_bytes().decode("utf-8"), +}})) +""" + + +def test_generated_files_are_utf8_under_a_non_utf8_locale() -> None: + """Written artifacts are UTF-8 regardless of the machine's locale. + + `Path.write_text` with no `encoding` uses the locale's, which through + Python 3.14 is whatever the environment says -- so on an ASCII locale + every description carrying an em-dash raised `UnicodeEncodeError` and no + file was written at all. Generated Python is decoded as UTF-8 by PEP + 3120 and JSON is UTF-8 by RFC 8259, so the locale is never the right + answer here. + + The subprocess is the point: the encoding is bound at interpreter start, + so an in-process check would read this machine's UTF-8 and pass either + way. + """ + proc = subprocess.run( + [sys.executable, "-c", _WRITE_PROBE.format(text=_NON_ASCII)], + capture_output=True, + text=True, + env={ + **os.environ, + "LC_ALL": "C", + "LANG": "C", + "PYTHONUTF8": "0", + "PYTHONCOERCECLOCALE": "0", + }, + ) + + assert proc.returncode == 0, proc.stderr + payload = json.loads(proc.stdout) + if payload["encoding"].lower().replace("-", "") in {"utf8", "utf"}: + pytest.skip(f"locale could not be forced off UTF-8: {payload['encoding']}") + assert payload["text"] == _NON_ASCII diff --git a/packages/overture-schema-codegen/tests/test_model_extraction.py b/packages/overture-schema-codegen/tests/test_model_extraction.py index e8b3fd2c7..80c2eac29 100644 --- a/packages/overture-schema-codegen/tests/test_model_extraction.py +++ b/packages/overture-schema-codegen/tests/test_model_extraction.py @@ -15,6 +15,7 @@ from overture.schema.codegen.extraction.field_walk import terminal_of from overture.schema.codegen.extraction.length_constraints import ArrayMinLen from overture.schema.codegen.extraction.model_extraction import extract_model +from overture.schema.codegen.extraction.specs import NO_DEFAULT from overture.schema.common.scoping.vehicle import VehicleSelector @@ -179,3 +180,93 @@ class M(BaseModel): assert isinstance(items_field.shape, ArrayOf) constraints = [cs.constraint for cs in items_field.shape.constraints] assert ArrayMinLen(min_length=2) in constraints + + +def test_declared_default_reaches_field_spec() -> None: + """A field's declared default lands on `FieldSpec.default`.""" + + class M(BaseModel): + level: int = 3 + + spec = extract_model(M) + field = next(f for f in spec.fields if f.name == "level") + + assert field.default == 3 + + +def test_field_without_default_is_no_default_not_none() -> None: + """No default extracts as `NO_DEFAULT`, which `None` cannot stand in for. + + `default=None` is a legal declaration, so collapsing the two would make a + field that defaults to null indistinguishable from one that has no default. + """ + + class M(BaseModel): + required: int + nullable_default: int | None = None + + spec = extract_model(M) + fields = {f.name: f for f in spec.fields} + + assert fields["required"].default is NO_DEFAULT + assert fields["nullable_default"].default is None + + +def test_default_factory_does_not_become_a_default() -> None: + """A `default_factory` extracts as `NO_DEFAULT`. + + Calling it here would freeze one sample of a value whose whole point is to + be produced per instance, so extraction records that there is no static + default rather than inventing one. + """ + + class M(BaseModel): + tags: list[str] = Field(default_factory=list) + + spec = extract_model(M) + field = next(f for f in spec.fields if f.name == "tags") + + assert field.default is NO_DEFAULT + + +def test_deprecated_message_sets_flag_and_keeps_prose() -> None: + """A string `deprecated` sets the flag and preserves the message beside it.""" + + class M(BaseModel): + old: Annotated[str, Field(deprecated="use `new`")] = "x" + + spec = extract_model(M) + field = next(f for f in spec.fields if f.name == "old") + + assert field.is_deprecated is True + assert field.deprecation_message == "use `new`" + + +def test_bare_deprecated_true_carries_no_message() -> None: + """A bare `deprecated=True` sets the flag with no prose to carry.""" + + class M(BaseModel): + old: Annotated[str, Field(deprecated=True)] = "x" + + spec = extract_model(M) + field = next(f for f in spec.fields if f.name == "old") + + assert field.is_deprecated is True + assert field.deprecation_message is None + + +def test_undeclared_deprecation_is_false() -> None: + """A field that says nothing about deprecation is not deprecated. + + The did-happen half of the pair above: without this, a reader cannot tell a + working flag from one that is wired to a slot nothing ever sets. + """ + + class M(BaseModel): + current: str + + spec = extract_model(M) + field = next(f for f in spec.fields if f.name == "current") + + assert field.is_deprecated is False + assert field.deprecation_message is None diff --git a/packages/overture-schema-codegen/tests/test_stac_table_columns.py b/packages/overture-schema-codegen/tests/test_stac_table_columns.py new file mode 100644 index 000000000..bc8fbf4fd --- /dev/null +++ b/packages/overture-schema-codegen/tests/test_stac_table_columns.py @@ -0,0 +1,431 @@ +"""Tests for the STAC `table:columns` renderer. + +A Column Object is `name`, `type`, `description`, so most of what this renderer +knows it cannot say. These cover the decisions that survive that flattening: how +an absent description is rendered, what happens at a union collision, and the two +cases where guessing is worse than refusing. +""" + +import json +from typing import Annotated, Literal + +import pytest +from pydantic import BaseModel, Field + +from overture.schema.codegen.extraction.model_extraction import extract_model +from overture.schema.codegen.stac_table_columns.exceptions import ( + TableColumnsUnrepresentable, +) +from overture.schema.codegen.stac_table_columns.pipeline import ( + generate_table_columns_documents, +) +from overture.schema.codegen.stac_table_columns.renderer import ( + VECTOR_EXTENSION_URI, + render_table_columns, +) +from overture.schema.system.geometric import ( + BBox, + Geometry, + GeometryType, + GeometryTypeConstraint, +) +from overture.schema.system.numeric import float32, float64, int32, int64 + + +def _capabilities(doc: object) -> set[str]: + return {g.capability for g in doc.gaps} # type: ignore[attr-defined] + + +def _columns(model: type[BaseModel]) -> dict[str, dict[str, object]]: + doc = render_table_columns(extract_model(model)) + return {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + +def test_undescribed_column_omits_the_key_entirely() -> None: + """A column with no description omits `description`, never emits `""`. + + The two are not interchangeable downstream: an omitted key leaves the + undescribed set as something a consumer can query for, while an empty string + is a description that happens to say nothing, and no count can tell it from + one that was authored badly. + """ + + class M(BaseModel): + described: Annotated[str, Field(description="what it is")] + bare: str + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert columns["described"]["description"] == "what it is" + assert "description" not in columns["bare"] + + +def test_primary_geometry_is_emitted_when_exactly_one_geometry_exists() -> None: + """One geometry column resolves `table:primary_geometry` unambiguously.""" + + class M(BaseModel): + geometry: Annotated[Geometry, Field(description="the shape")] + name: str + + doc = render_table_columns(extract_model(M)) + + assert doc.stac_fields()["table:primary_geometry"] == "geometry" + + +def test_two_geometries_omit_primary_rather_than_guessing() -> None: + """Nothing in the IR says which of two geometry columns is primary. + + Picking the first would invent a fact about the data, so the key is omitted + and the ambiguity is logged as an IR gap. This is the direction that + discriminates: the shipped models all carry exactly one geometry column, so + without a synthetic pair the branch is never exercised. + """ + + class M(BaseModel): + footprint: Annotated[Geometry, Field(description="one")] + centroid: Annotated[Geometry, Field(description="another")] + + doc = render_table_columns(extract_model(M)) + + assert "table:primary_geometry" not in doc.stac_fields() + assert any( + g.kind == "ir-gap" and g.capability == "primary geometry" for g in doc.gaps + ) + + +def test_row_count_is_never_emitted() -> None: + """`table:row_count` is a property of data; this path has only a schema. + + Emitting a placeholder would be a fabricated measurement, so the key is + absent by construction rather than by omission. + """ + + class M(BaseModel): + id: str + + assert "table:row_count" not in render_table_columns(extract_model(M)).stac_fields() + + +def test_union_name_collision_is_logged_not_silently_dropped() -> None: + """Two arms contributing one column name emit a `flatten-collision`. + + Both arms stringify to the same physical type, so nothing downstream can + refuse the merge and nothing else records that an arm's meaning was lost. + This gap is the only trace. + """ + + class Base(BaseModel): + pass + + class Left(Base): + kind: Literal["left"] + value: Annotated[str, Field(description="a left value")] + + class Right(Base): + kind: Literal["right"] + value: Annotated[str, Field(description="a right value")] + + class Holder(BaseModel): + item: Annotated[Left | Right, Field(discriminator="kind")] + + doc = render_table_columns(extract_model(Holder)) + + assert any(g.kind == "flatten-collision" for g in doc.gaps) + + +def test_optionality_is_reported_as_lost_for_every_column() -> None: + """A Column Object has no required, nullable or optional keyword. + + Unlike a constraint, this one cannot be expressed at all, so it is logged + for every column rather than only for the ones that declare something. + """ + + class M(BaseModel): + required: str + optional: str | None = None + + doc = render_table_columns(extract_model(M)) + + assert "optionality" in _capabilities(doc) + + +def test_strict_mode_raises_rather_than_returning_a_gap_log() -> None: + """`strict=True` turns the log into a refusal. + + The default is to emit and report, because a flattening target that refused + on every loss would never emit at all; strict is for a caller that wants the + losses to be fatal. + """ + + class M(BaseModel): + anything: str | None = None + + with pytest.raises(TableColumnsUnrepresentable): + render_table_columns(extract_model(M), strict=True) + + +def test_scalar_columns_use_arrow_names() -> None: + """`type` uses Arrow's names, the ones a reader of the released GeoParquet + reports -- `double` and `float`, not `float64`/`float32`, and `string`, + `bool`, `binary` rather than DuckDB's `varchar`, `boolean`, `blob`. Every + name here came from `pyarrow`'s own stringifier.""" + + class M(BaseModel): + big: float64 + small: float32 + label: str + flag: bool + raw: bytes + count: int64 + + columns = _columns(M) + + assert columns["big"]["type"] == "double" + assert columns["small"]["type"] == "float" + assert columns["label"]["type"] == "string" + assert columns["flag"]["type"] == "bool" + assert columns["raw"]["type"] == "binary" + assert columns["count"]["type"] == "int64" + + +def test_composite_columns_use_arrow_angle_bracket_grammar() -> None: + """Structs, lists and maps are the part most likely to drift. + + DuckDB wrote these `struct(name type, ...)`, `type[]` and + `map(varchar, varchar)`; Arrow uses angle brackets and names a + list's child `item`. Pinning the whole nested string is the point -- a + per-scalar check would pass against either grammar. + """ + + class Inner(BaseModel): + depth: float64 + tag: str + + class M(BaseModel): + nested: Inner + labels: list[str] + matrix: list[list[int32]] + lookup: dict[str, str] + + columns = _columns(M) + + assert columns["nested"]["type"] == "struct" + assert columns["labels"]["type"] == "list" + assert columns["matrix"]["type"] == "list>" + assert columns["lookup"]["type"] == "map" + + +def test_geometry_is_binary_and_the_erasure_is_logged() -> None: + """Arrow has no geometry type; GeoParquet stores WKB in a binary column. + + `binary` is what an Arrow reader reports for Overture's own released + geometry column, so it is what this renderer emits. The semantic the + DuckDB name carried in the type string is gone, and the gap log is + what records that -- `vector:geometry_types` carries only the part the + schema can assert. + """ + + class M(BaseModel): + geometry: Annotated[Geometry, GeometryTypeConstraint(GeometryType.POINT)] + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert columns["geometry"]["type"] == "binary" + assert columns["geometry"]["vector:geometry_types"] == ["Point"] + assert any(g.kind == "target-dialect" and "geometry" in g.detail for g in doc.gaps) + + +def test_type_and_data_type_differ_for_floats_on_purpose() -> None: + """The two fields speak different vocabularies, and neither is bent. + + `data_type` is bound to STAC common metadata, which names floats + `float32`/`float64`; `type` is bound to Arrow, which names the same two + `float`/`double`. + """ + + class M(BaseModel): + small: float32 + big: float64 + + columns = _columns(M) + + assert columns["small"]["type"] == "float" + assert columns["small"]["data_type"] == "float32" + assert columns["big"]["type"] == "double" + assert columns["big"]["data_type"] == "float64" + + +def test_bbox_struct_members_follow_the_model_not_a_publisher() -> None: + """The bbox struct's member order is the `BBox` class's declaration order. + + It does not match the order Overture ships (`xmin, xmax, ymin, ymax`, what + the PySpark `BBOX_STRUCT` declares). That discrepancy is accepted: this + renderer derives types from models, and matching one publisher's file + layout would encode a fact about that publisher rather than the schema. + """ + + class M(BaseModel): + bbox: BBox + + assert ( + _columns(M)["bbox"]["type"] + == "struct" + ) + + +def test_data_type_maps_a_numeric_column_to_the_stac_vocabulary() -> None: + """A numeric base type resolves to STAC common metadata's own name. + + Overture's `int32`/`float64` names already ARE the vocabulary's names, so + this is a pass-through, not a translation. + """ + + class M(BaseModel): + count: int32 + ratio: float64 + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert columns["count"]["data_type"] == "int32" + assert columns["ratio"]["data_type"] == "float64" + + +def test_data_type_falls_back_to_other_for_a_non_numeric_column() -> None: + """A string, a struct, and everything else with no numeric identity get + `other` -- a member of the STAC vocabulary, not an omission.""" + + class Nested(BaseModel): + value: str + + class M(BaseModel): + label: str + nested: Nested + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert columns["label"]["data_type"] == "other" + assert columns["nested"]["data_type"] == "other" + + +def test_geometry_type_constraint_becomes_vector_geometry_types() -> None: + """A single allowed geometry type becomes a one-element `vector:geometry_types`. + + v1.3.0's README defines no `geometry_type` on the Column Object at all, + only `vector:geometry_types` from the separate Vector extension (see the + renderer module docstring). + """ + + class M(BaseModel): + geometry: Annotated[ + Geometry, + GeometryTypeConstraint(GeometryType.POINT), + Field(description="the shape"), + ] + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert columns["geometry"]["vector:geometry_types"] == ["Point"] + + +def test_multiple_allowed_geometry_types_are_sorted_geojson_names() -> None: + """Several allowed types become a sorted list of GeoJSON (not snake_case) names.""" + + class M(BaseModel): + geometry: Annotated[ + Geometry, + GeometryTypeConstraint(GeometryType.MULTI_POLYGON, GeometryType.POLYGON), + ] + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert columns["geometry"]["vector:geometry_types"] == ["MultiPolygon", "Polygon"] + + +def test_unconstrained_geometry_omits_vector_geometry_types_and_logs_a_gap() -> None: + """No `GeometryTypeConstraint` means no claim about which types can appear. + + Emitting all seven would assert something the schema never said; omitting + silently would look identical to forgetting the field. The gap log is the + only record. + """ + + class M(BaseModel): + geometry: Geometry + + doc = render_table_columns(extract_model(M)) + columns = {c["name"]: c for c in doc.stac_fields()["table:columns"]} + + assert "vector:geometry_types" not in columns["geometry"] + assert any( + g.kind == "ir-gap" and g.capability == "geometry types" for g in doc.gaps + ) + + +def test_pipeline_declares_the_vector_extension_only_when_used() -> None: + """`stac_extensions` gains the Vector extension exactly when a column needs it. + + A `vector:` field with its extension undeclared would not validate; a + model with no geometry constraint should not carry a URI for a field it + never emits. + """ + + class Constrained(BaseModel): + geometry: Annotated[Geometry, GeometryTypeConstraint(GeometryType.POINT)] + + class Unconstrained(BaseModel): + geometry: Geometry + + [with_vector] = generate_table_columns_documents([extract_model(Constrained)]) + [without_vector] = generate_table_columns_documents([extract_model(Unconstrained)]) + + assert VECTOR_EXTENSION_URI in json.loads(with_vector.stac)["stac_extensions"] + assert ( + VECTOR_EXTENSION_URI not in json.loads(without_vector.stac)["stac_extensions"] + ) + + +def test_gap_log_is_non_empty_for_a_real_model() -> None: + """The did-happen control for every assertion above. + + Each test that asserts a *particular* capability was logged would pass just + as well against a renderer whose log was accidentally empty of everything + else. This one fails if the log stops being populated at all. + """ + + class M(BaseModel): + bounded: Annotated[int, Field(ge=0, le=10)] + + doc = render_table_columns(extract_model(M)) + + assert doc.gaps + assert "constraint" in _capabilities(doc) + + +def test_fragment_carries_non_ascii_text_literally() -> None: + """Descriptions serialize as UTF-8, not as `\\uXXXX` escapes. + + Both spellings parse to the same string, so nothing downstream depends on + this -- but the fragment is read by people as often as by parsers, and an + escaped em-dash reads as a defect in a document whose whole content is + prose lifted from the models. + """ + + class M(BaseModel): + field: str = Field(description="An em-dash — and a café.") + + [doc] = generate_table_columns_documents([extract_model(M)]) + [column] = [ + c + for c in json.loads(doc.stac)["properties"]["table:columns"] + if c["name"] == "field" + ] + + assert "An em-dash — and a café." in doc.stac + assert "\\u" not in doc.stac + assert column["description"] == "An em-dash — and a café."