Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Added a `stac-table-columns` codegen format rendering the STAC Table extension's (v1.3.0) `table:columns` from the models, with a gap log recording what a flat column list cannot hold. Column types use Arrow's names, the vocabulary a reader of Overture's released GeoParquet reports. Each Column Object also carries STAC common metadata's `data_type` and, for the geometry column, the Vector extension's `vector:geometry_types` (declared in `stac_extensions` alongside the Table extension when emitted). Field defaults and deprecation now carry through extraction so the renderer can report both.
1 change: 1 addition & 0 deletions packages/overture-schema-codegen/changelog.d/724.bugfix.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Fixed `generate --output-dir` raising `UnicodeEncodeError` on a non-UTF-8 locale: generated files are now written as UTF-8 rather than in the locale's encoding, which through Python 3.14 could be ASCII and refused the first em-dash in a model description.
Original file line number Diff line number Diff line change
Expand Up @@ -23,12 +23,13 @@
from .markdown.pipeline import generate_markdown_pages
from .pyspark.pipeline import generate_pyspark_modules
from .spec_discovery import extract_alias_spec, extract_model_spec
from .stac_table_columns.pipeline import generate_table_columns_documents

log = logging.getLogger(__name__)

__all__ = ["cli"]

_OUTPUT_FORMATS = ("markdown", "pyspark")
_OUTPUT_FORMATS = ("markdown", "pyspark", "stac-table-columns")

_FEATURE_FRONTMATTER = "---\nsidebar_position: 1\n---\n\n"

Expand All @@ -42,7 +43,10 @@ def _write_output(
if output_dir:
file_path = output_dir / output_path
file_path.parent.mkdir(parents=True, exist_ok=True)
file_path.write_text(content)
# UTF-8, not the locale's encoding: generated Python is decoded as
# UTF-8 (PEP 3120) and JSON is UTF-8 by RFC 8259, and on an ASCII
# locale the default raises on the first em-dash in a description.
file_path.write_text(content, encoding="utf-8")
else:
click.echo(content)
click.echo() # separate entries with a blank line in stdout mode
Expand Down Expand Up @@ -121,6 +125,8 @@ def generate(

if output_format == "pyspark":
_generate_pyspark(model_specs, output_dir, test_output_dir)
elif output_format == "stac-table-columns":
_generate_table_columns(model_specs, output_dir)
else:
# RootModel entry points yield no ModelSpec, so they document as
# named aliases -- reachable no other way, since a RootModel field
Expand Down Expand Up @@ -176,6 +182,22 @@ def _generate_pyspark(
_write_output(mod.content, test_output_dir, mod.path)


def _generate_table_columns(
model_specs: list[ModelSpec],
output_dir: Path | None,
) -> None:
"""Generate one STAC `table:` properties fragment per model.

Every model emits, including the discriminated-union root: a flat column
list is what a columnar sink stores for a union. The gap count is logged
per model because it, not the fragment, is what a flattening target has to
be judged on.
"""
for doc in generate_table_columns_documents(model_specs):
_write_output(doc.stac, output_dir, doc.stac_path)
log.info("%s: %d gaps", doc.model, len(doc.gaps))


def _ancestor_dirs(paths: set[PurePosixPath]) -> set[PurePosixPath]:
"""Collect all ancestor directories for a set of file paths."""
dirs: set[PurePosixPath] = set()
Expand Down Expand Up @@ -216,7 +238,7 @@ def _write_category_files(

file_path = output_dir / dir_path / "_category_.json"
file_path.parent.mkdir(parents=True, exist_ok=True)
file_path.write_text(json.dumps(category, indent=2) + "\n")
file_path.write_text(json.dumps(category, indent=2) + "\n", encoding="utf-8")


def main() -> None:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
from collections.abc import Mapping

from pydantic import BaseModel
from pydantic.experimental.missing_sentinel import MISSING
from pydantic.fields import FieldInfo
from pydantic_core import PydanticUndefined

Expand All @@ -15,7 +16,7 @@
ModelRef,
UnionRef,
)
from .specs import FieldSpec, RecordSpec, is_model_class
from .specs import NO_DEFAULT, FieldSpec, RecordSpec, is_model_class
from .type_analyzer import (
ModelResolver,
UnionResolver,
Expand Down Expand Up @@ -46,6 +47,44 @@ def resolve_field_alias(field_name: str, field_info: FieldInfo) -> str:
return field_name


def _field_default(field_info: FieldInfo) -> object:
"""Return the field's declared default, or `NO_DEFAULT`.

Typed `object` rather than `Any`: a default really is an arbitrary value, but
`object` says so without switching off type checking at every call site.

`default_factory` is deliberately not consulted: a factory is behavior, and
calling it here would capture one sample of a value meant to be produced per
instance. A field with only a factory extracts as `NO_DEFAULT` -- which is
accurate about what a static schema can carry, not a silent drop.

Two sentinels mean "no default", not one. `PydanticUndefined` is Pydantic's,
and `MISSING` is the one `Omitable[T]` installs (`Field(default=MISSING)`) to
get JSON Schema omissibility instead of Pydantic nullability -- see
`overture.schema.system.optionality`. `MISSING` is machinery for "this key may
be absent", never a value anyone declared, so extracting it as a default makes
every `Omitable` field claim a default it does not have.
"""
if field_info.default is PydanticUndefined or field_info.default is MISSING:
return NO_DEFAULT
return field_info.default


def _field_deprecation(field_info: FieldInfo) -> tuple[bool, str | None]:
"""Return `(is_deprecated, message)` from Pydantic's `deprecated`.

Pydantic admits `True`, a string message, or a `warnings.deprecated`
instance. All three mean deprecated; only the string carries prose, and it is
kept beside the flag rather than folded into it.
"""
deprecated = field_info.deprecated
if deprecated is None or deprecated is False:
return False, None
if isinstance(deprecated, str):
return True, deprecated
return True, None


def _is_field_required(field_info: FieldInfo, is_optional: bool) -> bool:
"""Determine whether a field is required (no default and not Optional)."""
has_default = (
Expand Down Expand Up @@ -178,13 +217,17 @@ def _extract_model_recursive(
# misses those constraints. Reattach them at the topmost
# constraint-bearing layer.
shape = attach_field_metadata(shape, field_info)
is_deprecated, deprecation_message = _field_deprecation(field_info)
fields.append(
FieldSpec(
name=resolve_field_alias(field_name, field_info),
shape=shape,
description=field_info.description or ti_description,
is_required=_is_field_required(field_info, is_optional),
is_optional=is_optional,
default=_field_default(field_info),
is_deprecated=is_deprecated,
deprecation_message=deprecation_message,
)
)

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -120,6 +120,23 @@ class EnumSpec(_SourceTypeIdentityMixin):
source_type: type | None = None


class _NoDefault:
"""The absence of a default, distinct from a default of `None`.

`None` is a legal default value, so `default=None` cannot mean "no default"
without collapsing the two. A dedicated sentinel keeps them apart, and reads
at a use site (`spec.default is NO_DEFAULT`) as the question being asked.
"""

__slots__ = ()

def __repr__(self) -> str: # pragma: no cover - debugging aid
return "NO_DEFAULT"


NO_DEFAULT = _NoDefault()


@dataclass
class FieldSpec:
"""Specification for a model field: header metadata plus structural shape.
Expand All @@ -134,6 +151,21 @@ class FieldSpec:
description: str | None = None
is_required: bool = True
is_optional: bool = False
# The field's declared default, or `NO_DEFAULT` when it has none. A
# `default_factory` does NOT land here: a factory is behavior, and calling it
# at extraction time would freeze one sample of a value whose whole point is
# to be produced per instance.
default: Any = NO_DEFAULT
# Pydantic's `deprecated`, normalized to a flag. Pydantic admits `True` or a
# deprecation *message*; a message narrows to `True` here, because every
# target this IR feeds has a boolean keyword and none has anywhere to put the
# prose. The renderer logs that narrowing where it applies -- the IR's job is
# to carry the fact, not to decide what a target does about it.
is_deprecated: bool = False
# The message form of `deprecated`, kept beside the flag so a target that
# grows somewhere to put it does not have to re-extract. `None` when the
# field declared a bare `True`, or nothing at all.
deprecation_message: str | None = None


@dataclass
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
"""STAC Table extension (`table:columns`) rendering.

A flattening target: a Column Object is `name`, `type`, `description` and
nothing else, so this renderer's output is as much a record of what a flat
column list cannot hold as it is the list itself.
"""
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
"""Typed record of everything the `table:columns` emit could not carry.

`flatten-collision` is the kind only a flattening target produces: two union
arms contributing the same column name, where a columnar sink keeps one and the
other's meaning is gone with no trace in the output. `unrepresentable-in-pydantic`
runs the other way -- the target asks for something Pydantic has no way to
declare, so the gap is a possible Overture Schema feature rather than a defect.
"""

from __future__ import annotations

from dataclasses import dataclass
from typing import Literal

__all__ = ["Kind", "TableColumnsGap", "TableColumnsUnrepresentable"]

Kind = Literal[
"ir-gap",
"target-gap",
"target-dialect",
"flatten-collision",
"unrepresentable-in-pydantic",
"renderer-gap",
]


@dataclass(frozen=True, slots=True)
class TableColumnsGap:
"""One capability that did not survive the emit.

`path` locates it in the emitted document, `capability` names what was lost
in the vocabulary of whichever side owns the loss, and `detail` carries the
mechanism. Classification is by mechanism, not by keyword.
"""

model: str
path: str
kind: Kind
capability: str
detail: str


class TableColumnsUnrepresentable(Exception):
"""Raised in strict mode, or when the renderer cannot proceed at all."""

def __init__(self, gaps: tuple[TableColumnsGap, ...]) -> None:
self.gaps = gaps
head = gaps[0]
super().__init__(
f"{head.model}{head.path}: {head.capability} ({head.kind}) -- {head.detail}"
+ (f" [+{len(gaps) - 1} more]" if len(gaps) > 1 else "")
)
Original file line number Diff line number Diff line change
@@ -0,0 +1,69 @@
"""STAC `table:columns` generation pipeline: render documents without I/O.

One artifact per model -- a STAC Item properties fragment carrying the `table:`
fields -- and the gap log beside it, which for a target this thin is as much the
deliverable as the fragment.
"""

from __future__ import annotations

import json
from collections.abc import Sequence
from dataclasses import dataclass
from pathlib import PurePosixPath

from overture.schema.system.case import to_snake_case

from ..extraction.specs import ModelSpec
from .exceptions import TableColumnsGap
from .renderer import TABLE_EXTENSION_URI, VECTOR_EXTENSION_URI, render_table_columns

__all__ = ["TableColumnsOutput", "generate_table_columns_documents"]


@dataclass(frozen=True, slots=True)
class TableColumnsOutput:
"""A rendered STAC fragment and everything it could not hold."""

model: str
stac: str
stac_path: PurePosixPath
gaps: tuple[TableColumnsGap, ...]


def generate_table_columns_documents(
model_specs: Sequence[ModelSpec],
) -> list[TableColumnsOutput]:
"""Render one STAC fragment per spec, plus the gap log."""
outputs: list[TableColumnsOutput] = []
for spec in model_specs:
rendered = render_table_columns(spec)
stem = to_snake_case(spec.name)
# A bare properties fragment rather than a whole Item: the extension
# fields are what this target emits, and an Item would need id, geometry,
# bbox, datetime and links, none of which come from a schema.
extensions = [TABLE_EXTENSION_URI]
if rendered.uses_vector_extension:
# A `vector:` field without the Vector extension declared here
# would not validate against either extension's schema. Every
# feature type currently emits one, so this is unconditional in
# practice; it is a condition because a model with no geometry
# column, or a geometry column carrying no `GeometryTypeConstraint`,
# must not declare an extension it never uses.
extensions.append(VECTOR_EXTENSION_URI)
fragment = {
"stac_extensions": extensions,
"properties": rendered.stac_fields(),
}
outputs.append(
TableColumnsOutput(
model=spec.name,
# ensure_ascii=False: RFC 8259 mandates UTF-8 for interchange,
# and these fragments carry model prose -- an escaped em-dash
# parses the same and reads as a defect.
stac=json.dumps(fragment, indent=2, ensure_ascii=False) + "\n",
stac_path=PurePosixPath(f"{stem}.json"),
gaps=rendered.gaps,
)
)
return outputs
Loading
Loading