From a42ddf5cf53ceb5a4ed2fe18e003e97874f2b70e Mon Sep 17 00:00:00 2001 From: Seth Fitzsimmons Date: Tue, 21 Jul 2026 12:09:37 -0700 Subject: [PATCH] docs: clear pydocstyle errors in codegen and pyspark `make docformat` (pydocstyle, numpy convention) flagged eight docstrings across overture-schema-codegen and overture-schema-pyspark: - D301 (raw string for backslashes): `_bare_map_side_name`, `normalize_anchor`, and `check_pattern` move to `r"""` and drop the doubled backslashes, so the rendered text is unchanged. - D401 (imperative mood): `_condition_value` and `absent_column` reword the summary to lead with a verb. - D401 on `check_string_min_length` / `check_string_max_length`: pydocstyle's stemmer trips on a leading "String", so the summary now reads "Minimum/Maximum character length check for strings", matching the sibling `check_array_*` docstrings. - D403 on `map_runtime_helper`: the check lowercases the proper noun "PySpark" to "Pyspark", so the summary is reworded to keep PySpark out of first position. Docstring text only, no behavior change. Refs #581 Signed-off-by: Seth Fitzsimmons --- .../overture/schema/codegen/markdown/type_format.py | 4 ++-- .../schema/codegen/pyspark/_render_common.py | 2 +- .../schema/codegen/pyspark/constraint_dispatch.py | 4 ++-- .../schema/codegen/pyspark/test_data/base_row.py | 2 +- .../src/overture/schema/pyspark/cli.py | 2 +- .../pyspark/expressions/constraint_expressions.py | 12 ++++++------ 6 files changed, 13 insertions(+), 13 deletions(-) diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/markdown/type_format.py b/packages/overture-schema-codegen/src/overture/schema/codegen/markdown/type_format.py index 55dde02cc..da72e186b 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/markdown/type_format.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/markdown/type_format.py @@ -238,7 +238,7 @@ def _map_side_link(shape: FieldShape, ctx: LinkContext | None) -> str | None: def _bare_map_side_name(shape: FieldShape) -> str: - """Bare markdown name for a map key/value, recursing through containers. + r"""Bare markdown name for a map key/value, recursing through containers. Every variant resolves to a real name: `list<...>` / `map<...>` wrappers recurse, scalars use their registry name (so `Any` is `Any`, @@ -248,7 +248,7 @@ def _bare_map_side_name(shape: FieldShape) -> str: bug, not a placeholder. A union-valued map is the one shape left unrendered: no schema field - uses one, and its `\\|`-separated members do not compose cleanly into + uses one, and its `\|`-separated members do not compose cleanly into a bare `map<...>` span. It raises so the gap surfaces loudly when a field first needs it, rather than shipping a half-rendered value. """ diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/pyspark/_render_common.py b/packages/overture-schema-codegen/src/overture/schema/codegen/pyspark/_render_common.py index bfa76ca0b..1afac65c8 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/pyspark/_render_common.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/pyspark/_render_common.py @@ -84,7 +84,7 @@ def map_runtime_helper(projection: MapProjection) -> str: - """PySpark column-patterns helper name for a map projection. + """Map a projection to its PySpark column-patterns helper name. `MapProjection.KEY` -> `map_keys_check`; `MapProjection.VALUE` -> `map_values_check`. This is a pyspark-layer diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/pyspark/constraint_dispatch.py b/packages/overture-schema-codegen/src/overture/schema/codegen/pyspark/constraint_dispatch.py index c79c1f455..f81ada61e 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/pyspark/constraint_dispatch.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/pyspark/constraint_dispatch.py @@ -173,11 +173,11 @@ def compiled_pattern_source(pattern: re.Pattern[str]) -> str: def normalize_anchor(pattern: str) -> str: - """Replace trailing `$` with `\\z` for Java/Spark regex compatibility. + r"""Replace trailing `$` with `\z` for Java/Spark regex compatibility. Uses backslash-parity to distinguish a real anchor from an escaped literal `$`. Counts the run of backslashes immediately before the - final `$`: an even count means `$` is unescaped (convert to `\\z`); + final `$`: an even count means `$` is unescaped (convert to `\z`); an odd count means it is an escaped literal `$` (leave unchanged). """ if not pattern.endswith("$"): diff --git a/packages/overture-schema-codegen/src/overture/schema/codegen/pyspark/test_data/base_row.py b/packages/overture-schema-codegen/src/overture/schema/codegen/pyspark/test_data/base_row.py index 70a20e0ee..82b7a6f2c 100644 --- a/packages/overture-schema-codegen/src/overture/schema/codegen/pyspark/test_data/base_row.py +++ b/packages/overture-schema-codegen/src/overture/schema/codegen/pyspark/test_data/base_row.py @@ -245,7 +245,7 @@ def _build_arm_rows( def _condition_value(field_eq: FieldEq) -> object: - """The condition's comparison value, with an `Enum` unwrapped to its value. + """Return the condition's comparison value, with an `Enum` unwrapped to its value. Row dicts store raw scalar values, so an `Enum`-typed condition value is compared and written as its underlying `.value`. diff --git a/packages/overture-schema-pyspark/src/overture/schema/pyspark/cli.py b/packages/overture-schema-pyspark/src/overture/schema/pyspark/cli.py index ee433814b..ed8dae2b8 100644 --- a/packages/overture-schema-pyspark/src/overture/schema/pyspark/cli.py +++ b/packages/overture-schema-pyspark/src/overture/schema/pyspark/cli.py @@ -34,7 +34,7 @@ class ReadSpec: def absent_column(exc: AnalysisException, columns: Collection[str]) -> str | None: - """The top-level column an unresolved-column error names, if absent from data. + """Return the top-level column named by an unresolved-column error, if absent. Returns the column name only when `exc` is an `UNRESOLVED_COLUMN` error whose target is genuinely missing from `columns` -- the case a re-run with diff --git a/packages/overture-schema-pyspark/src/overture/schema/pyspark/expressions/constraint_expressions.py b/packages/overture-schema-pyspark/src/overture/schema/pyspark/expressions/constraint_expressions.py index afb946d02..a6a374d30 100644 --- a/packages/overture-schema-pyspark/src/overture/schema/pyspark/expressions/constraint_expressions.py +++ b/packages/overture-schema-pyspark/src/overture/schema/pyspark/expressions/constraint_expressions.py @@ -133,14 +133,14 @@ def check_required(col: Column) -> Column: def check_pattern(col: Column, pattern: str, *, label: str) -> Column: - """Regex pattern check via rlike. Returns error string or null. + r"""Regex pattern check via rlike. Returns error string or null. Parameters ---------- col Column to validate. pattern - Java regex pattern (use `\\z` for absolute end-of-input). + Java regex pattern (use `\z` for absolute end-of-input). label Human-readable description used in error messages: `"invalid {label}: got '...'"` @@ -150,8 +150,8 @@ def check_pattern(col: Column, pattern: str, *, label: str) -> Column: `rlike` runs Java's regex engine against patterns authored for Python's `re` (the engine Pydantic validates with). The dialects coincide on the ASCII character ranges the schema patterns use, but diverge on the - shorthand classes: Java's `\\d \\s \\w \\S` are ASCII-only while Python's - are Unicode, so e.g. `^\\S+$` accepts a non-breaking space here that + shorthand classes: Java's `\d \s \w \S` are ASCII-only while Python's + are Unicode, so e.g. `^\S+$` accepts a non-breaking space here that Pydantic rejects, and `.` excludes a different set of line terminators. These divergences are accepted -- the affected inputs (Unicode digits, exotic whitespace) do not occur in practice for the constrained fields. @@ -243,12 +243,12 @@ def check_array_max_length(col: Column, max_len: int) -> Column: def check_string_min_length(col: Column, min_len: int) -> Column: - """String minimum character length check. Returns error string or null.""" + """Minimum character length check for strings. Returns error string or null.""" return _check_length(col, F.length(col), min_len, direction="minimum") def check_string_max_length(col: Column, max_len: int) -> Column: - """String maximum character length check. Returns error string or null.""" + """Maximum character length check for strings. Returns error string or null.""" return _check_length(col, F.length(col), max_len, direction="maximum")