From 708789d19788bed443b15b04cc6f422c77976180 Mon Sep 17 00:00:00 2001 From: Zio Gabber <78922322+Gabrymi93@users.noreply.github.com> Date: Sun, 4 Oct 2026 14:59:15 +0100 Subject: [PATCH 1/2] feat(clean): support ODS via pandas engine odf OpenDocument Spreadsheet (.ods) leggibile nel layer CLEAN, stessa path di Excel/xlsx. - SUPPORTED_INPUT_EXTS + .ods (duckdb_read, input_selection) - read_excel: engine odf per .ods; colonne compatto clean_name:TYPE - odfpy in extra pandas (pyproject) - .ods non troncabili in HTTP Range; sniff/preview raw - test: read ODS first sheet --- docs/config-schema.md | 1 + pyproject.toml | 1 + tests/test_clean_duckdb_read.py | 34 ++++++++++++++++++++++++++++++++ toolkit/clean/input_selection.py | 2 +- toolkit/core/duckdb_read.py | 3 ++- toolkit/core/read_excel.py | 29 ++++++++++++++++++++------- toolkit/domain/layer.py | 6 +++--- toolkit/plugins/_http_utils.py | 1 + toolkit/raw/run.py | 1 + 9 files changed, 66 insertions(+), 12 deletions(-) diff --git a/docs/config-schema.md b/docs/config-schema.md index 704c915b..9cb4f0b5 100644 --- a/docs/config-schema.md +++ b/docs/config-schema.md @@ -196,6 +196,7 @@ Note pratiche: `remove_dot_thousands`). Possono essere usate in qualsiasi `clean.sql` senza importazioni né configurazioni. Dettaglio: [standard-macros.md](standard-macros.md). - i file `.xlsx` sono supportati nel layer CLEAN via `engine="openpyxl"` +- i file `.ods` (OpenDocument) sono supportati nel layer CLEAN via `engine="odf"` (odfpy) - i file `.xls` (Excel 97-2003) sono supportati via `engine="xlrd"` - RAW conserva il workbook originale senza convertirlo - per Excel le opzioni principali sono `header`, `skip`, `columns`, `trim_whitespace`, `sheet_name` diff --git a/pyproject.toml b/pyproject.toml index 90f1ef97..5c198953 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -41,6 +41,7 @@ pandas = [ "pandas>=2.0", "openpyxl>=3.1.0", "xlrd>=2.0.0", + "odfpy>=1.4.0", ] dev = [ "build>=1.4.3", diff --git a/tests/test_clean_duckdb_read.py b/tests/test_clean_duckdb_read.py index 23e3e37d..0b6f00d6 100644 --- a/tests/test_clean_duckdb_read.py +++ b/tests/test_clean_duckdb_read.py @@ -276,6 +276,40 @@ def test_read_raw_to_relation_reads_xlsx_first_sheet(tmp_path: Path): assert rows == [(2022, "Lazio", 123.4), (2022, "Umbria", 56.7)] +@pytest.mark.policy +def test_read_raw_to_relation_reads_ods_with_odf_engine(tmp_path: Path): + """ODS (OpenDocument) read via pandas engine odf — ACI Auto-Trend e simili.""" + import pandas as pd + + input_file = tmp_path / "ok.ods" + pd.DataFrame( + [ + {"ANNO": 2022, "MESE": "GENNAIO", "UFFICIO": "ROMA", "QUANTITA": "1.600"}, + {"ANNO": 2022, "MESE": "FEBBRAIO", "UFFICIO": "TORINO", "QUANTITA": "987"}, + ] + ).to_excel(input_file, index=False, engine="odf") + + with safe_connect() as con: + logger = logging.getLogger("tests.clean.duckdb_read.ods") + + info = duckdb_read.read_raw_to_relation( + con, + [input_file], + {"header": True}, + "fallback", + logger, + ) + + rows = con.execute( + 'SELECT "ANNO", "MESE", "UFFICIO", "QUANTITA" FROM raw_input ORDER BY "MESE"' + ).fetchall() + assert info.source == "ods" + assert rows == [ + (2022, "FEBBRAIO", "TORINO", "987"), + (2022, "GENNAIO", "ROMA", "1.600"), + ] + + @pytest.mark.policy def test_read_raw_to_relation_reads_xlsx_with_explicit_sheet_and_columns(tmp_path: Path): input_file = tmp_path / "sheeted.xlsx" diff --git a/toolkit/clean/input_selection.py b/toolkit/clean/input_selection.py index 011fea80..49c940a8 100644 --- a/toolkit/clean/input_selection.py +++ b/toolkit/clean/input_selection.py @@ -14,7 +14,7 @@ def is_supported_input_file(path: Path) -> bool: return False if name.endswith((".csv.gz", ".tsv.gz", ".txt.gz", ".nt.gz")): return True - if path.suffix.lower() in {".csv", ".tsv", ".txt", ".parquet", ".xlsx", ".xls"}: + if path.suffix.lower() in {".csv", ".tsv", ".txt", ".parquet", ".xlsx", ".xls", ".ods"}: return True return False diff --git a/toolkit/core/duckdb_read.py b/toolkit/core/duckdb_read.py index 29cf3d99..e22a650f 100644 --- a/toolkit/core/duckdb_read.py +++ b/toolkit/core/duckdb_read.py @@ -29,6 +29,7 @@ ".txt.gz", ".xlsx", ".xls", + ".ods", ".nt.gz", } @@ -443,7 +444,7 @@ def read_raw_to_relation( info = _execute_parquet_read(con, normalized) logger.info("read_csv params used: source=parquet params={}") return info - if exts <= {".xlsx", ".xls"}: + if exts <= {".xlsx", ".xls", ".ods"}: result = _execute_excel_read(con, [f.path for f in normalized], read_cfg, logger=logger) return ReadInfo(source=result["source"], params_used=result["params_used"]) diff --git a/toolkit/core/read_excel.py b/toolkit/core/read_excel.py index ff305229..ab66dde3 100644 --- a/toolkit/core/read_excel.py +++ b/toolkit/core/read_excel.py @@ -49,7 +49,12 @@ def _load_excel_frame( ext = input_file.suffix.lower() # Literal richiesto dagli overload pandas (engine non è str generico) - engine: Literal["xlrd", "openpyxl"] = "xlrd" if ext == ".xls" else "openpyxl" + if ext == ".xls": + engine: Literal["xlrd", "openpyxl", "odf"] = "xlrd" + elif ext == ".ods": + engine = "odf" + else: + engine = "openpyxl" df = pd.read_excel( input_file, sheet_name=sheet_name, @@ -60,13 +65,19 @@ def _load_excel_frame( ) if columns: - expected_columns = list(columns.keys()) - if len(expected_columns) != len(df.columns): + from toolkit.core.sql_utils import parse_column_value + + # Supporta formato compatto "clean_name:DUCKDB_TYPE" come nel path CSV + resolved_names: list[str] = [] + for raw_name, value in columns.items(): + clean_name, _ = parse_column_value(raw_name, value) + resolved_names.append(clean_name) + if len(resolved_names) != len(df.columns): raise ValueError( "Excel input columns mismatch. " - f"Configured={len(expected_columns)} detected={len(df.columns)} file={input_file}" + f"Configured={len(resolved_names)} detected={len(df.columns)} file={input_file}" ) - df.columns = expected_columns + df.columns = resolved_names elif not header: df.columns = [f"col{i}" for i in range(len(df.columns))] @@ -108,8 +119,12 @@ def _execute_excel_read( used = dict(params_used or {}) if used.get("columns") is None: used.pop("columns", None) + source_label = "excel" + if input_files and input_files[0].suffix.lower() == ".ods": + source_label = "ods" logger.info( - "read_excel params used: source=excel params=%s", + "read_excel params used: source=%s params=%s", + source_label, json.dumps(used, ensure_ascii=False, sort_keys=True), ) - return {"source": "excel", "params_used": used} + return {"source": source_label, "params_used": used} diff --git a/toolkit/domain/layer.py b/toolkit/domain/layer.py index 2180e466..4d8fae54 100644 --- a/toolkit/domain/layer.py +++ b/toolkit/domain/layer.py @@ -308,11 +308,11 @@ def raw_preview(config_path: str, year: int | None = None, limit: int = 20) -> d from toolkit.domain.profile import csv_preview as _csv_preview return _csv_preview(str(raw_file), limit=limit) - elif suffix in (".xlsx", ".xls"): + elif suffix in (".xlsx", ".xls", ".ods"): return { "path": str(raw_file), - "format": "xlsx", - "note": "File binario XLSX. Usa mode='schema' per lo schema colonne.", + "format": suffix.lstrip("."), + "note": f"File binario {suffix.lstrip('.').upper()}. Usa mode='schema' per lo schema colonne.", "dataset": paths.get("dataset"), "year": paths.get("year"), } diff --git a/toolkit/plugins/_http_utils.py b/toolkit/plugins/_http_utils.py index 02905c4d..8a840ac8 100644 --- a/toolkit/plugins/_http_utils.py +++ b/toolkit/plugins/_http_utils.py @@ -16,6 +16,7 @@ ".zip", ".xlsx", ".xls", + ".ods", ".gz", ".bz2", ".7z", diff --git a/toolkit/raw/run.py b/toolkit/raw/run.py index 60a49b4f..ab1cc5ae 100644 --- a/toolkit/raw/run.py +++ b/toolkit/raw/run.py @@ -161,6 +161,7 @@ def run_raw( ".txt", ".xlsx", ".xls", + ".ods", }: try: profile_hints = sniff_source_file(primary_output_path) From be4ddf4b4ba13ea5c092a35d52a433928edd7e51 Mon Sep 17 00:00:00 2001 From: Zio Gabber <78922322+Gabrymi93@users.noreply.github.com> Date: Sun, 4 Oct 2026 15:46:47 +0100 Subject: [PATCH 2/2] fix(clean): test colonne compatto Excel/ODS + source_label batch + hint odfpy Review PR #495: - policy test clean_name:TYPE rename su .xlsx e .ods (consumer open-aci Auto-Trend) - source_label da tutti i suffix del batch (excel / ods / excel+ods) - ImportError .ods con hint install extra pandas/odfpy --- tests/test_clean_duckdb_read.py | 100 ++++++++++++++++++++++++++++++++ toolkit/core/read_excel.py | 33 +++++++---- 2 files changed, 123 insertions(+), 10 deletions(-) diff --git a/tests/test_clean_duckdb_read.py b/tests/test_clean_duckdb_read.py index 0b6f00d6..aa01b1fa 100644 --- a/tests/test_clean_duckdb_read.py +++ b/tests/test_clean_duckdb_read.py @@ -310,6 +310,106 @@ def test_read_raw_to_relation_reads_ods_with_odf_engine(tmp_path: Path): ] +@pytest.mark.policy +def test_read_excel_compact_columns_rename_xlsx(tmp_path: Path): + """policy: colonne compatto clean_name:TYPE rinominano nel path Excel/ODS. + + open-aci Auto-Trend usa columns in formato compatto su CSV e ODS: + senza rename, raw_input terrebbe i nomi grezzi e clean.sql fallirebbe. + """ + input_file = tmp_path / "compact.xlsx" + pd.DataFrame( + [ + {"ANNO": 2022, "UFFICIO PRA DI COMPETENZA": "ROMA", "QUANTITA'": "1.600"}, + {"ANNO": 2022, "UFFICIO PRA DI COMPETENZA": "TORINO", "QUANTITA'": "987"}, + ] + ).to_excel(input_file, index=False) + + with safe_connect() as con: + logger = logging.getLogger("tests.clean.duckdb_read.xlsx_compact_cols") + + info = duckdb_read.read_raw_to_relation( + con, + [input_file], + { + "header": True, + "columns": { + "ANNO": "anno:INTEGER", + "UFFICIO PRA DI COMPETENZA": "ufficio_pra:VARCHAR", + "QUANTITA'": "quantita:VARCHAR", + }, + }, + "fallback", + logger, + ) + + cols = [r[0] for r in con.execute("DESCRIBE raw_input").fetchall()] + assert cols == ["anno", "ufficio_pra", "quantita"] + rows = con.execute( + "SELECT anno, ufficio_pra, quantita FROM raw_input ORDER BY ufficio_pra" + ).fetchall() + assert info.source == "excel" + assert rows == [(2022, "ROMA", "1.600"), (2022, "TORINO", "987")] + + +@pytest.mark.policy +def test_read_excel_compact_columns_rename_ods(tmp_path: Path): + """policy: stesso rename compatto su .ods (consumer open-aci Auto-Trend ODS).""" + input_file = tmp_path / "compact.ods" + pd.DataFrame( + [ + {"ANNO": 2022, "UFFICIO PRA DI COMPETENZA": "ROMA", "QUANTITA'": "1.600"}, + {"ANNO": 2022, "UFFICIO PRA DI COMPETENZA": "TORINO", "QUANTITA'": "987"}, + ] + ).to_excel(input_file, index=False, engine="odf") + + with safe_connect() as con: + logger = logging.getLogger("tests.clean.duckdb_read.ods_compact_cols") + + info = duckdb_read.read_raw_to_relation( + con, + [input_file], + { + "header": True, + "columns": { + "ANNO": "anno:INTEGER", + "UFFICIO PRA DI COMPETENZA": "ufficio_pra:VARCHAR", + "QUANTITA'": "quantita:VARCHAR", + }, + }, + "fallback", + logger, + ) + + cols = [r[0] for r in con.execute("DESCRIBE raw_input").fetchall()] + assert cols == ["anno", "ufficio_pra", "quantita"] + rows = con.execute( + "SELECT anno, ufficio_pra, quantita FROM raw_input ORDER BY ufficio_pra" + ).fetchall() + assert info.source == "ods" + assert rows == [(2022, "ROMA", "1.600"), (2022, "TORINO", "987")] + + +@pytest.mark.policy +def test_read_excel_source_label_uses_all_batch_suffixes(tmp_path: Path): + """policy: source_label considera tutti i suffix del batch, non solo il primo.""" + xlsx = tmp_path / "a.xlsx" + ods = tmp_path / "b.ods" + pd.DataFrame([{"Anno": 2022, "Val": 1}]).to_excel(xlsx, index=False) + pd.DataFrame([{"Anno": 2022, "Val": 2}]).to_excel(ods, index=False, engine="odf") + + with safe_connect() as con: + logger = logging.getLogger("tests.clean.duckdb_read.batch_label") + info = duckdb_read.read_raw_to_relation( + con, + [xlsx, ods], + {"header": True}, + "fallback", + logger, + ) + assert info.source == "excel+ods" + + @pytest.mark.policy def test_read_raw_to_relation_reads_xlsx_with_explicit_sheet_and_columns(tmp_path: Path): input_file = tmp_path / "sheeted.xlsx" diff --git a/toolkit/core/read_excel.py b/toolkit/core/read_excel.py index ab66dde3..ba59e796 100644 --- a/toolkit/core/read_excel.py +++ b/toolkit/core/read_excel.py @@ -55,14 +55,23 @@ def _load_excel_frame( engine = "odf" else: engine = "openpyxl" - df = pd.read_excel( - input_file, - sheet_name=sheet_name, - header=0 if header else None, - skiprows=skip, - dtype=object, - engine=engine, - ) + try: + df = pd.read_excel( + input_file, + sheet_name=sheet_name, + header=0 if header else None, + skiprows=skip, + dtype=object, + engine=engine, + ) + except ImportError as e: + if engine == "odf": + raise ImportError( + "Lettura .ods richiede odfpy. Installa le dipendenze pandas: " + "pip install 'dataciviclab-toolkit[pandas]' " + "(include odfpy>=1.4)." + ) from e + raise if columns: from toolkit.core.sql_utils import parse_column_value @@ -119,9 +128,13 @@ def _execute_excel_read( used = dict(params_used or {}) if used.get("columns") is None: used.pop("columns", None) - source_label = "excel" - if input_files and input_files[0].suffix.lower() == ".ods": + suffixes = {f.suffix.lower() for f in input_files if f.suffix} + if suffixes <= {".ods"}: source_label = "ods" + elif suffixes & {".xlsx", ".xls"} and suffixes & {".ods"}: + source_label = "excel+ods" + else: + source_label = "excel" logger.info( "read_excel params used: source=%s params=%s", source_label,