diff --git a/docs/config-schema.md b/docs/config-schema.md index 704c915..9cb4f0b 100644 --- a/docs/config-schema.md +++ b/docs/config-schema.md @@ -196,6 +196,7 @@ Note pratiche: `remove_dot_thousands`). Possono essere usate in qualsiasi `clean.sql` senza importazioni né configurazioni. Dettaglio: [standard-macros.md](standard-macros.md). - i file `.xlsx` sono supportati nel layer CLEAN via `engine="openpyxl"` +- i file `.ods` (OpenDocument) sono supportati nel layer CLEAN via `engine="odf"` (odfpy) - i file `.xls` (Excel 97-2003) sono supportati via `engine="xlrd"` - RAW conserva il workbook originale senza convertirlo - per Excel le opzioni principali sono `header`, `skip`, `columns`, `trim_whitespace`, `sheet_name` diff --git a/pyproject.toml b/pyproject.toml index 90f1ef9..5c19895 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -41,6 +41,7 @@ pandas = [ "pandas>=2.0", "openpyxl>=3.1.0", "xlrd>=2.0.0", + "odfpy>=1.4.0", ] dev = [ "build>=1.4.3", diff --git a/tests/test_clean_duckdb_read.py b/tests/test_clean_duckdb_read.py index 23e3e37..aa01b1f 100644 --- a/tests/test_clean_duckdb_read.py +++ b/tests/test_clean_duckdb_read.py @@ -276,6 +276,140 @@ def test_read_raw_to_relation_reads_xlsx_first_sheet(tmp_path: Path): assert rows == [(2022, "Lazio", 123.4), (2022, "Umbria", 56.7)] +@pytest.mark.policy +def test_read_raw_to_relation_reads_ods_with_odf_engine(tmp_path: Path): + """ODS (OpenDocument) read via pandas engine odf — ACI Auto-Trend e simili.""" + import pandas as pd + + input_file = tmp_path / "ok.ods" + pd.DataFrame( + [ + {"ANNO": 2022, "MESE": "GENNAIO", "UFFICIO": "ROMA", "QUANTITA": "1.600"}, + {"ANNO": 2022, "MESE": "FEBBRAIO", "UFFICIO": "TORINO", "QUANTITA": "987"}, + ] + ).to_excel(input_file, index=False, engine="odf") + + with safe_connect() as con: + logger = logging.getLogger("tests.clean.duckdb_read.ods") + + info = duckdb_read.read_raw_to_relation( + con, + [input_file], + {"header": True}, + "fallback", + logger, + ) + + rows = con.execute( + 'SELECT "ANNO", "MESE", "UFFICIO", "QUANTITA" FROM raw_input ORDER BY "MESE"' + ).fetchall() + assert info.source == "ods" + assert rows == [ + (2022, "FEBBRAIO", "TORINO", "987"), + (2022, "GENNAIO", "ROMA", "1.600"), + ] + + +@pytest.mark.policy +def test_read_excel_compact_columns_rename_xlsx(tmp_path: Path): + """policy: colonne compatto clean_name:TYPE rinominano nel path Excel/ODS. + + open-aci Auto-Trend usa columns in formato compatto su CSV e ODS: + senza rename, raw_input terrebbe i nomi grezzi e clean.sql fallirebbe. + """ + input_file = tmp_path / "compact.xlsx" + pd.DataFrame( + [ + {"ANNO": 2022, "UFFICIO PRA DI COMPETENZA": "ROMA", "QUANTITA'": "1.600"}, + {"ANNO": 2022, "UFFICIO PRA DI COMPETENZA": "TORINO", "QUANTITA'": "987"}, + ] + ).to_excel(input_file, index=False) + + with safe_connect() as con: + logger = logging.getLogger("tests.clean.duckdb_read.xlsx_compact_cols") + + info = duckdb_read.read_raw_to_relation( + con, + [input_file], + { + "header": True, + "columns": { + "ANNO": "anno:INTEGER", + "UFFICIO PRA DI COMPETENZA": "ufficio_pra:VARCHAR", + "QUANTITA'": "quantita:VARCHAR", + }, + }, + "fallback", + logger, + ) + + cols = [r[0] for r in con.execute("DESCRIBE raw_input").fetchall()] + assert cols == ["anno", "ufficio_pra", "quantita"] + rows = con.execute( + "SELECT anno, ufficio_pra, quantita FROM raw_input ORDER BY ufficio_pra" + ).fetchall() + assert info.source == "excel" + assert rows == [(2022, "ROMA", "1.600"), (2022, "TORINO", "987")] + + +@pytest.mark.policy +def test_read_excel_compact_columns_rename_ods(tmp_path: Path): + """policy: stesso rename compatto su .ods (consumer open-aci Auto-Trend ODS).""" + input_file = tmp_path / "compact.ods" + pd.DataFrame( + [ + {"ANNO": 2022, "UFFICIO PRA DI COMPETENZA": "ROMA", "QUANTITA'": "1.600"}, + {"ANNO": 2022, "UFFICIO PRA DI COMPETENZA": "TORINO", "QUANTITA'": "987"}, + ] + ).to_excel(input_file, index=False, engine="odf") + + with safe_connect() as con: + logger = logging.getLogger("tests.clean.duckdb_read.ods_compact_cols") + + info = duckdb_read.read_raw_to_relation( + con, + [input_file], + { + "header": True, + "columns": { + "ANNO": "anno:INTEGER", + "UFFICIO PRA DI COMPETENZA": "ufficio_pra:VARCHAR", + "QUANTITA'": "quantita:VARCHAR", + }, + }, + "fallback", + logger, + ) + + cols = [r[0] for r in con.execute("DESCRIBE raw_input").fetchall()] + assert cols == ["anno", "ufficio_pra", "quantita"] + rows = con.execute( + "SELECT anno, ufficio_pra, quantita FROM raw_input ORDER BY ufficio_pra" + ).fetchall() + assert info.source == "ods" + assert rows == [(2022, "ROMA", "1.600"), (2022, "TORINO", "987")] + + +@pytest.mark.policy +def test_read_excel_source_label_uses_all_batch_suffixes(tmp_path: Path): + """policy: source_label considera tutti i suffix del batch, non solo il primo.""" + xlsx = tmp_path / "a.xlsx" + ods = tmp_path / "b.ods" + pd.DataFrame([{"Anno": 2022, "Val": 1}]).to_excel(xlsx, index=False) + pd.DataFrame([{"Anno": 2022, "Val": 2}]).to_excel(ods, index=False, engine="odf") + + with safe_connect() as con: + logger = logging.getLogger("tests.clean.duckdb_read.batch_label") + info = duckdb_read.read_raw_to_relation( + con, + [xlsx, ods], + {"header": True}, + "fallback", + logger, + ) + assert info.source == "excel+ods" + + @pytest.mark.policy def test_read_raw_to_relation_reads_xlsx_with_explicit_sheet_and_columns(tmp_path: Path): input_file = tmp_path / "sheeted.xlsx" diff --git a/toolkit/clean/input_selection.py b/toolkit/clean/input_selection.py index 011fea8..49c940a 100644 --- a/toolkit/clean/input_selection.py +++ b/toolkit/clean/input_selection.py @@ -14,7 +14,7 @@ def is_supported_input_file(path: Path) -> bool: return False if name.endswith((".csv.gz", ".tsv.gz", ".txt.gz", ".nt.gz")): return True - if path.suffix.lower() in {".csv", ".tsv", ".txt", ".parquet", ".xlsx", ".xls"}: + if path.suffix.lower() in {".csv", ".tsv", ".txt", ".parquet", ".xlsx", ".xls", ".ods"}: return True return False diff --git a/toolkit/core/duckdb_read.py b/toolkit/core/duckdb_read.py index 29cf3d9..e22a650 100644 --- a/toolkit/core/duckdb_read.py +++ b/toolkit/core/duckdb_read.py @@ -29,6 +29,7 @@ ".txt.gz", ".xlsx", ".xls", + ".ods", ".nt.gz", } @@ -443,7 +444,7 @@ def read_raw_to_relation( info = _execute_parquet_read(con, normalized) logger.info("read_csv params used: source=parquet params={}") return info - if exts <= {".xlsx", ".xls"}: + if exts <= {".xlsx", ".xls", ".ods"}: result = _execute_excel_read(con, [f.path for f in normalized], read_cfg, logger=logger) return ReadInfo(source=result["source"], params_used=result["params_used"]) diff --git a/toolkit/core/read_excel.py b/toolkit/core/read_excel.py index ff30522..ba59e79 100644 --- a/toolkit/core/read_excel.py +++ b/toolkit/core/read_excel.py @@ -49,24 +49,44 @@ def _load_excel_frame( ext = input_file.suffix.lower() # Literal richiesto dagli overload pandas (engine non è str generico) - engine: Literal["xlrd", "openpyxl"] = "xlrd" if ext == ".xls" else "openpyxl" - df = pd.read_excel( - input_file, - sheet_name=sheet_name, - header=0 if header else None, - skiprows=skip, - dtype=object, - engine=engine, - ) + if ext == ".xls": + engine: Literal["xlrd", "openpyxl", "odf"] = "xlrd" + elif ext == ".ods": + engine = "odf" + else: + engine = "openpyxl" + try: + df = pd.read_excel( + input_file, + sheet_name=sheet_name, + header=0 if header else None, + skiprows=skip, + dtype=object, + engine=engine, + ) + except ImportError as e: + if engine == "odf": + raise ImportError( + "Lettura .ods richiede odfpy. Installa le dipendenze pandas: " + "pip install 'dataciviclab-toolkit[pandas]' " + "(include odfpy>=1.4)." + ) from e + raise if columns: - expected_columns = list(columns.keys()) - if len(expected_columns) != len(df.columns): + from toolkit.core.sql_utils import parse_column_value + + # Supporta formato compatto "clean_name:DUCKDB_TYPE" come nel path CSV + resolved_names: list[str] = [] + for raw_name, value in columns.items(): + clean_name, _ = parse_column_value(raw_name, value) + resolved_names.append(clean_name) + if len(resolved_names) != len(df.columns): raise ValueError( "Excel input columns mismatch. " - f"Configured={len(expected_columns)} detected={len(df.columns)} file={input_file}" + f"Configured={len(resolved_names)} detected={len(df.columns)} file={input_file}" ) - df.columns = expected_columns + df.columns = resolved_names elif not header: df.columns = [f"col{i}" for i in range(len(df.columns))] @@ -108,8 +128,16 @@ def _execute_excel_read( used = dict(params_used or {}) if used.get("columns") is None: used.pop("columns", None) + suffixes = {f.suffix.lower() for f in input_files if f.suffix} + if suffixes <= {".ods"}: + source_label = "ods" + elif suffixes & {".xlsx", ".xls"} and suffixes & {".ods"}: + source_label = "excel+ods" + else: + source_label = "excel" logger.info( - "read_excel params used: source=excel params=%s", + "read_excel params used: source=%s params=%s", + source_label, json.dumps(used, ensure_ascii=False, sort_keys=True), ) - return {"source": "excel", "params_used": used} + return {"source": source_label, "params_used": used} diff --git a/toolkit/domain/layer.py b/toolkit/domain/layer.py index 2180e46..4d8fae5 100644 --- a/toolkit/domain/layer.py +++ b/toolkit/domain/layer.py @@ -308,11 +308,11 @@ def raw_preview(config_path: str, year: int | None = None, limit: int = 20) -> d from toolkit.domain.profile import csv_preview as _csv_preview return _csv_preview(str(raw_file), limit=limit) - elif suffix in (".xlsx", ".xls"): + elif suffix in (".xlsx", ".xls", ".ods"): return { "path": str(raw_file), - "format": "xlsx", - "note": "File binario XLSX. Usa mode='schema' per lo schema colonne.", + "format": suffix.lstrip("."), + "note": f"File binario {suffix.lstrip('.').upper()}. Usa mode='schema' per lo schema colonne.", "dataset": paths.get("dataset"), "year": paths.get("year"), } diff --git a/toolkit/plugins/_http_utils.py b/toolkit/plugins/_http_utils.py index 02905c4..8a840ac 100644 --- a/toolkit/plugins/_http_utils.py +++ b/toolkit/plugins/_http_utils.py @@ -16,6 +16,7 @@ ".zip", ".xlsx", ".xls", + ".ods", ".gz", ".bz2", ".7z", diff --git a/toolkit/raw/run.py b/toolkit/raw/run.py index 60a49b4..ab1cc5a 100644 --- a/toolkit/raw/run.py +++ b/toolkit/raw/run.py @@ -161,6 +161,7 @@ def run_raw( ".txt", ".xlsx", ".xls", + ".ods", }: try: profile_hints = sniff_source_file(primary_output_path)