From 296de991833f5e3f8c01db7693c741123d80e635 Mon Sep 17 00:00:00 2001 From: jstet Date: Sun, 27 Sep 2026 19:31:39 +0200 Subject: [PATCH] feat(ddi2xlsform): DDI back to XLSForm, round trips proven on the model (#154) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - `ddiToXlsform(xml, { onWarning })` / `formtransform ddi2xlsform`: a DDI codebook, or a // fragment, → `{ survey, choices, settings }`. `src/instrument/fromDdi.ts` reads standard DDI first, then the cdl: notes; `src/utils/xmlParse.ts` is a dependency-free XML reader, so it runs in the browser bundle too. - Never refuses DDI it can read: a non-CDL codebook gets one `ddi-field-missing` warning per field only CDL carries, an unknown responseDomainType is text (`ddi-type-unknown`), a fragment's reference outside it is `ddi-reference-outside`. Broken XML is `ddi-invalid`. - The Instrument → XLSForm emitter moves to `src/xlsform/fromInstrument.ts` and is shared with lstsv2xlsform: groups keep their names, grid members their own fields, a shared list is written once, exclusive choices come from the choice rows too, form_id/version/per-language titles are kept. lstsv2xlsform: a LimeSurvey group's description comes back as its hint, and a group name that is a valid XLSForm name is kept (`Later`), else slugified and made unique. - Round trips: every survey fixture and registry entity, and 200 random forms (fast-check), give back their Instrument after XLSForm → DDI, and after XLSForm → DDI → XLSForm sheets, up to the losses src/pipelines/ddi2xlsform/README.md lists. Each survey's output is blessed as ddi2xlsform.json and validated with pyxform. - Schematron: the cdl: note vocabulary (known types, the subject each needs, one note per type and language), generated from the registry. - Emitter fixes the round trip found: a note before a semi-open question was dropped, and a per-language form_title was written as "[object Object]"; it is now titl plus parTitl per language. Co-Authored-By: Claude Opus 5.5 (1M context) --- ARCHITECTURE.md | 8 +- README.md | 26 +- codegen/schematron.py | 45 + .../schematron/ddi_custom_rules.sch | 30 + package-lock.json | 41 + package.json | 1 + scripts/bless-survey-snapshots.mjs | 16 +- scripts/bless.sh | 2 +- src/api.ts | 17 + src/cli.ts | 53 +- src/ddi/codebook.ts | 62 +- src/diagnostics.ts | 5 + src/index.ts | 8 +- src/instrument/fromDdi.ts | 669 +++++++++++++++ src/instrument/fromLstsv.ts | 30 +- src/pipelines/README.md | 46 +- src/pipelines/ddi2xlsform/README.md | 85 ++ src/pipelines/ddi2xlsform/index.ts | 31 + src/pipelines/lstsv2xlsform/README.md | 7 +- src/pipelines/lstsv2xlsform/toXlsform.ts | 341 +------- .../xlsform2lstsv/surveySettingsEmitter.ts | 5 +- src/utils/xmlParse.ts | 173 ++++ src/xlsform/fromInstrument.ts | 317 +++++++ .../languageNames.ts | 0 src/xlsform/types.ts | 3 +- .../surveys/all_types_survey/ddi2xlsform.json | 326 ++++++++ .../appearances_survey/ddi2xlsform.json | 124 +++ .../surveys/basic_survey/ddi2xlsform.json | 45 + .../surveys/bilingual_survey/ddi2xlsform.json | 207 +++++ .../surveys/complex_survey/ddi2xlsform.json | 115 +++ .../complex_xpath_survey/ddi2xlsform.json | 65 ++ .../surveys/grid_survey/ddi2xlsform.json | 85 ++ .../surveys/hints_survey/ddi2xlsform.json | 63 ++ .../multilingual_survey/ddi2xlsform.json | 107 +++ .../surveys/multipage_survey/ddi2xlsform.json | 66 ++ .../surveys/settings_survey/ddi2xlsform.json | 61 ++ tests/fixtures/surveys/testA/ddi2xlsform.json | 430 ++++++++++ tests/fixtures/surveys/testB/ddi2xlsform.json | 774 ++++++++++++++++++ .../ddi2xlsform.json | 97 +++ tests/ts/contract/canonicalInstrument.ts | 212 +++++ tests/ts/contract/ddiRoundtrip.test.ts | 151 ++++ .../ts/contract/ddiRoundtripGenerated.test.ts | 218 +++++ tests/ts/integration/browserBundle.test.ts | 6 + tests/ts/integration/cli.test.ts | 30 + tests/ts/unit/ddi/fromDdi.test.ts | 136 +++ .../lstsv2xlsform/reverseGroups.test.ts | 2 +- .../pipelines/lstsv2xlsform/toXlsform.test.ts | 8 +- .../test_registry_schematron_conformance.py | 16 + tests/validation/test_xlsform_pyxform.py | 24 +- 49 files changed, 4987 insertions(+), 402 deletions(-) create mode 100644 src/instrument/fromDdi.ts create mode 100644 src/pipelines/ddi2xlsform/README.md create mode 100644 src/pipelines/ddi2xlsform/index.ts create mode 100644 src/utils/xmlParse.ts create mode 100644 src/xlsform/fromInstrument.ts rename src/{pipelines/lstsv2xlsform => xlsform}/languageNames.ts (100%) create mode 100644 tests/fixtures/surveys/all_types_survey/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/appearances_survey/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/basic_survey/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/bilingual_survey/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/complex_survey/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/complex_xpath_survey/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/grid_survey/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/hints_survey/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/multilingual_survey/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/multipage_survey/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/settings_survey/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/testA/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/testB/ddi2xlsform.json create mode 100644 tests/fixtures/surveys/validation_relevance_survey/ddi2xlsform.json create mode 100644 tests/ts/contract/canonicalInstrument.ts create mode 100644 tests/ts/contract/ddiRoundtrip.test.ts create mode 100644 tests/ts/contract/ddiRoundtripGenerated.test.ts create mode 100644 tests/ts/unit/ddi/fromDdi.test.ts diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 1642850..460f5dd 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -149,18 +149,16 @@ The migration runs in phases, each keeping every snapshot byte-identical: stores the expression text, not the AST: emitters that need structure (the EM transpiler) parse it, and XPath is what XLSForm writes back. -### Why there is no `ddi2xlsform` or `ddi2lstsv` +### The way back from DDI (`ddi2xlsform`) -**DDI is the terminus of the pipeline graph for now.** It describes a *dataset*, not an *instrument*, and a reverse path needs the whole instrument. #155 plans one. +A DDI codebook describes a *dataset*, not an *instrument*. A codebook formtransform wrote carries the whole instrument too (#155), so `ddi2xlsform` (#154) turns it back into its form: `src/instrument/fromDdi.ts` reads the standard elements and the `cdl:` notes into the Instrument, and the XLSForm emitter both reverse paths share (`src/xlsform/fromInstrument.ts`) writes the sheets. Any other DDI converts as far as its standard elements go, with a warning for each missing field. The round trip is tested on the model, never on bytes; `src/pipelines/ddi2xlsform/README.md` lists the losses. A `ddi2lstsv` would be `ddi2xlsform` → `xlsform2lstsv` and earns no module of its own. -The canonical `Variable` (`src/ddi/types.ts`) is what survives an emit: +What a CDL codebook carries: - **`relevant`, `constraint`, `constraint_message`, `required`** survive (#151). DDI Codebook 2.5 has no expression syntax, so each goes in twice: readable (`` prose, `` for a simple numeric range) and exact, in a typed ``. `convention:logicMapping` (`ddiEncoding`) defines the notes. A group's own condition is on its `varGrp`; a variable's `` states its groups' conditions too. - **Groups and order** survive (#152): every group is a `` (a plain one `type="section"`), nested through `@varGrp`, and the ``s and data columns follow the survey. - **Every other form field** (#153, `convention:ddiFields`): standard DDI where it has a home (the hint as `postQTxt`, `guidance_hint` as `ivuInstr`, `varFormat/@category` for date and time, `var/@dcml="0"` for integer, `valrng` for a range, `qstn/@seqNo` and `qstn/backward`), else a typed note (`cdl:default`, `cdl:appearance`, `cdl:parameters`, a group's `cdl:hint`, `cdl:exclusive`, `cdl:setting`). A unit test fails when a model field has neither nor a documented loss. Not carried: `calculation` and other unlifted columns, a choice list's name, the `or_other` shorthand as such. -Compare `lstsv2xlsform`, which *is* implemented: a LimeSurvey structure TSV carries `relevance`, `em_validation_q`, `mandatory`, `default` and the `!`/`T` type overrides. It is a form definition in a different dialect, so reversing it is a translation problem. Reversing a codebook without those pieces is a *reconstruction* problem, and they cannot be inferred from it. - ## Development Workflow ### Adding a New Question Type diff --git a/README.md b/README.md index da26649..e227cec 100644 --- a/README.md +++ b/README.md @@ -67,6 +67,7 @@ import { xlsformToDdi, lstsvToDdi, lstsvToXlsform, + ddiToXlsform, } from '@correlaid/formtransform'; const bytes = await file.arrayBuffer(); @@ -74,6 +75,7 @@ const tsv = await xlsformToLstsv(bytes); const xml = xlsformToDdi(bytes); const fromLs = lstsvToDdi(tsv); const { survey, choices, settings } = lstsvToXlsform(tsv); +const back = ddiToXlsform(xml); // a codebook, or a / fragment ``` Each rejects input outside its subset with a `ConversionError` unless you pass @@ -124,6 +126,9 @@ formtransform lstsv2ddi survey.tsv -o codebook.xml --data responses.csv # Convert from LimeSurvey TSV back to XLSForm (emitted as JSON sheets) formtransform lstsv2xlsform survey.tsv -o recovered.json + +# Convert a DDI codebook (or a / fragment) back to XLSForm (JSON sheets) +formtransform ddi2xlsform codebook.xml -o form.json ``` ### Asking the library what exists @@ -183,21 +188,22 @@ records how each maps onto it. | [LimeSurvey TSV](https://www.limesurvey.org/manual/Tab_Separated_Value_survey_structure) | **Deployment** — recreate the survey in LimeSurvey | type codes (`L`, `M`, `F`, `N`) | | [DDI Codebook 2.5](https://ddialliance.org/Specification/DDI-Codebook/2.5/) | **Documentation** — describe the resulting dataset | interval class + response domains (`category`, `multiple`) | -**Supported directions:** four, one per module under `src/pipelines/` — +**Supported directions:** five, one per module under `src/pipelines/` — `xlsform2lstsv` (deploy the survey), `xlsform2ddi` (document the dataset), -`lstsv2ddi` and `lstsv2xlsform` (the reverse paths). All are lossy for some +`lstsv2ddi`, `lstsv2xlsform` and `ddi2xlsform` (the reverse paths). All are lossy for some types: nested groups flatten in LimeSurvey, choice codes over 5 chars truncate, `select_multiple` becomes N binary variables, and the reverse paths cannot recover a select's authored `list_name`. -**DDI has no way back yet:** there is no `ddi2xlsform` or `ddi2lstsv`. A CDL -codebook carries skip logic, validation and `required`: each condition as a -readable `` sentence (and a simple numeric range as ``), plus -the exact expression in a typed note such as `` (`convention:logicMapping`). Groups, order, hints, -defaults, appearances and parameters are in it too, in standard DDI where it -has a place and typed notes where not (`convention:ddiFields`). The reverse -parser that reads them back is #154. +**DDI goes back to XLSForm:** a CDL codebook carries the whole form. Skip +logic, validation and `required` are each a readable `` sentence +(a simple numeric range also ``), plus the exact expression in a typed +note such as `` +(`convention:logicMapping`). Groups, order, hints, defaults, appearances and +parameters are in standard DDI where it has a place and typed notes where not +(`convention:ddiFields`). `ddiToXlsform` reads it back; any other DDI converts +as far as its standard elements go, with a warning per missing field +([`src/pipelines/ddi2xlsform/README.md`](src/pipelines/ddi2xlsform/README.md)). ## Errors and warnings diff --git a/codegen/schematron.py b/codegen/schematron.py index d890c29..eb08130 100644 --- a/codegen/schematron.py +++ b/codegen/schematron.py @@ -33,7 +33,20 @@ def rdt_of(slug: str) -> str | None: companion_type = other.get("companionType", "text") comp_ddi = registry.get(f"type:{companion_type}", {}).get("ddi", {}) + # The `cdl:` note vocabulary (convention:logicMapping ddiEncoding, convention:ddiFields): + # each type, and the @subject a type requires (a fixed value, or `True` for "any"). + logic = registry.get("convention:logicMapping", {}).get("rule", {}).get("ddiEncoding", {}) + fields = registry.get("convention:ddiFields", {}).get("rule", {}) + note_specs = [*logic.get("notes", {}).values(), *fields.get("notes", {}).values()] + note_types = sorted({n["type"] for n in note_specs}) + subject = logic.get("noteSubject") + note_subjects = { + n["type"]: (n["subject"] if n.get("subject") == subject else True) for n in note_specs if n.get("subject") + } + return { + "note_types": note_types, + "note_subjects": note_subjects, "cat_rdts": cat_rdts, "vg_types": vg_types, "category_rdt": rdt_of("select_one"), @@ -175,11 +188,43 @@ def generate_schematron(registry: dict[str, Any], output: Path) -> None: """ + note_test = " or ".join(f"@type = '{t}'" for t in f["note_types"]) + note_msg = ", ".join(f["note_types"]) + subject_rules = "\n".join( + ( + f" " + f'A {t} note needs subject="{subj}": its text is an expression in that syntax.' + ) + if subj is not True + else ( + f" " + f"A {t} note needs a subject: the name of what it holds." + ) + for t, subj in sorted(f["note_subjects"].items()) + ) + cdl_notes = f"""\ + + Note type "" is not in the CDL vocabulary ({note_msg}). +{subject_rules} + +""" + # At most one note of each cdl: type per element and language (per subject on stdyDscr). + cdl_note_uniqueness = """\ + + has more than one note of one cdl: type in one language. + + + A setting has more than one cdl:setting note. + +""" + patterns = [ ("uniqueness", uniqueness), ("essentials", essentials), ("logic", logic), ("other_variables", other_variables), + ("cdl_notes", cdl_notes), + ("cdl_note_uniqueness", cdl_note_uniqueness), ] out: list[str] = [ diff --git a/ddi-validation/schematron/ddi_custom_rules.sch b/ddi-validation/schematron/ddi_custom_rules.sch index 077bea4..8477fe5 100644 --- a/ddi-validation/schematron/ddi_custom_rules.sch +++ b/ddi-validation/schematron/ddi_custom_rules.sch @@ -214,4 +214,34 @@ + + + Note type "" is not in the CDL vocabulary (cdl:appearance, cdl:constraint, cdl:constraint_message, cdl:default, cdl:exclusive, cdl:hint, cdl:parameters, cdl:relevant, cdl:required, cdl:setting). + A cdl:constraint note needs subject="xlsform-xpath": its text is an expression in that syntax. + A cdl:relevant note needs subject="xlsform-xpath": its text is an expression in that syntax. + A cdl:setting note needs a subject: the name of what it holds. + + + Note type "" is not in the CDL vocabulary (cdl:appearance, cdl:constraint, cdl:constraint_message, cdl:default, cdl:exclusive, cdl:hint, cdl:parameters, cdl:relevant, cdl:required, cdl:setting). + A cdl:constraint note needs subject="xlsform-xpath": its text is an expression in that syntax. + A cdl:relevant note needs subject="xlsform-xpath": its text is an expression in that syntax. + A cdl:setting note needs a subject: the name of what it holds. + + + + + + has more than one note of one cdl: type in one language. + + + A setting has more than one cdl:setting note. + + + has more than one note of one cdl: type in one language. + + + A setting has more than one cdl:setting note. + + + diff --git a/package-lock.json b/package-lock.json index e262260..203e32b 100644 --- a/package-lock.json +++ b/package-lock.json @@ -25,6 +25,7 @@ "concurrently": "^10.0.3", "esbuild": "^0.28.1", "eslint": "^10.7.0", + "fast-check": "^4.10.2", "globals": "^17.7.0", "husky": "^9.1.7", "jscpd": "^5.0.12", @@ -4511,6 +4512,29 @@ "node": ">=12.0.0" } }, + "node_modules/fast-check": { + "version": "4.10.2", + "resolved": "https://registry.npmjs.org/fast-check/-/fast-check-4.10.2.tgz", + "integrity": "sha512-iK2f+YrcmoeGqk6fA0ea2bptcu/itMIm4NfEozq6N25+aG6h7s5HZbB/k1aV7b5w5sFLMCbbtRUsTVR+BgC3xw==", + "dev": true, + "funding": [ + { + "type": "individual", + "url": "https://github.com/sponsors/dubzzz" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/fast-check" + } + ], + "license": "MIT", + "dependencies": { + "pure-rand": "^8.0.0" + }, + "engines": { + "node": ">=12.17.0" + } + }, "node_modules/fast-deep-equal": { "version": "3.1.3", "resolved": "https://registry.npmjs.org/fast-deep-equal/-/fast-deep-equal-3.1.3.tgz", @@ -7364,6 +7388,23 @@ "node": ">=6" } }, + "node_modules/pure-rand": { + "version": "8.4.2", + "resolved": "https://registry.npmjs.org/pure-rand/-/pure-rand-8.4.2.tgz", + "integrity": "sha512-vvuOGgcuPJAirlHvuQw1TrOiw7ptaIXXmIbNuiNOY6lNGJJH49PQ1Kj4nd783nPdQhQdicgOjVI2yI/9BD6/Ng==", + "dev": true, + "funding": [ + { + "type": "individual", + "url": "https://github.com/sponsors/dubzzz" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/fast-check" + } + ], + "license": "MIT" + }, "node_modules/queue-microtask": { "version": "1.2.3", "resolved": "https://registry.npmjs.org/queue-microtask/-/queue-microtask-1.2.3.tgz", diff --git a/package.json b/package.json index 1cbe0d8..9701879 100644 --- a/package.json +++ b/package.json @@ -78,6 +78,7 @@ "concurrently": "^10.0.3", "esbuild": "^0.28.1", "eslint": "^10.7.0", + "fast-check": "^4.10.2", "globals": "^17.7.0", "husky": "^9.1.7", "jscpd": "^5.0.12", diff --git a/scripts/bless-survey-snapshots.mjs b/scripts/bless-survey-snapshots.mjs index 2744d88..589d834 100644 --- a/scripts/bless-survey-snapshots.mjs +++ b/scripts/bless-survey-snapshots.mjs @@ -1,7 +1,7 @@ #!/usr/bin/env node /** * Re-bless the frozen whole-survey snapshots - * (tests/fixtures/surveys//{tsv.tsv,ddi.xml}) using the locally built + * (tests/fixtures/surveys//{tsv.tsv,ddi.xml,ddi2xlsform.json}) using the locally built * library (dist/index.js). * * The per-entity snapshots under registry/entities/ pin *one question type* at a @@ -30,9 +30,8 @@ if (!fs.existsSync(entry)) { console.error('dist/index.js missing — run `npm run build` first.'); process.exit(1); } -const { XLSFormToTSVConverter, XLSLoader, buildDdiXml } = await import( - pathToFileURL(entry).href -); +const { XLSFormToTSVConverter, XLSLoader, buildDdiXml, ddiToXlsform } = + await import(pathToFileURL(entry).href); // Fixed so a re-bless on another day is not a diff (the snapshot test scrubs it // too, but keeping the file stable makes `git diff` mean something). @@ -96,7 +95,14 @@ for (const name of dirs) { }); fs.writeFileSync(path.join(dir, 'ddi.xml'), ddi); - console.log(`blessed ${name} → tsv.tsv + ddi.xml`); + // And back (#154): the form ddi2xlsform gives, which pyxform validates. + const back = ddiToXlsform(ddi, { onWarning: () => {} }); + fs.writeFileSync( + path.join(dir, 'ddi2xlsform.json'), + JSON.stringify(back, null, 2) + '\n', + ); + + console.log(`blessed ${name} → tsv.tsv + ddi.xml + ddi2xlsform.json`); written++; } catch (e) { // testA carries unimplemented types on purpose; a fixture that cannot be diff --git a/scripts/bless.sh b/scripts/bless.sh index 1d0e8fb..1bfaa71 100755 --- a/scripts/bless.sh +++ b/scripts/bless.sh @@ -5,7 +5,7 @@ # npm run bless -- tsv # registry/entities//tsv.tsv # npm run bless -- ddi # registry/entities//ddi.xml # npm run bless -- examples # meta.json + xlsform.xlsx (codegen) -# npm run bless -- surveys # tests/fixtures/surveys//{tsv.tsv,ddi.xml} +# npm run bless -- surveys # tests/fixtures/surveys//{tsv.tsv,ddi.xml,ddi2xlsform.json} # npm run bless -- responses# tests/live/limesurvey/expected/ (needs docker) # # Snapshots are a manual gate: default codegen never touches ddi.xml/tsv.tsv, so diff --git a/src/api.ts b/src/api.ts index a3c0d9d..cca547b 100644 --- a/src/api.ts +++ b/src/api.ts @@ -18,6 +18,11 @@ import type { ChoiceRow, SettingsRow, SurveyRow } from './xlsform/types.js'; import { XLSFormToTSVConverter } from './pipelines/xlsform2lstsv/index.js'; import { buildDdiXml } from './pipelines/xlsform2ddi/index.js'; import { lstsvToDdiXml } from './pipelines/lstsv2ddi/index.js'; +import { + ddiToXlsform as ddiToXlsformSheets, + type DdiToXlsformOptions, + type XlsformOutput, +} from './pipelines/ddi2xlsform/index.js'; import type { LstsvToDdiOptions } from './pipelines/lstsv2ddi/index.js'; /** An XLSForm: the `.xlsx` bytes, or its loaded sheets. */ @@ -157,6 +162,18 @@ export function xlsformToDdi( }); } +/** + * DDI-Codebook 2.5 XML (a codebook or a fragment) → XLSForm sheets + * `{ survey, choices, settings }` (#154). Never refuses DDI it can read: + * each field it can't supply is a warning. + */ +export function ddiToXlsform( + xml: string, + options: DdiToXlsformOptions = {}, +): XlsformOutput { + return ddiToXlsformSheets(xml, { onWarning: onceEach(options.onWarning) }); +} + /** LimeSurvey structure TSV → DDI-Codebook 2.5 XML. */ export function lstsvToDdi( tsv: string, diff --git a/src/cli.ts b/src/cli.ts index c0fdc0f..0cbeb86 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -8,6 +8,7 @@ import { ConversionError } from './diagnostics.js'; import { resolveFileChoices } from './fileChoices.js'; import { lstsvToDataCsv, lstsvToDdiXml } from './pipelines/lstsv2ddi/index.js'; import { lstsvToXlsform } from './pipelines/lstsv2xlsform/index.js'; +import { ddiToXlsform } from './pipelines/ddi2xlsform/index.js'; import type { Submission } from './pipelines/xlsform2ddi/index.js'; import type { XLSFormData } from './xlsform/types.js'; import { XLSValidator } from './xlsform/validate.js'; @@ -39,6 +40,7 @@ Commands: xlsform2ddi Convert an XLSForm (.xlsx) to a DDI-Codebook 2.5 XML lstsv2ddi Convert a LimeSurvey structure TSV to a DDI-Codebook 2.5 XML lstsv2xlsform Convert a LimeSurvey structure TSV to an XLSForm (.json) + ddi2xlsform Convert a DDI-Codebook 2.5 XML to an XLSForm (.json) Run "${PROG} --help" for command options. `, @@ -139,7 +141,7 @@ Arguments: Emits { survey, choices, settings } as JSON. Not a full reverse of xlsform2lstsv — see src/pipelines/lstsv2xlsform/README.md for scope and known lossy fields -(a select's list_name is synthesized; integer/decimal are indistinguishable). +(a select's list_name is synthesized). Options: -o, --output Write JSON to (default: stdout) @@ -465,6 +467,52 @@ function cmdLstsv2xlsform(argv: string[]): void { emit(json, values.output as string | undefined); } +function ddi2xlsformHelp(): void { + process.stdout.write( + `${PROG} ddi2xlsform — DDI-Codebook 2.5 XML → XLSForm (JSON) + +Usage: + ${PROG} ddi2xlsform [-o output.json] + +Arguments: + input.xml A DDI codebook, or a fragment (, , ) + +Emits { survey, choices, settings } as JSON. A codebook formtransform wrote +gives back its form; any other DDI is read as far as its standard elements +go, with a warning for each field it can't supply. See +src/pipelines/ddi2xlsform/README.md. + +Options: + -o, --output Write JSON to (default: stdout) + -h, --help Show this help +`, + ); +} + +function cmdDdi2xlsform(argv: string[]): void { + const { values, positionals } = parse(argv, { + output: { type: 'string', short: 'o' }, + help: { type: 'boolean', short: 'h', default: false }, + }); + + if (values.help) return ddi2xlsformHelp(); + + const bytes = readInput(positionals, ddi2xlsformHelp); + + let json: string; + try { + const xlsform = ddiToXlsform(bytes.toString('utf-8'), { + onWarning: (w) => + process.stderr.write(`${PROG}: warning: ${w.message}\n`), + }); + json = JSON.stringify(xlsform, null, 2) + '\n'; + } catch (err) { + return die(`conversion failed: ${(err as Error).message}`); + } + + emit(json, values.output as string | undefined); +} + async function main(): Promise { const [command, ...rest] = process.argv.slice(2); @@ -489,6 +537,9 @@ async function main(): Promise { case 'lstsv2xlsform': cmdLstsv2xlsform(rest); break; + case 'ddi2xlsform': + cmdDdi2xlsform(rest); + break; default: topHelp(); die(`unknown command: ${command}`); diff --git a/src/ddi/codebook.ts b/src/ddi/codebook.ts index 63c9396..86ae54f 100644 --- a/src/ddi/codebook.ts +++ b/src/ddi/codebook.ts @@ -359,6 +359,7 @@ function emitOtherPattern( dataDscr: XmlElement, p: OtherPattern, ctx: LogicContext, + notes: InlineNotes, ): void { const { base, otherVar } = p; const label = base.label; @@ -382,8 +383,9 @@ function emitOtherPattern( }); localizedChild(parentEl, 'txt', label, labelTranslations); parentEl.textChild('concept', label); - // No var has the question's name: the parent group carries its logic. - addGroupLogic(parentEl, base, ctx); + // No var has the question's name: the parent group carries its logic, + // and a note before it. + addGroupLogic(parentEl, base, ctx, { note: notes, name: baseName }); const childEl = dataDscr.child('varGrp', { ID: childId, @@ -410,6 +412,7 @@ function emitOtherPatternVars( dataDscr: XmlElement, p: OtherPattern, ctx: LogicContext, + notes: InlineNotes, ): void { const { base, otherVar } = p; const baseName = base.name; @@ -426,6 +429,10 @@ function emitOtherPatternVars( label: base.label, varType: base.type, choices: base.choices, + opts: { + preQTxt: notes.text[baseName] ?? '', + preQTxtTranslations: notes.translations[baseName], + }, hint: base.hint, guidanceHint: base.guidanceHint, ...specTranslations(base), @@ -449,7 +456,8 @@ function emitOtherPatternVars( /** Optional settings that shape study-level metadata. */ export interface DdiSettings { - form_title?: string; + /** A plain title, or one per language (`{ en: …, es: … }`). */ + form_title?: string | Record; /** Study ID (`IDNo`); `id_string` is Kobo's older name for it. */ form_id?: string | number; id_string?: string | number; @@ -530,7 +538,7 @@ export function splitDataVars(dataVars: Variable[]): DataVarBuckets { function addStudyDscr( root: XmlElement, settings: DdiSettings, - title: string, + title: StudyTitle, prodDate: string, orphanNotes: Variable[], ): void { @@ -538,7 +546,11 @@ function addStudyDscr( const citation = stdy.child('citation'); const titlStmt = citation.child('titlStmt'); - titlStmt.textChild('titl', title); + titlStmt.textChild('titl', title.title); + // A title in another language is DDI's parallel title. + for (const [lang, text] of Object.entries(title.parallel)) { + titlStmt.textChild('parTitl', text, { 'xml:lang': lang }); + } const studyId = settings.form_id ?? settings.id_string; if (studyId) titlStmt.textChild('IDNo', String(studyId)); @@ -557,6 +569,39 @@ function addStudyDscr( addSettingNotes(stdy, settings); } +interface StudyTitle { + title: string; + /** `form_title` in the form's other languages, by tag. */ + parallel: Translations; +} + +/** + * The study title: `assetName`, else `form_title` (a `{ lang: text }` one in + * the base language, the others as parallel titles), else `Untitled`. + */ +function studyTitle(assetName: string, settings: DdiSettings): StudyTitle { + const raw = settings.form_title; + if (assetName.trim()) return { title: assetName.trim(), parallel: {} }; + if (raw === null || typeof raw !== 'object') { + return { title: String(raw ?? '').trim() || 'Untitled', parallel: {} }; + } + const base = + typeof settings.default_language === 'string' + ? languageTagOf(settings.default_language) + : null; + const byTag = Object.entries(raw as Record) + .map(([key, v]): [string, string] => [ + languageTagOf(key) ?? key, + typeof v === 'string' ? v.trim() : '', + ]) + .filter(([, v]) => v); + const main = byTag.find(([tag]) => tag === base) ?? byTag[0]; + return { + title: main?.[1] ?? 'Untitled', + parallel: Object.fromEntries(byTag.filter((e) => e !== main)), + }; +} + /** Emit `` with `caseQnty` set to the submissions count. */ function addFileDscr( root: XmlElement, @@ -723,7 +768,7 @@ function addVarGroups( } for (const p of otherPatterns.values()) { - emitOtherPattern(dataDscr, p, ctx); + emitOtherPattern(dataDscr, p, ctx, notes); } } @@ -747,7 +792,7 @@ function addVars( ); } } else if (unit.kind === 'other') { - emitOtherPatternVars(dataDscr, unit.p, ctx); + emitOtherPatternVars(dataDscr, unit.p, ctx, notes); } else if (unit.grid) { const group = getGroupLabel(dataVars, unit.grid); // preQTxt must equal the group's txt (Schematron); the member's own @@ -818,8 +863,7 @@ export function buildDdiCodebook( translations: classified.inlinePreqtxtTranslations, }; - const title = - assetName.trim() || String(settings.form_title ?? '').trim() || 'Untitled'; + const title = studyTitle(assetName, settings); const root = new XmlElement('codeBook'); root.setAttr('xmlns', NS); diff --git a/src/diagnostics.ts b/src/diagnostics.ts index 56961d6..d75f5c6 100644 --- a/src/diagnostics.ts +++ b/src/diagnostics.ts @@ -56,6 +56,11 @@ export type DiagnosticCode = // LimeSurvey TSV (reverse) | 'lstsv-outside-subset' | 'em-unsupported' + // DDI (reverse, #154) + | 'ddi-invalid' + | 'ddi-field-missing' + | 'ddi-reference-outside' + | 'ddi-type-unknown' // responses | 'responses-invalid' | 'response-ambiguous' diff --git a/src/index.ts b/src/index.ts index 8b53fb4..ffa9869 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1,5 +1,10 @@ // ── Conversions (one function per direction) ─────────────────────────── -export { xlsformToLstsv, xlsformToDdi, lstsvToDdi } from './api.js'; +export { + xlsformToLstsv, + xlsformToDdi, + lstsvToDdi, + ddiToXlsform, +} from './api.js'; export type { XlsformSource, XlsformToLstsvOptions, @@ -11,6 +16,7 @@ export type { XlsformOutput, } from './pipelines/lstsv2xlsform/index.js'; export type { LstsvToDdiOptions } from './pipelines/lstsv2ddi/index.js'; +export type { DdiToXlsformOptions } from './pipelines/ddi2xlsform/index.js'; // ── Diagnostics ──────────────────────────────────────────────────────── export { ConversionError, consoleWarning } from './diagnostics.js'; diff --git a/src/instrument/fromDdi.ts b/src/instrument/fromDdi.ts new file mode 100644 index 0000000..089fdc0 --- /dev/null +++ b/src/instrument/fromDdi.ts @@ -0,0 +1,669 @@ +/** + * DDI Codebook 2.5 → {@link Instrument} (#154): the reverse of the DDI + * emitter, reading standard DDI first and the `cdl:` notes where DDI has no + * element (`convention:logicMapping` `ddiEncoding`, `convention:ddiFields`). + * + * A codebook formtransform wrote gives back its form, up to the losses + * `src/pipelines/ddi2xlsform/README.md` lists. Any other DDI (hand-written, + * another tool's, a fragment) is read as far as its standard elements go: + * never refused, with a warning for each field it can't supply. + * + * Input may be a whole ``, a ``, or bare `` / + * `` elements. + */ +import conventions from '../generated/conventions.js'; +import { + ConversionError, + warning, + type WarningHandler, +} from '../diagnostics.js'; +import { GRID_APPEARANCE } from '../conventions/grid.js'; +import { EXCLUSIVE_RULE } from '../conventions/exclusive.js'; +import { OTHER_CODE, otherLabelFor } from '../conventions/other.js'; +import { fromFileTypeFor } from '../conventions/fromFile.js'; +import { TYPE_MAPPINGS } from '../generated/TypeMappings.js'; +import { parseParameters } from '../utils/parameters.js'; +import { + childNamed, + childrenNamed, + parseXml, + textContent, + type XmlNode, +} from '../utils/xmlParse.js'; +import type { + GroupItem, + Instrument, + InstrumentChoice, + Item, + QuestionItem, + Text, +} from './types.js'; + +const LOGIC = conventions.conventions.logicMapping.ddiEncoding.notes; +const FIELDS = conventions.conventions.ddiFields.notes; + +/** What reading one codebook needs. */ +interface ReadState { + /** The codebook's base language: `codeBook/@xml:lang`, else untagged. */ + base: string; + /** Every language seen, base first. */ + languages: string[]; + vars: Map; + groups: Map; + /** Section / grid a `var` or `varGrp` is directly in. */ + parent: Map; + /** multipleResp / other `varGrp` a `var` belongs to. */ + owner: Map; + lists: Record; + /** Choice set (as a key) → its list's name. */ + listByKey: Map; + names: Set; + onWarning?: WarningHandler; +} + +// ── texts ──────────────────────────────────────────────────────────────── + +/** The elements' texts by language: untagged is the base. */ +function texts(nodes: XmlNode[], state: ReadState): Text { + const out: Text = {}; + for (const node of nodes) { + const lang = node.attrs['xml:lang'] ?? state.base; + const value = textContent(node).trim(); + if (!value || lang in out) continue; + out[lang] = value; + if (!state.languages.includes(lang)) state.languages.push(lang); + } + return out; +} + +const childTexts = (node: XmlNode | undefined, name: string, s: ReadState) => + node ? texts(childrenNamed(node, name), s) : {}; + +/** The typed notes of one `cdl:` type. */ +function notesOf(node: XmlNode, type: string): XmlNode[] { + return childrenNamed(node, 'notes').filter((n) => n.attrs['type'] === type); +} + +/** A typed note's base-language text (`''` when absent). */ +function noteText(node: XmlNode, type: string, state: ReadState): string { + const notes = notesOf(node, type); + const base = notes.find( + (n) => (n.attrs['xml:lang'] ?? state.base) === state.base, + ); + const note = base ?? notes[0]; + return note ? textContent(note).trim() : ''; +} + +const ids = (value: string | undefined) => + (value ?? '').split(/\s+/).filter(Boolean); + +// ── structure ──────────────────────────────────────────────────────────── + +/** The `var` / `varGrp` elements, wherever the input put them. */ +function dataElements(roots: XmlNode[]): XmlNode[] { + const out: XmlNode[] = []; + const walk = (node: XmlNode) => { + if (node.name === 'var' || node.name === 'varGrp') out.push(node); + else if (node.name === 'codeBook' || node.name === 'dataDscr') { + node.children.forEach(walk); + } + }; + roots.forEach(walk); + return out; +} + +const STRUCTURE_TYPES = new Set(['section', 'grid']); + +/** Index the groups and who contains whom. */ +function indexStructure(elements: XmlNode[], state: ReadState): void { + for (const el of elements) { + const id = el.attrs['ID']; + if (!id) continue; + if (el.name === 'var') state.vars.set(id, el); + else state.groups.set(id, el); + } + for (const [id, grp] of state.groups) { + const type = grp.attrs['type'] ?? ''; + const members = [...ids(grp.attrs['var']), ...ids(grp.attrs['varGrp'])]; + for (const member of members) { + if (!state.vars.has(member) && !state.groups.has(member)) { + state.onWarning?.( + warning( + 'ddi-reference-outside', + `varGrp ${id} refers to ${member}, which is not in the input`, + ), + ); + } + if (STRUCTURE_TYPES.has(type)) state.parent.set(member, id); + else if (type === 'multipleResp' || type === 'other') { + state.owner.set(member, id); + } + } + } +} + +/** The groups a `var` / `varGrp` is in, outermost first. */ +function chainOf(id: string, state: ReadState): string[] { + const chain: string[] = []; + for (let at = state.parent.get(id); at; at = state.parent.get(at)) { + if (chain.includes(at)) break; + chain.unshift(at); + } + return chain; +} + +// ── choices ────────────────────────────────────────────────────────────── + +/** + * A choice set's list: an identical set already read shares its list (the + * list's own name is not in the DDI), else a new one named `preferred`. + */ +function listFor( + choices: InstrumentChoice[], + preferred: string, + state: ReadState, +): string { + const key = JSON.stringify( + choices.map((c) => [c.name, c.label, c.row[EXCLUSIVE_RULE.choicesColumn]]), + ); + const known = state.listByKey.get(key); + if (known) return known; + let name = preferred; + for (let i = 2; name in state.lists; i++) name = `${preferred}_${i}`; + state.lists[name] = choices; + state.listByKey.set(key, name); + return name; +} + +function categories(v: XmlNode, state: ReadState): InstrumentChoice[] { + return childrenNamed(v, 'catgry').map((c) => ({ + name: textContent(childNamed(c, 'catValu') ?? c).trim(), + label: childTexts(c, 'labl', state), + row: {}, + })); +} + +// ── questions ──────────────────────────────────────────────────────────── + +function emptyQuestion(name: string): QuestionItem { + return { + kind: 'question', + name, + label: {}, + hint: {}, + relevant: '', + appearance: '', + row: {}, + type: 'text', + rawType: 'text', + list: '', + file: '', + orOther: false, + guidanceHint: {}, + constraint: '', + constraintMessage: {}, + required: false, + default: '', + parameters: '', + }; +} + +/** The logic and field notes on a `var` or a question's `varGrp`. */ +function readNotes(q: QuestionItem, node: XmlNode, state: ReadState): void { + q.relevant = noteText(node, LOGIC.relevant.type, state); + q.constraint = noteText(node, LOGIC.constraint.type, state); + q.constraintMessage = texts( + notesOf(node, LOGIC.constraint_message.type), + state, + ); + q.required = noteText(node, LOGIC.required.type, state) !== ''; + q.default = noteText(node, FIELDS.default.type, state); + q.appearance = noteText(node, FIELDS.appearance.type, state); + q.parameters = noteText(node, FIELDS.parameters.type, state); +} + +/** `qstn`'s texts: label, hint, guidance hint. */ +function readQstn(q: QuestionItem, v: XmlNode, state: ReadState): void { + const qstn = childNamed(v, 'qstn'); + q.label = childTexts(qstn, 'qstnLit', state); + q.hint = childTexts(qstn, 'postQTxt', state); + q.guidanceHint = childTexts(qstn, 'ivuInstr', state); +} + +/** A range's `valrng` bounds back into its `parameters`, where not defaults. */ +function rangeParameters(v: XmlNode, notes: string): string { + const range = childNamed(childNamed(v, 'valrng') ?? v, 'range'); + const defaults = TYPE_MAPPINGS['range']?.parameters ?? {}; + const own = parseParameters(notes); + const bounds = [ + ['start', range?.attrs['min']], + ['end', range?.attrs['max']], + ] + .filter(([key, value]) => value && value !== defaults[key!] && !own[key!]) + .map(([key, value]) => `${key}=${value}`); + return [...bounds, notes].filter(Boolean).join(' '); +} + +/** A number: `dcml="0"` is an integer; bounds with no constraint, a range. */ +function numericType(v: XmlNode): string { + if (v.attrs['dcml'] === '0') return 'integer'; + const bounded = !!childNamed(v, 'valrng'); + return bounded && !notesOf(v, LOGIC.constraint.type).length + ? 'range' + : 'decimal'; +} + +/** Text, or the date / time its `varFormat/@category` says. */ +function textType(v: XmlNode): string { + const category = childNamed(v, 'varFormat')?.attrs['category']; + return category === 'date' || category === 'time' ? category : 'text'; +} + +/** A plain `var`'s XLSForm type, from its standard DDI. */ +function varType(v: XmlNode, state: ReadState): string { + const domain = childNamed(v, 'qstn')?.attrs['responseDomainType'] ?? 'text'; + const vocab = childNamed(v, 'concept')?.attrs['vocab']; + const select = (base: string) => (vocab ? fromFileTypeFor(base) : base); + if (domain === 'category') return select('select_one'); + if (domain === 'multiple') return select('select_multiple'); + if (domain === 'numeric') return numericType(v); + if (domain !== 'text') { + state.onWarning?.( + warning( + 'ddi-type-unknown', + `responseDomainType "${domain}" of ${v.attrs['name']} is not one formtransform writes; read as text`, + v.attrs['name'], + ), + ); + } + return textType(v); +} + +/** One plain `var` as a question. */ +function plainQuestion( + v: XmlNode, + state: ReadState, + listName = v.attrs['name'] ?? '', +): QuestionItem { + const q = emptyQuestion(v.attrs['name'] ?? v.attrs['ID'] ?? ''); + readQstn(q, v, state); + readNotes(q, v, state); + q.type = varType(v, state); + const vocab = childNamed(v, 'concept')?.attrs['vocab']; + if (vocab) { + q.file = `${vocab}.csv`; + } else if (q.type === 'select_one' || q.type === 'select_multiple') { + q.list = listFor(categories(v, state), listName, state); + } + if (q.type === 'range') q.parameters = rangeParameters(v, q.parameters); + q.rawType = [q.type, q.file || q.list].filter(Boolean).join(' '); + return q; +} + +/** Exclusive codes of a select_multiple (`cdl:exclusive`). */ +function exclusive(grp: XmlNode, state: ReadState): Set { + return new Set(ids(noteText(grp, FIELDS.exclusive.type, state))); +} + +/** + * A select_multiple from its `varGrp` (`multipleResp`, or the semi-open + * pair's `other` around it): label and logic from the group, choices and + * hints from its binary `var`s. + */ +function multiQuestion( + grp: XmlNode, + binaries: XmlNode[], + withOther: boolean, + state: ReadState, +): QuestionItem { + const name = grp.attrs['name'] ?? ''; + const q = emptyQuestion(name); + q.type = 'select_multiple'; + q.label = childTexts(grp, 'txt', state); + readNotes(q, grp, state); + const first = binaries[0]; + if (first) { + const qstn = childNamed(first, 'qstn'); + q.hint = childTexts(qstn, 'postQTxt', state); + q.guidanceHint = childTexts(qstn, 'ivuInstr', state); + } + const excl = exclusive(grp, state); + const choices: InstrumentChoice[] = binaries.map((b) => { + const code = (b.attrs['name'] ?? '').slice(name.length + 1); + return { + name: code, + label: childTexts(childNamed(b, 'qstn'), 'qstnLit', state), + row: excl.has(code) + ? { [EXCLUSIVE_RULE.choicesColumn]: EXCLUSIVE_RULE.trueValues[0] } + : {}, + }; + }); + if (withOther) { + // Its binary isn't written; the label is the convention's. + const label: Text = {}; + for (const lang of state.languages) + label[lang] = otherLabelFor(lang || 'en'); + choices.push({ name: OTHER_CODE, label, row: {} }); + } + q.list = listFor(choices, name, state); + q.rawType = `select_multiple ${q.list}`; + return q; +} + +// ── notes ──────────────────────────────────────────────────────────────── + +/** A note row, whose name the DDI doesn't keep. */ +function noteItem(label: Text, near: string, state: ReadState): QuestionItem { + let name = `${near}_note`; + for (let i = 2; state.names.has(name); i++) name = `${near}_note_${i}`; + state.names.add(name); + return { ...emptyQuestion(name), type: 'note', rawType: 'note', label }; +} + +/** A lead-in note kept as `preQTxt` (standalone) or an untyped group note. */ +function leadIn(label: Text, near: string, state: ReadState): QuestionItem[] { + return Object.keys(label).length ? [noteItem(label, near, state)] : []; +} + +const untypedNotes = (node: XmlNode, state: ReadState) => + texts( + childrenNamed(node, 'notes').filter((n) => !n.attrs['type']), + state, + ); + +// ── tree ───────────────────────────────────────────────────────────────── + +/** A question in survey order, with the element that places it in groups. */ +interface Placed { + items: QuestionItem[]; + /** The `var` / `varGrp` ID its enclosing groups contain. */ + anchor: string; +} + +/** The `var`s an element's `@var` lists that are in the input. */ +function varsOf(grp: XmlNode, state: ReadState): XmlNode[] { + return ids(grp.attrs['var']) + .map((m) => state.vars.get(m)) + .filter((m): m is XmlNode => !!m); +} + +/** A `var`'s lead-in note (its `preQTxt`) and the question itself. */ +function withLeadIn(v: XmlNode, state: ReadState): QuestionItem[] { + const note = childTexts(childNamed(v, 'qstn'), 'preQTxt', state); + return [ + ...leadIn(note, v.attrs['name'] ?? '', state), + plainQuestion(v, state), + ]; +} + +/** A `var` in no multipleResp / other group: a standalone or grid question. */ +function placePlain(id: string, v: XmlNode, state: ReadState): Placed { + const parentId = state.parent.get(id) ?? ''; + // A grid member's preQTxt is the grid's text, not a note. + if (state.groups.get(parentId)?.attrs['type'] !== 'grid') { + return { items: withLeadIn(v, state), anchor: id }; + } + return { + items: [plainQuestion(v, state, gridName(parentId, state))], + anchor: id, + }; +} + +/** A select_multiple, and with an `other` pair around it, its companion. */ +function placeMulti(multiId: string, multi: XmlNode, state: ReadState): Placed { + const pairId = state.owner.get(multiId); + const pair = pairId ? state.groups.get(pairId) : undefined; + const q = multiQuestion(pair ?? multi, varsOf(multi, state), !!pair, state); + const lead = untypedNotes(pair ?? multi, state); + const items = [...leadIn(lead, q.name, state), q]; + if (pair) + items.push(...varsOf(pair, state).map((m) => plainQuestion(m, state))); + return { items, anchor: pairId ?? multiId }; +} + +/** The question(s) one `var` starts, or none if an earlier one covered it. */ +function placeVar( + id: string, + v: XmlNode, + seen: Set, + state: ReadState, +): Placed | null { + const ownerId = state.owner.get(id); + const owner = ownerId ? state.groups.get(ownerId) : undefined; + if (!owner || !ownerId) return placePlain(id, v, state); + // A multi pair's companion is in its `other` group, next to the multi. + const multiId = + owner.attrs['type'] === 'multipleResp' + ? ownerId + : ids(owner.attrs['varGrp']).find((g) => state.groups.has(g)); + const key = multiId ? (state.owner.get(multiId) ?? multiId) : ownerId; + if (seen.has(key)) return null; + seen.add(key); + if (multiId) return placeMulti(multiId, state.groups.get(multiId)!, state); + // A semi-open select_one: its `other` group holds the select and its text. + return { + items: varsOf(owner, state).flatMap((m) => withLeadIn(m, state)), + anchor: ownerId, + }; +} + +/** A grid's own name (its path's last part), its members' list name. */ +function gridName(id: string, state: ReadState): string { + const path = state.groups.get(id)?.attrs['name'] ?? id; + return path.slice(path.lastIndexOf('/') + 1); +} + +/** A section / grid `varGrp` as an (empty) group item. */ +function groupItem(id: string, state: ReadState): GroupItem { + const grp = state.groups.get(id)!; + const grid = grp.attrs['type'] === 'grid'; + const lead = grid + ? leadIn(untypedNotes(grp, state), gridName(id, state), state) + : []; + return { + kind: 'group', + name: gridName(id, state), + label: childTexts(grp, 'txt', state), + hint: texts(notesOf(grp, FIELDS.hint.type), state), + relevant: noteText(grp, LOGIC.relevant.type, state), + appearance: + noteText(grp, FIELDS.appearance.type, state) || + (grid ? GRID_APPEARANCE : ''), + row: {}, + children: lead, + closed: true, + }; +} + +/** Seat each placed question in its groups, opening them as they come. */ +function buildTree(placed: Placed[], state: ReadState): Item[] { + const body: Item[] = []; + const open: Array<{ id: string; item: GroupItem }> = []; + for (const { items, anchor } of placed) { + const chain = chainOf(anchor, state); + let depth = 0; + while (depth < open.length && open[depth].id === chain[depth]) depth++; + open.length = depth; + for (const id of chain.slice(depth)) { + const item = groupItem(id, state); + (open.length ? open[open.length - 1].item.children : body).push(item); + open.push({ id, item }); + } + (open.length ? open[open.length - 1].item.children : body).push(...items); + } + return body; +} + +/** `var`s in survey order: element order, `qstn/@seqNo` where given. */ +function inSurveyOrder(vars: Map): Array<[string, XmlNode]> { + const seq = (v: XmlNode) => + Number(childNamed(v, 'qstn')?.attrs['seqNo'] ?? Number.NaN); + const entries = [...vars.entries()]; + if (!entries.every(([, v]) => Number.isFinite(seq(v)))) return entries; + return entries + .map((e, i) => ({ e, i })) + .sort((a, b) => seq(a.e[1]) - seq(b.e[1]) || a.i - b.i) + .map(({ e }) => e); +} + +// ── study ──────────────────────────────────────────────────────────────── + +/** The text at a `stdyDscr/citation` path (`titlStmt/titl`), `''` if none. */ +function citationText(stdy: XmlNode, path: string[]): string { + let at: XmlNode | undefined = childNamed(stdy, 'citation'); + for (const name of path) at = at && childNamed(at, name); + return at ? textContent(at).trim() : ''; +} + +function readSettings( + root: XmlNode | undefined, + state: ReadState, +): Record { + const stdy = root && childNamed(root, 'stdyDscr'); + if (!stdy) return {}; + const fields: Array<[string, string]> = [ + ['form_title', citationText(stdy, ['titlStmt', 'titl'])], + ['form_id', citationText(stdy, ['titlStmt', 'IDNo'])], + ['version', citationText(stdy, ['verStmt', 'version'])], + ['default_language', state.base], + ]; + const out: Record = Object.fromEntries( + fields.filter(([, value]) => value && value !== 'Untitled'), + ); + // Parallel titles: form_title in the other languages. + const titlStmt = childNamed(childNamed(stdy, 'citation') ?? stdy, 'titlStmt'); + const parallel = titlStmt ? childrenNamed(titlStmt, 'parTitl') : []; + if (parallel.length && typeof out['form_title'] === 'string') { + out['form_title'] = { + ...texts(parallel, state), + [state.base]: out['form_title'], + }; + } + for (const note of notesOf(stdy, FIELDS.setting.type)) { + const key = note.attrs['subject']; + if (key) out[key] = textContent(note).trim(); + } + return out; +} + +/** The study's notes (`type="instruction"`): notes with no question after them. */ +function orphanNotes( + root: XmlNode | undefined, + state: ReadState, +): QuestionItem[] { + const stdy = root && childNamed(root, 'stdyDscr'); + if (!stdy) return []; + const bySubject = new Map(); + for (const note of childrenNamed(stdy, 'notes')) { + if (note.attrs['type'] !== 'instruction') continue; + const subject = note.attrs['subject'] ?? ''; + bySubject.set(subject, [...(bySubject.get(subject) ?? []), note]); + } + return [...bySubject].map(([subject, notes]) => { + state.names.add(subject); + return { + ...emptyQuestion(subject), + type: 'note', + rawType: 'note', + label: texts(notes, state), + }; + }); +} + +// ── provenance ─────────────────────────────────────────────────────────── + +/** Fields only a CDL codebook carries, reported once each when absent. */ +const CDL_ONLY = [ + 'question order (qstn/@seqNo)', + 'skip logic (cdl:relevant)', + 'validation (cdl:constraint)', + 'required (cdl:required)', + 'default (cdl:default)', + 'appearance (cdl:appearance)', +]; + +function isCdl(elements: XmlNode[]): boolean { + return elements.some( + (el) => + childNamed(el, 'qstn')?.attrs['seqNo'] !== undefined || + childrenNamed(el, 'notes').some((n) => + (n.attrs['type'] ?? '').startsWith('cdl:'), + ), + ); +} + +function warnNotCdl(onWarning: WarningHandler | undefined): void { + for (const field of CDL_ONLY) { + onWarning?.( + warning( + 'ddi-field-missing', + `Not a CDL codebook: no ${field}; the form gets none (formtransform#155)`, + ), + ); + } +} + +/** The languages seen, base first; `['']` for an untagged one-language form. */ +function languagesOf(state: ReadState): string[] { + const languages = state.languages.filter((l, i, all) => all.indexOf(l) === i); + return state.base || languages.length > 1 ? languages : ['']; +} + +// ── entry point ────────────────────────────────────────────────────────── + +export interface DdiParseOptions { + /** Receives what the DDI can't supply (not a CDL codebook, a fragment). */ + onWarning?: WarningHandler; +} + +/** + * Parse DDI Codebook 2.5 XML (a whole codebook or a fragment) into an + * Instrument. Throws `ddi-invalid` when the XML is not well-formed or holds + * neither a `codeBook` nor a `var`. + */ +export function instrumentFromDdi( + xml: string, + options: DdiParseOptions = {}, +): Instrument { + const roots = parseXml(xml); + const root = roots.find((r) => r.name === 'codeBook'); + const base = root?.attrs['xml:lang'] ?? ''; + const state: ReadState = { + base, + languages: [base], + vars: new Map(), + groups: new Map(), + parent: new Map(), + owner: new Map(), + lists: {}, + listByKey: new Map(), + names: new Set(), + onWarning: options.onWarning, + }; + const elements = dataElements(roots); + indexStructure(elements, state); + if (!root && state.vars.size === 0) { + throw new ConversionError( + 'ddi-invalid', + 'The input holds no and no .', + ); + } + if (!isCdl(elements)) warnNotCdl(options.onWarning); + for (const [, v] of state.vars) state.names.add(v.attrs['name'] ?? ''); + + const seen = new Set(); + const placed: Placed[] = []; + for (const [id, v] of inSurveyOrder(state.vars)) { + const p = placeVar(id, v, seen, state); + if (p) placed.push(p); + } + const settings = readSettings(root, state); + const body = [...buildTree(placed, state), ...orphanNotes(root, state)]; + return { + languages: languagesOf(state), + ...(base ? { defaultLanguage: base } : {}), + settings, + lists: state.lists, + body, + }; +} diff --git a/src/instrument/fromLstsv.ts b/src/instrument/fromLstsv.ts index 2946a02..3f96f01 100644 --- a/src/instrument/fromLstsv.ts +++ b/src/instrument/fromLstsv.ts @@ -85,6 +85,8 @@ interface ParseState { lists: Record; /** Selects with `other=Y`, by name: their `other` text is the companion. */ otherSelects: Set; + /** Group names given so far, to keep them unique. */ + groupNames: Set; } const text = (state: ParseState, key: string): Text => ({ @@ -202,6 +204,31 @@ function addChoice( }); } +/** A name XLSForm (and a DDI `xs:ID`) accepts. */ +const XLSFORM_NAME = /^[A-Za-z_][A-Za-z0-9._-]*$/; + +/** + * A plain group's name: LimeSurvey's group name is its title, so one that is + * no XLSForm name (`Über Sie`) is slugified (`bersie`), made unique. + */ +function groupName(title: string, state: ParseState): string { + let name = title; + if (!XLSFORM_NAME.test(name)) { + const slug = title + .toLowerCase() + .replace(/[^a-z0-9]+/g, ' ') + .trim() + .split(/\s+/) + .slice(0, 4) + .join(''); + name = slug && !/^[0-9]/.test(slug) ? slug : `group${slug}`; + } + const base = name; + for (let i = 2; state.groupNames.has(name); i++) name = `${base}_${i}`; + state.groupNames.add(name); + return name; +} + /** Walk the base-language rows into groups, questions and choice lists. */ function parseBody(baseRows: Row[], state: ParseState): Item[] { const body: Item[] = []; @@ -217,7 +244,7 @@ function parseBody(baseRows: Row[], state: ParseState): Item[] { group = { kind: 'group', ...emptyItem(row), - name: cell(row, 'name'), + name: groupName(cell(row, 'name'), state), label: text(state, `label:G:${cell(row, 'type/scale')}`), hint: text(state, `hint:G:${cell(row, 'type/scale')}`), children: [], @@ -395,6 +422,7 @@ export function instrumentFromLstsv( tr: collectTranslations(rows), lists: {}, otherSelects: new Set(), + groupNames: new Set(), }; const baseRows = rows.filter( (r) => diff --git a/src/pipelines/README.md b/src/pipelines/README.md index f5bb763..e28bd86 100644 --- a/src/pipelines/README.md +++ b/src/pipelines/README.md @@ -50,15 +50,16 @@ then hands the rows to the same `buildDataCsv`: These LimeSurvey quirks stay in `lstsv2ddi/`; the shared emitter and the Kobo path never guess at them. -## Why there is no `ddi2xlsform` or `ddi2lstsv` +## The way back from DDI (`ddi2xlsform`) -Not yet. **DDI is the terminus of the pipeline graph for now**: it describes a -*dataset*, not an *instrument*, and a reverse path needs the whole instrument. -The plan to carry all of it is #155 (#151–#154). +A codebook describes a *dataset*, not an *instrument*, but one formtransform +wrote carries the whole instrument too (#155, #151–#153). `ddi2xlsform` +(#154) reads it back: [`ddi2xlsform/README.md`](ddi2xlsform/README.md) has +the mapping, the input it accepts and the known losses. -The canonical `Variable` (`src/ddi/types.ts`) is what survives an emit: `name`, -`type`, `label`, group path/label/appearance, `listName`, `vocab`, `choices`, -and since #151–#153 the rest of the form: +What a CDL codebook carries beyond the canonical `Variable` +(`src/ddi/types.ts`: `name`, `type`, `label`, group path/label/appearance, +`listName`, `vocab`, `choices`): - **`relevant`, `constraint`, `constraint_message`, `required`.** DDI Codebook 2.5 has no expression syntax, so each is written twice: readable @@ -76,25 +77,12 @@ and since #151–#153 the rest of the form: `varFormat/@category`, `var/@dcml`, `valrng`, `qstn/@seqNo`, `qstn/backward`), else a typed note. `convention:ddiFields` ([`registry/conventions/ddiFields.jsonld`](../../registry/conventions/ddiFields.jsonld)) - maps every model field, or names it a loss: `calculation` and other - unlifted columns, a list's name, the `or_other` shorthand as such. - -Compare `lstsv2xlsform`, which *is* implemented: a LimeSurvey structure TSV -carries `relevance`, `em_validation_q`, `mandatory`, `default` and the `!`/`T` -type overrides. It is a form definition in a different dialect, so reversing it -is a translation problem. Reversing a codebook without those pieces is a -*reconstruction* problem, and they cannot be inferred from it. - -The failure mode matters more than the missing feature. A `ddi2xlsform` that -reads a codebook without the CDL notes (someone else's, or one from before -formtransform#151–#153) emits a survey that looks correct and behaves wrongly. Silently producing a broken instrument is worse -than declining to produce one — the same reasoning behind -`validateLstsvSubset` rejecting out-of-subset input rather than guessing at it. - -**If you need this anyway**, the honest shape is a *skeleton* generator: variable -names, labels, types, choice lists and group structure out of someone else's -codebook, as a starting point for authoring. That is a legitimate tool, but it -must never be described or tested as a round-trip, and the reconstructed form -must be reviewed before deployment. Route it as DDI → XLSForm → `xlsform2lstsv`; -a separate `ddi2lstsv` earns nothing but a second lossy reconstructor to keep in -sync. + maps every model field, or names it a loss. + +DDI formtransform didn't write (a hand-written seed study, another tool's +codebook) has none of the `cdl:` notes. `ddi2xlsform` still converts it, as a +skeleton: names, labels, types, choices and groups, with a `ddi-field-missing` +warning for each field it can't supply. Such a form must be reviewed before it +is deployed: it has no skip logic, validation or required answers unless +someone adds them. There is no `ddi2lstsv`; DDI → XLSForm → `xlsform2lstsv` +does it without a second reconstructor to keep in sync. diff --git a/src/pipelines/ddi2xlsform/README.md b/src/pipelines/ddi2xlsform/README.md new file mode 100644 index 0000000..4f3a468 --- /dev/null +++ b/src/pipelines/ddi2xlsform/README.md @@ -0,0 +1,85 @@ +# ddi2xlsform + +DDI Codebook 2.5 → XLSForm sheets `{ survey, choices, settings }` (#154). + +```ts +import { ddiToXlsform } from '@correlaid/formtransform'; + +const { survey, choices, settings } = ddiToXlsform(xml, { + onWarning: (w) => console.warn(w.code, w.message), +}); +``` + +CLI: `formtransform ddi2xlsform codebook.xml -o form.json`. + +The XML is parsed into the Instrument (`src/instrument/fromDdi.ts`, with the +dependency-free reader `src/utils/xmlParse.ts`), then written by the XLSForm +emitter `lstsv2xlsform` uses too (`src/xlsform/fromInstrument.ts`). + +## What it reads + +Standard DDI first, `cdl:` notes where DDI has no element +(`convention:ddiFields`, `convention:logicMapping` `ddiEncoding`): + +| XLSForm | DDI | +|---|---| +| type | `qstn/@responseDomainType`; `varFormat/@category` (date, time); `var/@dcml="0"` (integer); `valrng` without a constraint (range); `concept/@vocab` (`select_*_from_file`) | +| label / hint / guidance_hint | `qstnLit` / `postQTxt` / `ivuInstr`, every `xml:lang` | +| a note row before a question | its `preQTxt` (a grid member's is the grid's text) or its group's untyped `notes` | +| choices | `catgry` (`catValu`, `labl`); a select_multiple's binary `var`s | +| groups | `varGrp type="section"` / `"grid"`, nested by `@varGrp`; label `txt`, hint `cdl:hint` | +| order | `var` order, `qstn/@seqNo` | +| relevant, constraint, constraint_message, required | `cdl:relevant`, `cdl:constraint`, `cdl:constraint_message`, `cdl:required` | +| default, appearance, parameters | `cdl:default`, `cdl:appearance`, `cdl:parameters`; a range's `valrng/range` | +| exclusive | `cdl:exclusive` on the select_multiple's `varGrp` | +| settings | `titl` (+ `parTitl` per language), `IDNo`, `verStmt/version`, `codeBook/@xml:lang`, `cdl:setting` | + +## Input it accepts + +- **A codebook formtransform wrote** gives back its form, up to the losses + below. `tests/ts/contract/ddiRoundtrip.test.ts` checks every fixture on the + Instrument model, and `ddiRoundtripGenerated.test.ts` checks random forms + (fast-check). Every survey's output is pinned as `ddi2xlsform.json` and + validated with pyxform. +- **Any other DDI** is never refused. It is read as far as its standard + elements go. Without `qstn/@seqNo` or any `cdl:` note it is not a CDL + codebook, and each field only CDL carries gets one `ddi-field-missing` + warning. An unknown `responseDomainType` is read as text + (`ddi-type-unknown`). +- **A fragment**, i.e. a `` or bare `` / `` elements, is + read the same way. A `varGrp` that refers to a `var` outside the fragment + gets `ddi-reference-outside`. +- XML that is not well-formed, or holds neither a `codeBook` nor a `var`, + throws `ddi-invalid`. + +## Known losses + +What a CDL codebook doesn't give back (the round-trip tests fold these out): + +1. **List names.** Identical category sets come back as one list, named + after the first question that uses it (a grid's after the grid). +2. **The `or_other` shorthand** comes back as the explicit pair (an `other` + choice plus a `_other` text question with its `relevant`), which + behaves the same. The `other` choice's label is what LimeSurvey shows + (`convention:other` `limesurveyOtherText`), or for a select_multiple the + convention's label. +3. **Note rows:** + - Their names are rebuilt as `_note`. + - Consecutive notes before one question come back as one note. + - A note with no question after it in its group (an intro or outro) comes + back at the end of the survey, under its own name. + - A note's hint is lost. +4. **Rows DDI has no variable for**: device and session metadata (`start`, + `end`, `deviceid`, …), matrix header rows, `calculate` and other + unregistered types. A group with none of its questions left is dropped. +5. **Language names.** Columns use the BCP 47 tag (`label::de`), and + `default_language` gets an English name (`German (de)`). +6. **Defaults are written as defaults.** A range's `start`/`end` equal to + the registry defaults (1, 10) aren't written back. A group without a label + comes back labelled with its name. `required` is `yes` or absent. +7. A `constraint_message` without a `constraint` is not in the DDI. +8. **Choice columns** other than `exclusive` (media, filters) and **settings** + other than `form_title`, `form_id`, `version`, `default_language` and + `style` are not carried. +9. **The title.** A codebook built with an explicit study title (`assetName`) + has that title, not `form_title`. diff --git a/src/pipelines/ddi2xlsform/index.ts b/src/pipelines/ddi2xlsform/index.ts new file mode 100644 index 0000000..c1426f4 --- /dev/null +++ b/src/pipelines/ddi2xlsform/index.ts @@ -0,0 +1,31 @@ +/** + * DDI Codebook 2.5 → XLSForm (#154): parsed into the Instrument + * (`src/instrument/fromDdi.ts`), then emitted by the shared XLSForm emitter + * (`src/xlsform/fromInstrument.ts`). See `./README.md` for what a CDL + * codebook gives back and what any other DDI can't. + */ +import type { WarningHandler } from '../../diagnostics.js'; +import { instrumentFromDdi } from '../../instrument/fromDdi.js'; +import { + xlsformFromInstrument, + type XlsformOutput, +} from '../../xlsform/fromInstrument.js'; + +export type { XlsformOutput }; + +export interface DdiToXlsformOptions { + /** Receives what the DDI can't supply; the conversion never stops for it. */ + onWarning?: WarningHandler; +} + +/** + * The XLSForm sheets (`{ survey, choices, settings }`) of a DDI codebook or + * fragment (``, ``, ``). Throws `ddi-invalid` only + * when the XML is not well-formed or holds no ``. + */ +export function ddiToXlsform( + xml: string, + options: DdiToXlsformOptions = {}, +): XlsformOutput { + return xlsformFromInstrument(instrumentFromDdi(xml, options)); +} diff --git a/src/pipelines/lstsv2xlsform/README.md b/src/pipelines/lstsv2xlsform/README.md index 4cf3838..fa0fde0 100644 --- a/src/pipelines/lstsv2xlsform/README.md +++ b/src/pipelines/lstsv2xlsform/README.md @@ -112,8 +112,11 @@ listed here must match exactly. arithmetic `+`; `concat()` is never reconstructed. 4. **Non-grid group machine names are best-effort.** A grid's machine name *is* recoverable (from its `F` question's own name — exact). An explicit non-grid - group has none in the TSV, so `slugifyGroupName` falls back to: lowercase, - strip non-alphanumeric, join the first 4 words, `'group'` if empty. + group has only its LimeSurvey group name (its title). The parser keeps it + when it is a valid XLSForm name (`Later`) and otherwise slugifies it + (`groupName` in `src/instrument/fromLstsv.ts`): lowercase, strip + non-alphanumeric, join the first 4 words, `group` if empty, `_2`, `_3` … + for a repeat. 5. **Markdown vs HTML in labels is not distinguishable.** `htmlToMarkdown()` (`../../utils/markdownRenderer.ts`) reverses the HTML constructs `marked` produces (``, ``, ``, ``, `

`, `