From 5c8939a766fe501658b7cfd0f8744424dc988c10 Mon Sep 17 00:00:00 2001 From: jstet Date: Sun, 27 Sep 2026 20:10:42 +0200 Subject: [PATCH] feat(ddi)!: a codebook gives back its whole form; round trips through LimeSurvey (#160) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A CDL codebook now carries what ddi2xlsform used to lose, in cdl: notes where DDI 2.5 has no element (convention:ddiFields): - cdl:list: a select's list name, when it isn't the question's - cdl:or_other: an added other pair, `shorthand` (the type cell's or_other) or `added` (LimeSurvey's other=Y); cdl:other_label for a select_multiple pair's authored other label - cdl:note_names: the note rows a preQTxt joins - stdyDscr: cdl:row (+ cdl:row_label) for rows without data (metadata rows, matrix headers), cdl:row_hint / row_relevant / row_appearance for note rows, cdl:position for orphan notes, rows and groups of notes only (which now get a section varGrp) - cdl:setting for every setting, default_language as authored; cdl:language for the form's language names and order - a range's start/end as authored in cdl:parameters The LimeSurvey TSV → DDI path declares the survey's language (codeBook/@xml:lang), passes its settings (style), and carries exclude_all_others onto the choices. format=G's field-list groups are inferred by the LimeSurvey parser, not the shared XLSForm emitter. New round trips, on every fixture and on generated forms (fast-check): DDI → XLSForm → DDI is byte-identical; XLSForm → DDI → XLSForm → LimeSurvey equals XLSForm → LimeSurvey byte for byte; LimeSurvey → DDI → Instrument equals the TSV's own; XLSForm → LimeSurvey → DDI → XLSForm equals XLSForm → LimeSurvey → XLSForm. The model comparison is strict now. BREAKING CHANGE: lstsv2ddi declares codeBook/@xml:lang for a single-language survey too, so its universe prose is in the survey's language. extractVariables / variablesFromInstrument also return rows without data (Variable.row set). instrumentFromLstsv's settings no longer hold default_language (defaultLanguage does). A CDL codebook's identical category sets are no longer merged into one list. New cdl: note types and section varGrps without members appear in the DDI. Co-Authored-By: Claude Opus 5.5 (1M context) --- ARCHITECTURE.md | 2 +- README.md | 8 +- codegen/schematron.py | 5 +- .../schematron/ddi_custom_rules.sch | 22 +- registry/conventions/ddiFields.jsonld | 118 ++++- registry/entities/grid/ddi.xml | 4 +- registry/entities/note/ddi.xml | 3 +- registry/entities/range/ddi.xml | 4 +- registry/entities/select_multiple/ddi.xml | 3 +- .../entities/select_multiple_other/ddi.xml | 4 +- registry/entities/select_one_other/ddi.xml | 3 +- src/conventions/metadata.ts | 10 +- src/ddi/codebook.ts | 173 ++++++- src/ddi/fields.ts | 110 ++++- src/ddi/fromInstrument.ts | 49 +- src/ddi/notes.ts | 126 +++++- src/ddi/types.ts | 14 + src/generated/conventions.json | 118 ++++- src/generated/conventions.ts | 118 ++++- src/instrument/fromDdi.ts | 422 +++++++++++++++--- src/instrument/fromLstsv.ts | 46 +- src/pipelines/README.md | 4 + src/pipelines/ddi2xlsform/README.md | 86 ++-- src/pipelines/lstsv2ddi/index.ts | 21 +- src/pipelines/lstsv2ddi/toVariables.ts | 17 +- src/pipelines/lstsv2xlsform/README.md | 3 +- src/pipelines/xlsform2ddi/index.ts | 38 +- src/xlsform/fromInstrument.ts | 51 ++- .../fixtures/surveys/all_types_survey/ddi.xml | 32 ++ .../surveys/all_types_survey/ddi2xlsform.json | 147 +++--- .../surveys/appearances_survey/ddi.xml | 11 + .../appearances_survey/ddi2xlsform.json | 34 +- tests/fixtures/surveys/basic_survey/ddi.xml | 4 + .../surveys/basic_survey/ddi2xlsform.json | 10 +- .../fixtures/surveys/bilingual_survey/ddi.xml | 8 + .../surveys/bilingual_survey/ddi2xlsform.json | 49 +- tests/fixtures/surveys/complex_survey/ddi.xml | 3 + .../surveys/complex_survey/ddi2xlsform.json | 17 +- .../surveys/complex_xpath_survey/ddi.xml | 1 + .../complex_xpath_survey/ddi2xlsform.json | 1 + tests/fixtures/surveys/grid_survey/ddi.xml | 6 + .../surveys/grid_survey/ddi2xlsform.json | 26 +- tests/fixtures/surveys/hints_survey/ddi.xml | 3 + .../surveys/hints_survey/ddi2xlsform.json | 10 +- .../surveys/multilingual_survey/ddi.xml | 3 + .../multilingual_survey/ddi2xlsform.json | 1 + .../fixtures/surveys/multipage_survey/ddi.xml | 3 + .../surveys/multipage_survey/ddi2xlsform.json | 15 +- .../fixtures/surveys/settings_survey/ddi.xml | 4 + .../surveys/settings_survey/ddi2xlsform.json | 19 +- tests/fixtures/surveys/testA/ddi.xml | 20 +- tests/fixtures/surveys/testA/ddi2xlsform.json | 73 +-- tests/fixtures/surveys/testB/ddi.xml | 79 ++++ tests/fixtures/surveys/testB/ddi2xlsform.json | 285 ++++++++---- .../validation_relevance_survey/ddi.xml | 3 + .../ddi2xlsform.json | 9 +- tests/ts/contract/canonicalInstrument.ts | 222 +++++---- tests/ts/contract/ddiRoundtrip.test.ts | 84 +--- .../ts/contract/ddiRoundtripGenerated.test.ts | 366 ++++++++++++--- tests/ts/contract/koboRealExports.test.ts | 3 +- tests/ts/contract/lstsv2ddiRoundtrip.test.ts | 51 ++- tests/ts/contract/lstsvDdiRoundtrip.test.ts | 100 +++++ tests/ts/contract/roundtripCases.ts | 92 ++++ tests/ts/unit/ddi/fields.test.ts | 7 +- tests/ts/unit/ddi/fromDdi.test.ts | 34 +- tests/ts/unit/ddi/multilingual.test.ts | 4 +- tests/ts/unit/ddi/structure.test.ts | 11 +- tests/ts/unit/instrument/fromLstsv.test.ts | 7 +- .../pipelines/xlsform2ddi/codebook.test.ts | 5 +- .../test_registry_schematron_conformance.py | 11 + 70 files changed, 2620 insertions(+), 835 deletions(-) create mode 100644 tests/ts/contract/lstsvDdiRoundtrip.test.ts create mode 100644 tests/ts/contract/roundtripCases.ts diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 460f5dd..ed2b0a8 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -151,7 +151,7 @@ The migration runs in phases, each keeping every snapshot byte-identical: ### The way back from DDI (`ddi2xlsform`) -A DDI codebook describes a *dataset*, not an *instrument*. A codebook formtransform wrote carries the whole instrument too (#155), so `ddi2xlsform` (#154) turns it back into its form: `src/instrument/fromDdi.ts` reads the standard elements and the `cdl:` notes into the Instrument, and the XLSForm emitter both reverse paths share (`src/xlsform/fromInstrument.ts`) writes the sheets. Any other DDI converts as far as its standard elements go, with a warning for each missing field. The round trip is tested on the model, never on bytes; `src/pipelines/ddi2xlsform/README.md` lists the losses. A `ddi2lstsv` would be `ddi2xlsform` → `xlsform2lstsv` and earns no module of its own. +A DDI codebook describes a *dataset*, not an *instrument*. A codebook formtransform wrote carries the whole instrument too (#155), so `ddi2xlsform` (#154) turns it back into its form: `src/instrument/fromDdi.ts` reads the standard elements and the `cdl:` notes into the Instrument, and the XLSForm emitter both reverse paths share (`src/xlsform/fromInstrument.ts`) writes the sheets. Any other DDI converts as far as its standard elements go, with a warning for each missing field. The round trips are tested on the model, and where the target is text, on bytes: DDI → XLSForm → DDI and XLSForm → DDI → XLSForm → LimeSurvey give the same file (#160). `src/pipelines/ddi2xlsform/README.md` lists what stays lost. A `ddi2lstsv` would be `ddi2xlsform` → `xlsform2lstsv` and earns no module of its own. What a CDL codebook carries: diff --git a/README.md b/README.md index e227cec..5f3c00c 100644 --- a/README.md +++ b/README.md @@ -192,15 +192,17 @@ records how each maps onto it. `xlsform2lstsv` (deploy the survey), `xlsform2ddi` (document the dataset), `lstsv2ddi`, `lstsv2xlsform` and `ddi2xlsform` (the reverse paths). All are lossy for some types: nested groups flatten in LimeSurvey, choice codes over 5 chars truncate, -`select_multiple` becomes N binary variables, and the reverse paths cannot -recover a select's authored `list_name`. +`select_multiple` becomes N binary variables, and the LimeSurvey reverse paths +cannot recover a select's authored `list_name`. A CDL codebook holds the whole +form: XLSForm → DDI → XLSForm gives it back, and its codebook again. **DDI goes back to XLSForm:** a CDL codebook carries the whole form. Skip logic, validation and `required` are each a readable `` sentence (a simple numeric range also ``), plus the exact expression in a typed note such as `` (`convention:logicMapping`). Groups, order, hints, defaults, appearances and -parameters are in standard DDI where it has a place and typed notes where not +parameters, list names, note rows and metadata rows, settings and language +names are in standard DDI where it has a place and typed notes where not (`convention:ddiFields`). `ddiToXlsform` reads it back; any other DDI converts as far as its standard elements go, with a warning per missing field ([`src/pipelines/ddi2xlsform/README.md`](src/pipelines/ddi2xlsform/README.md)). diff --git a/codegen/schematron.py b/codegen/schematron.py index eb08130..f2fbcd1 100644 --- a/codegen/schematron.py +++ b/codegen/schematron.py @@ -208,13 +208,14 @@ def generate_schematron(registry: dict[str, Any], output: Path) -> None: {subject_rules} """ - # At most one note of each cdl: type per element and language (per subject on stdyDscr). + # At most one note of each cdl: type per element and language (on stdyDscr, + # where the subject names what a note is about: per subject too). cdl_note_uniqueness = """\ has more than one note of one cdl: type in one language. - A setting has more than one cdl:setting note. + The study has more than one note of one cdl: type about one subject in one language. """ diff --git a/ddi-validation/schematron/ddi_custom_rules.sch b/ddi-validation/schematron/ddi_custom_rules.sch index 8477fe5..1d60f10 100644 --- a/ddi-validation/schematron/ddi_custom_rules.sch +++ b/ddi-validation/schematron/ddi_custom_rules.sch @@ -216,15 +216,29 @@ - Note type "" is not in the CDL vocabulary (cdl:appearance, cdl:constraint, cdl:constraint_message, cdl:default, cdl:exclusive, cdl:hint, cdl:parameters, cdl:relevant, cdl:required, cdl:setting). + Note type "" is not in the CDL vocabulary (cdl:appearance, cdl:constraint, cdl:constraint_message, cdl:default, cdl:exclusive, cdl:hint, cdl:language, cdl:list, cdl:note_names, cdl:or_other, cdl:other_label, cdl:parameters, cdl:position, cdl:relevant, cdl:required, cdl:row, cdl:row_appearance, cdl:row_hint, cdl:row_label, cdl:row_relevant, cdl:setting). A cdl:constraint note needs subject="xlsform-xpath": its text is an expression in that syntax. + A cdl:language note needs a subject: the name of what it holds. + A cdl:position note needs a subject: the name of what it holds. A cdl:relevant note needs subject="xlsform-xpath": its text is an expression in that syntax. + A cdl:row note needs a subject: the name of what it holds. + A cdl:row_appearance note needs a subject: the name of what it holds. + A cdl:row_hint note needs a subject: the name of what it holds. + A cdl:row_label note needs a subject: the name of what it holds. + A cdl:row_relevant note needs a subject: the name of what it holds. A cdl:setting note needs a subject: the name of what it holds. - Note type "" is not in the CDL vocabulary (cdl:appearance, cdl:constraint, cdl:constraint_message, cdl:default, cdl:exclusive, cdl:hint, cdl:parameters, cdl:relevant, cdl:required, cdl:setting). + Note type "" is not in the CDL vocabulary (cdl:appearance, cdl:constraint, cdl:constraint_message, cdl:default, cdl:exclusive, cdl:hint, cdl:language, cdl:list, cdl:note_names, cdl:or_other, cdl:other_label, cdl:parameters, cdl:position, cdl:relevant, cdl:required, cdl:row, cdl:row_appearance, cdl:row_hint, cdl:row_label, cdl:row_relevant, cdl:setting). A cdl:constraint note needs subject="xlsform-xpath": its text is an expression in that syntax. + A cdl:language note needs a subject: the name of what it holds. + A cdl:position note needs a subject: the name of what it holds. A cdl:relevant note needs subject="xlsform-xpath": its text is an expression in that syntax. + A cdl:row note needs a subject: the name of what it holds. + A cdl:row_appearance note needs a subject: the name of what it holds. + A cdl:row_hint note needs a subject: the name of what it holds. + A cdl:row_label note needs a subject: the name of what it holds. + A cdl:row_relevant note needs a subject: the name of what it holds. A cdl:setting note needs a subject: the name of what it holds. @@ -234,13 +248,13 @@ has more than one note of one cdl: type in one language. - A setting has more than one cdl:setting note. + The study has more than one note of one cdl: type about one subject in one language. has more than one note of one cdl: type in one language. - A setting has more than one cdl:setting note. + The study has more than one note of one cdl: type about one subject in one language. diff --git a/registry/conventions/ddiFields.jsonld b/registry/conventions/ddiFields.jsonld index 7177b77..96fbfba 100644 --- a/registry/conventions/ddiFields.jsonld +++ b/registry/conventions/ddiFields.jsonld @@ -63,7 +63,7 @@ "var", "varGrp" ], - "text": "the parameters not in a standard element, space-separated key=value" + "text": "the parameters not in a standard element, space-separated key=value; a range's start and end as authored (valrng/range has them too)" }, "hint": { "type": "cdl:hint", @@ -85,21 +85,113 @@ "stdyDscr" ], "subject": "the settings key", - "keys": [ - "style" + "text": "the value of every string or number setting DDI has no element for (id_string too: IDNo is form_id's), default_language as authored (codeBook/@xml:lang is also set without one: a LimeSurvey survey's language, a multilingual form's first)", + "standard": [ + "form_title", + "form_id", + "version" ] + }, + "list": { + "type": "cdl:list", + "on": [ + "var", + "varGrp" + ], + "text": "the choice list's name, when it isn't the question's own name (a grid member's: its grid's)" + }, + "or_other": { + "type": "cdl:or_other", + "on": [ + "varGrp[@type='other']" + ], + "text": "the pair's other choice and companion question were added, not authored: 'shorthand' from the XLSForm type cell's or_other, 'added' from a source that only says the question has one (LimeSurvey's other=Y), whose XLSForm writes the explicit pair" + }, + "note_names": { + "type": "cdl:note_names", + "on": [ + "var", + "varGrp" + ], + "text": "the names of the note rows whose texts the preQTxt (or a group's untyped notes) joins with a blank line, space-separated, in order" + }, + "row": { + "type": "cdl:row", + "on": [ + "stdyDscr" + ], + "subject": "the row's name", + "text": "the type cell of a row with no data column: a metadata row (convention:unregisteredRows metadataRowTypes: start, deviceid, …) or a matrix header (an appearance with carriesData false)" + }, + "row_label": { + "type": "cdl:row_label", + "on": [ + "stdyDscr" + ], + "subject": "the row's name", + "localized": true, + "text": "a cdl:row's label" + }, + "row_hint": { + "type": "cdl:row_hint", + "on": [ + "stdyDscr" + ], + "subject": "the name of a note row or cdl:row", + "localized": true, + "text": "its hint" + }, + "row_relevant": { + "type": "cdl:row_relevant", + "on": [ + "stdyDscr" + ], + "subject": "the name of a note row or cdl:row", + "text": "its relevant (XLSForm XPath)" + }, + "row_appearance": { + "type": "cdl:row_appearance", + "on": [ + "stdyDscr" + ], + "subject": "the name of a note row or cdl:row", + "text": "its appearance" + }, + "position": { + "type": "cdl:position", + "on": [ + "stdyDscr" + ], + "subject": "the name of an intro/outro note or a cdl:row, or the path of a group with no data question", + "text": "where it is: in= after=" + }, + "language": { + "type": "cdl:language", + "on": [ + "stdyDscr" + ], + "subject": "a language's BCP 47 tag", + "text": "the form's own name for it (its column suffix, 'Deutsch (de)'): every language of a form in several, in the form's order; of a form in one, when its name isn't the bare tag" + }, + "other_label": { + "type": "cdl:other_label", + "on": [ + "varGrp[@type='other']" + ], + "localized": true, + "text": "a select_multiple's authored other choice's label: the pair has no binary var for it" } }, "fields": { "ItemBase.name": { - "ddi": "var/@name; a group's path in varGrp/@name" + "ddi": "var/@name; a group's path in varGrp/@name; a note's in cdl:note_names or notes/@subject" }, "ItemBase.label": { "ddi": "qstn/qstnLit; a group's varGrp/txt" }, "ItemBase.hint": { "ddi": "qstn/postQTxt; a group's in a cdl:hint note (varGrp has no hint element)", - "note": "CDL uses postQTxt for the hint shown to respondents with the question, though the XSD describes it as text about what follows the question: DDI 2.5 has no hint element." + "note": "CDL uses postQTxt for the hint shown to respondents with the question, though the XSD describes it as text about what follows the question: DDI 2.5 has no hint element. A note row's hint is a cdl:row_hint." }, "ItemBase.relevant": { "ddi": "universe + cdl:relevant note (convention:logicMapping)" @@ -127,17 +219,17 @@ "ddi": "qstn/@responseDomainType, var/@intrvl, varFormat/@type, varFormat/@category (date, time), var/@dcml='0' (integer)" }, "QuestionItem.rawType": { - "loss": "Rebuilt from type, list and the other pattern; the or_other shorthand comes back as the explicit pair, which behaves the same." + "ddi": "type, list and the other pattern; the or_other shorthand: cdl:or_other" }, "QuestionItem.list": { - "loss": "The list name: questions with identical category sets come back sharing one list." + "cdlNote": "cdl:list", + "note": "Only when the list isn't named after the question (a grid member: after its grid); without it the list has that name. DDI that isn't CDL gets identical category sets as one list." }, "QuestionItem.file": { "ddi": "concept/@vocab (the file is .csv)" }, "QuestionItem.orOther": { - "ddi": "varGrp[@type='other'] (convention:other)", - "note": "Comes back as the explicit pair, see rawType." + "ddi": "varGrp[@type='other'] (convention:other), with cdl:or_other when it was the shorthand" }, "QuestionItem.guidanceHint": { "ddi": "qstn/ivuInstr" @@ -172,20 +264,20 @@ "note": "The choice sheet's other columns are not carried." }, "Instrument.languages": { - "ddi": "xml:lang siblings of every text (convention:languageTagging)" + "ddi": "xml:lang siblings of every text (convention:languageTagging); a language's own name in a cdl:language note" }, "Instrument.defaultLanguage": { "ddi": "codeBook/@xml:lang" }, "Instrument.settings": { - "ddi": "form_title: titl; form_id: IDNo; version: verStmt/version; default_language: codeBook/@xml:lang; style: stdyDscr/notes[@type='cdl:setting'][@subject='style']", - "note": "Other settings are not carried." + "ddi": "form_title: titl (+ parTitl); form_id: IDNo; version: verStmt/version; every other setting, default_language as authored too: stdyDscr/notes[@type='cdl:setting'][@subject=]", + "note": "Settings that are neither a string nor a number are not carried." }, "Instrument.lists": { "ddi": "catgry of each var that uses a list" }, "Instrument.body": { - "ddi": "dataDscr: varGrp and var, in survey order" + "ddi": "dataDscr: varGrp and var, in survey order; note rows in preQTxt (names: cdl:note_names) or as stdyDscr notes type='instruction'; rows without data as cdl:row; their fields in cdl:row_*; where the stdyDscr ones are: cdl:position" } } } diff --git a/registry/entities/grid/ddi.xml b/registry/entities/grid/ddi.xml index b760b26..5a7d48d 100644 --- a/registry/entities/grid/ddi.xml +++ b/registry/entities/grid/ddi.xml @@ -6,7 +6,7 @@ grid - 2026-07-22 + 2026-09-27 @@ -52,6 +52,7 @@ Das Parlament + skala5 @@ -80,6 +81,7 @@ Die Polizei + skala5 diff --git a/registry/entities/note/ddi.xml b/registry/entities/note/ddi.xml index 61bbee9..bcbc908 100644 --- a/registry/entities/note/ddi.xml +++ b/registry/entities/note/ddi.xml @@ -6,10 +6,11 @@ note - 2026-07-22 + 2026-09-27 Im folgenden Abschnitt geht es um Ihre Lebenssituation. + in= after= diff --git a/registry/entities/range/ddi.xml b/registry/entities/range/ddi.xml index dad2e11..09b5d97 100644 --- a/registry/entities/range/ddi.xml +++ b/registry/entities/range/ddi.xml @@ -6,7 +6,7 @@ range - 2026-09-24 + 2026-09-27 @@ -30,7 +30,7 @@ Wie zufrieden sind Sie insgesamt? (0 = gar nicht, 10 = voll) - step=1 + start=0 end=10 step=1 diff --git a/registry/entities/select_multiple/ddi.xml b/registry/entities/select_multiple/ddi.xml index 1d9f471..8de0f0f 100644 --- a/registry/entities/select_multiple/ddi.xml +++ b/registry/entities/select_multiple/ddi.xml @@ -6,7 +6,7 @@ select_multiple - 2026-07-22 + 2026-09-27 @@ -24,6 +24,7 @@ An welchen Tagen des Wochenendes sind Sie erreichbar? An welchen Tagen des Wochenendes sind Sie erreichbar? + wochenendtage diff --git a/registry/entities/select_multiple_other/ddi.xml b/registry/entities/select_multiple_other/ddi.xml index c2c0a81..799f4d0 100644 --- a/registry/entities/select_multiple_other/ddi.xml +++ b/registry/entities/select_multiple_other/ddi.xml @@ -6,7 +6,7 @@ select_multiple_other - 2026-07-22 + 2026-09-27 @@ -24,6 +24,8 @@ Welche dieser Geräte besitzen Sie? Welche dieser Geräte besitzen Sie? + geraete + Sonstiges Welche dieser Geräte besitzen Sie? diff --git a/registry/entities/select_one_other/ddi.xml b/registry/entities/select_one_other/ddi.xml index 4f8844e..97ee46f 100644 --- a/registry/entities/select_one_other/ddi.xml +++ b/registry/entities/select_one_other/ddi.xml @@ -6,7 +6,7 @@ select_one_other - 2026-07-22 + 2026-09-27 @@ -47,6 +47,7 @@ Wie sind Sie auf unser Angebot aufmerksam geworden? + quelle diff --git a/src/conventions/metadata.ts b/src/conventions/metadata.ts index 3f0162f..1a8661d 100644 --- a/src/conventions/metadata.ts +++ b/src/conventions/metadata.ts @@ -1,8 +1,16 @@ /** * `convention:unregisteredRows`: XLSForm metadata rows (`start`, `end`, - * `today`, …) carry no question and are skipped by every emitter. + * `today`, …) carry no question. LimeSurvey skips them; the DDI keeps them + * as `cdl:row` notes (#160), without a variable. */ import conventions from '../generated/conventions.js'; export const METADATA_ROW_TYPES: readonly string[] = conventions.conventions.unregisteredRows.metadataRowTypes; + +const METADATA = new Set(METADATA_ROW_TYPES); + +/** A metadata row's type (`start`, `deviceid`, …). */ +export function isMetadataType(type: string): boolean { + return METADATA.has(type); +} diff --git a/src/ddi/codebook.ts b/src/ddi/codebook.ts index 86ae54f..2274e48 100644 --- a/src/ddi/codebook.ts +++ b/src/ddi/codebook.ts @@ -21,13 +21,19 @@ import { } from '../conventions/other.js'; import { isGridAppearance } from '../conventions/grid.js'; import { XmlElement } from './xml.js'; -import { classifyNotes } from './notes.js'; +import { classifyNotes, type Position } from './notes.js'; +import conventions from '../generated/conventions.js'; import { Choice, DdiGroup, Translations, Variable } from './types.js'; import { localizedChild, textsOf } from './translations.js'; import { addExclusiveNote, addFieldNotes, addGroupFieldNotes, + addLanguageNotes, + addListNote, + addNoteNames, + addRowFieldNotes, + addRowNotes, addSettingNotes, references, } from './fields.js'; @@ -43,6 +49,8 @@ import { import { registeredVocabCodes } from '../conventions/fromFile.js'; import { languageTagOf } from '../utils/languageUtils.js'; +const NOTES = conventions.conventions.ddiFields.notes; + const NS = 'ddi:codebook:2_5'; const XSI = 'http://www.w3.org/2001/XMLSchema-instance'; const SCHEMA_LOC = @@ -91,6 +99,10 @@ interface AddVarOpts { vocab?: string; preQTxt?: string; preQTxtTranslations?: Translations; + /** The note rows `preQTxt` joins, by name (`cdl:note_names`). */ + noteNames?: string[]; + /** The list name a reader assumes without `cdl:list` (a grid's name). */ + listDefault?: string; } interface AddVarSpec { @@ -189,7 +201,9 @@ function addVarElement(parent: XmlElement, spec: AddVarSpec): XmlElement { if (spec.logic) { addLogicNotes(varEl, spec.logic.v); addFieldNotes(varEl, spec.logic.v); + addListNote(varEl, spec.logic.v, spec.opts?.listDefault ?? name); } + addNoteNames(varEl, spec.opts?.noteNames); return varEl; } @@ -280,6 +294,14 @@ function addGroupLogic( addLogicNotes(grpEl, v); addFieldNotes(grpEl, v); addExclusiveNote(grpEl, v.choices); + addListNote(grpEl, v, v.name); +} + +/** A semi-open pair written with the `or_other` shorthand (#160). */ +function addOrOtherNote(grpEl: XmlElement, p: OtherPattern): void { + if (p.base.orOther && p.otherVar.synthesized) { + grpEl.textChild('notes', p.base.orOther, { type: NOTES.or_other.type }); + } } /** Append a binary 0/1 `` for one `select_multiple` option. */ @@ -386,6 +408,14 @@ function emitOtherPattern( // No var has the question's name: the parent group carries its logic, // and a note before it. addGroupLogic(parentEl, base, ctx, { note: notes, name: baseName }); + addOrOtherNote(parentEl, p); + // Its binary isn't written: the authored other answer's label (#160). + const other = base.choices.find((c) => c.name === OTHER_CODE); + if (other && !p.otherVar.synthesized) { + localizedChild(parentEl, 'notes', other.label, other.translations, { + type: NOTES.other_label.type, + }); + } const childEl = dataDscr.child('varGrp', { ID: childId, @@ -404,6 +434,7 @@ function emitOtherPattern( }); localizedChild(parentEl, 'txt', label, labelTranslations); parentEl.textChild('concept', label); + addOrOtherNote(parentEl, p); } } @@ -432,6 +463,7 @@ function emitOtherPatternVars( opts: { preQTxt: notes.text[baseName] ?? '', preQTxtTranslations: notes.translations[baseName], + noteNames: notes.names[baseName], }, hint: base.hint, guidanceHint: base.guidanceHint, @@ -477,6 +509,17 @@ export interface BuildDdiOptions { datasetFilename?: string; /** Override the `prodDate` (ISO `YYYY-MM-DD`); defaults to today. */ prodDate?: string; + /** + * The form's own name of each language, by tag (`{ de: 'Deutsch (de)' }`): + * a `cdl:language` note where it isn't the tag (#160). + */ + languageNames?: Record; + /** + * The base language's tag (`codeBook/@xml:lang`) when the form doesn't + * author one (`settings.default_language`, which wins): a LimeSurvey + * survey's language, a multilingual form's first. + */ + language?: string; } /** @@ -534,13 +577,30 @@ export function splitDataVars(dataVars: Variable[]): DataVarBuckets { return { otherPatterns, gridGroups, multiRespGroups, units }; } -/** Emit `` (citation, then orphan notes and `cdl:setting` notes). */ +/** What `` holds besides the citation. */ +interface StudyNotes { + /** Notes with no question after them in their group. */ + orphans: Variable[]; + /** Rows without data: metadata rows, matrix headers. */ + rows: Variable[]; + /** Every note row, for its fields. */ + notes: Variable[]; + /** Where the orphans and rows are. */ + positions: Position[]; + /** Language tag → the form's name for it. */ + languageNames: Record; +} + +/** + * Emit ``: the citation, then the orphan notes, the metadata rows + * and where they are (#160), and the `cdl:setting` / `cdl:language` notes. + */ function addStudyDscr( root: XmlElement, settings: DdiSettings, title: StudyTitle, prodDate: string, - orphanNotes: Variable[], + study: StudyNotes, ): void { const stdy = root.child('stdyDscr'); const citation = stdy.child('citation'); @@ -560,13 +620,28 @@ function addStudyDscr( const ver = settings.version; if (ver) citation.child('verStmt').textChild('version', String(ver)); - for (const note of orphanNotes) { + const kept = new Set(); + for (const note of study.orphans) { if (!note.label) continue; const attrs: Record = { type: 'instruction' }; if (note.name) attrs.subject = note.name; localizedChild(stdy, 'notes', note.label, textsOf(note, 'label'), attrs); + kept.add(note.name); + } + for (const row of study.rows) { + addRowNotes(stdy, row); + kept.add(row.name); + } + for (const note of study.notes) addRowFieldNotes(stdy, note); + for (const p of study.positions) { + if (!p.name || !(p.group || kept.has(p.name))) continue; + stdy.textChild('notes', `in=${p.in} after=${p.after}`, { + type: NOTES.position.type, + subject: p.name, + }); } addSettingNotes(stdy, settings); + addLanguageNotes(stdy, study.languageNames); } interface StudyTitle { @@ -579,16 +654,16 @@ interface StudyTitle { * The study title: `assetName`, else `form_title` (a `{ lang: text }` one in * the base language, the others as parallel titles), else `Untitled`. */ -function studyTitle(assetName: string, settings: DdiSettings): StudyTitle { +function studyTitle( + assetName: string, + settings: DdiSettings, + base: string | null, +): StudyTitle { const raw = settings.form_title; if (assetName.trim()) return { title: assetName.trim(), parallel: {} }; if (raw === null || typeof raw !== 'object') { return { title: String(raw ?? '').trim() || 'Untitled', parallel: {} }; } - const base = - typeof settings.default_language === 'string' - ? languageTagOf(settings.default_language) - : null; const byTag = Object.entries(raw as Record) .map(([key, v]): [string, string] => [ languageTagOf(key) ?? key, @@ -602,6 +677,18 @@ function studyTitle(assetName: string, settings: DdiSettings): StudyTitle { }; } +/** The codebook's language: the authored `default_language`, else `fallback`. */ +function baseLanguage( + settings: DdiSettings, + fallback: string | undefined, +): string | null { + const authored = + typeof settings.default_language === 'string' + ? languageTagOf(settings.default_language) + : null; + return authored ?? fallback ?? null; +} + /** Emit `` with `caseQnty` set to the submissions count. */ function addFileDscr( root: XmlElement, @@ -621,6 +708,8 @@ function addFileDscr( interface InlineNotes { text: Record; translations: Record; + /** The notes' names, in order. */ + names: Record; } /** A group's lead-in note (the one preceding `name`) as ``. */ @@ -630,7 +719,9 @@ function addGroupNote( name: string, ): void { const note = notes.text[name]; - if (note) localizedChild(grpEl, 'notes', note, notes.translations[name]); + if (!note) return; + localizedChild(grpEl, 'notes', note, notes.translations[name]); + addNoteNames(grpEl, notes.names[name]); } /** A group with what it directly contains, by `varGrp` ID reference. */ @@ -654,7 +745,10 @@ function unitVar(unit: EmitUnit): Variable { * its direct members (#152). A grid member belongs to its grid's `varGrp`; * a `select_multiple` and a semi-open pair are represented by theirs. */ -function groupTree(units: EmitUnit[]): Map { +function groupTree( + units: EmitUnit[], + emptyGroups: DdiGroup[][] = [], +): Map { const tree = new Map(); for (const unit of units) { const chain = unitVar(unit).groups ?? []; @@ -676,6 +770,18 @@ function groupTree(units: EmitUnit[]): Map { inner.groups.push(makeGrpId(unit.p.base.name)); else if (!unit.grid) inner.vars.push(makeVarId(unit.v.name)); } + // A group of notes only: its notes' cdl:position place it (#160). + for (const chain of emptyGroups) { + const group = chain[chain.length - 1]; + tree.set(group.path, { + group, + universe: chain.map((g) => g.relevant), + vars: [], + groups: [], + }); + const parent = chain[chain.length - 2]; + if (parent) tree.get(parent.path)?.groups.push(makeGrpId(group.path)); + } return tree; } @@ -724,11 +830,11 @@ function addVarGroups( dataDscr: XmlElement, dataVars: Variable[], notes: InlineNotes, - buckets: DataVarBuckets, + buckets: DataVarBuckets & { emptyGroups: DdiGroup[][] }, ctx: LogicContext, ): void { const { gridGroups, multiRespGroups, otherPatterns } = buckets; - const tree = groupTree(buckets.units); + const tree = groupTree(buckets.units, buckets.emptyGroups); for (const node of tree.values()) { if (!gridGroups.has(node.group.path)) addSection(dataDscr, node, ctx); @@ -803,7 +909,11 @@ function addVars( label: unit.v.label, varType: unit.v.type, choices: unit.v.choices, - opts: { preQTxt: group.label, preQTxtTranslations: group.translations }, + opts: { + preQTxt: group.label, + preQTxtTranslations: group.translations, + listDefault: unit.grid.slice(unit.grid.lastIndexOf('/') + 1), + }, hint: unit.v.hint, guidanceHint: unit.v.guidanceHint, ...specTranslations(unit.v), @@ -832,6 +942,7 @@ function addStandaloneVar( vocab: v.vocab, preQTxt: notes.text[v.name] ?? '', preQTxtTranslations: notes.translations[v.name], + noteNames: notes.names[v.name], }, hint: v.hint, guidanceHint: v.guidanceHint, @@ -854,35 +965,51 @@ export function buildDdiCodebook( submissions = [], datasetFilename = 'data.csv', prodDate = new Date().toISOString().slice(0, 10), + languageNames = {}, } = options; const classified = classifyNotes(variables); - const { dataVars, orphanNotes } = classified; + const { dataVars } = classified; const notes: InlineNotes = { text: classified.inlinePreqtxt, translations: classified.inlinePreqtxtTranslations, + names: classified.inlineNames, }; - const title = studyTitle(assetName, settings); + const title = studyTitle( + assetName, + settings, + baseLanguage(settings, options.language), + ); const root = new XmlElement('codeBook'); root.setAttr('xmlns', NS); root.setAttr('xmlns:xsi', XSI); root.setAttr('xsi:schemaLocation', SCHEMA_LOC); root.setAttr('version', '2.5'); - const lang = - typeof settings.default_language === 'string' - ? languageTagOf(settings.default_language) - : null; + const lang = baseLanguage(settings, options.language); if (lang) root.setAttr('xml:lang', lang); - addStudyDscr(root, settings, title, prodDate, orphanNotes); + addStudyDscr(root, settings, title, prodDate, { + orphans: classified.orphanNotes, + rows: classified.rows, + notes: variables.filter((v) => v.type === 'note' && v.row === undefined), + positions: classified.positions, + languageNames, + }); addFileDscr(root, datasetFilename, submissions.length); const dataDscr = root.child('dataDscr'); const buckets = splitDataVars(dataVars); - const ctx = logicContext(variables, lang, questionIds(buckets.units)); - addVarGroups(dataDscr, dataVars, notes, buckets, ctx); + const described = variables.filter((v) => v.row === undefined); + const ctx = logicContext(described, lang, questionIds(buckets.units)); + addVarGroups( + dataDscr, + dataVars, + notes, + { ...buckets, emptyGroups: classified.emptyGroups }, + ctx, + ); addVars(dataDscr, dataVars, notes, buckets, ctx); return root; diff --git a/src/ddi/fields.ts b/src/ddi/fields.ts index ccb4d5d..cc35225 100644 --- a/src/ddi/fields.ts +++ b/src/ddi/fields.ts @@ -7,28 +7,23 @@ import conventions from '../generated/conventions.js'; import { GRID_APPEARANCE } from '../conventions/grid.js'; import { parseParameters } from '../utils/parameters.js'; import { TYPE_MAPPINGS } from '../generated/TypeMappings.js'; -import { localizedChild } from './translations.js'; +import { localizedChild, textsOf } from './translations.js'; import type { Choice, DdiGroup, Variable } from './types.js'; import type { XmlElement } from './xml.js'; const NOTES = conventions.conventions.ddiFields.notes; -/** `range`'s own bounds, which its `valrng` carries. */ -const RANGE_BOUNDS = ['start', 'end']; - /** * The `parameters` DDI has no element for, as `key=value` tokens: without - * `guidance_hint` (`ivuInstr`) and a range's `start`/`end` (`valrng`). + * `guidance_hint` (`ivuInstr`). A range's `start`/`end` stay as authored + * (#160), though its `valrng` has them too: a bound equal to the default is + * otherwise not told apart from none. */ export function otherParameters(v: Variable): string { const tokens: string[] = []; for (const part of (v.parameters ?? '').split(';')) { if (/^\s*guidance_hint\s*=/.test(part)) continue; - for (const token of part.split(/[\s,]+/).filter(Boolean)) { - const key = token.split('=')[0].trim().toLowerCase(); - if (v.type === 'range' && RANGE_BOUNDS.includes(key)) continue; - tokens.push(token); - } + tokens.push(...part.split(/[\s,]+/).filter(Boolean)); } return tokens.join(' '); } @@ -83,24 +78,105 @@ export function addGroupFieldNotes( } } -/** The settings DDI has no element for, as `cdl:setting` notes. */ +/** Settings in a standard element: `titl`, `IDNo`, `verStmt/version`. */ +const STANDARD_SETTINGS = new Set(NOTES.setting.standard); + +/** + * Every setting DDI has no element for, as `cdl:setting` notes by key + * (#160). `default_language` too, as authored: `codeBook/@xml:lang` is also + * set without it (a LimeSurvey survey's language, a multilingual form's first). + */ export function addSettingNotes( stdy: XmlElement, settings: Record, ): void { - for (const key of NOTES.setting.keys) { - const value = settings[key]; + const keys = Object.keys(settings).sort(); + for (const [key, value] of keys.map((k) => [k, settings[k]] as const)) { + if (STANDARD_SETTINGS.has(key)) continue; if (typeof value !== 'string' && typeof value !== 'number') continue; const text = String(value).trim(); - if (text) { - stdy.textChild('notes', text, { - type: NOTES.setting.type, - subject: key, + if (!text) continue; + stdy.textChild('notes', text, { type: NOTES.setting.type, subject: key }); + } +} + +/** + * The form's languages (`cdl:language`), in its order, by tag with its own + * name: all of a form in several, else one whose name isn't its bare tag. + */ +export function addLanguageNotes( + stdy: XmlElement, + names: Record, +): void { + const all = Object.keys(names).length > 1; + for (const [tag, name] of Object.entries(names)) { + if (name && (all || name !== tag)) { + stdy.textChild('notes', name, { + type: NOTES.language.type, + subject: tag, }); } } } +/** + * The list's name (`cdl:list`), when it isn't `named` (the question's own + * name, a grid member's grid's): what a reader without the note assumes. + */ +export function addListNote(el: XmlElement, v: Variable, named: string): void { + if (v.listName && v.listName !== named) { + el.textChild('notes', v.listName, { type: NOTES.list.type }); + } +} + +/** + * A row without data on `stdyDscr` (#160): its type cell (`cdl:row`), label + * and fields, by its name. + */ +export function addRowNotes(stdy: XmlElement, v: Variable): void { + stdy.textChild('notes', v.row ?? v.type, { + type: NOTES.row.type, + subject: v.name, + }); + if (v.label) { + localizedChild(stdy, 'notes', v.label, textsOf(v, 'label'), { + type: NOTES.row_label.type, + subject: v.name, + }); + } + addRowFieldNotes(stdy, v); +} + +/** A note row's or data-less row's hint, relevant and appearance, by its name. */ +export function addRowFieldNotes(stdy: XmlElement, v: Variable): void { + const subject = v.name; + if (v.hint) { + localizedChild(stdy, 'notes', v.hint, textsOf(v, 'hint'), { + type: NOTES.row_hint.type, + subject, + }); + } + if (v.relevant) { + stdy.textChild('notes', v.relevant, { + type: NOTES.row_relevant.type, + subject, + }); + } + if (v.appearance) { + stdy.textChild('notes', v.appearance, { + type: NOTES.row_appearance.type, + subject, + }); + } +} + +/** The names of the note rows a lead-in text joins (`cdl:note_names`). */ +export function addNoteNames(el: XmlElement, names: string[] | undefined) { + if (names?.length) { + el.textChild('notes', names.join(' '), { type: NOTES.note_names.type }); + } +} + /** The `${name}`s an expression refers to, in order, each once. */ export function references(expression: string): string[] { return [ diff --git a/src/ddi/fromInstrument.ts b/src/ddi/fromInstrument.ts index 42e4692..f221172 100644 --- a/src/ddi/fromInstrument.ts +++ b/src/ddi/fromInstrument.ts @@ -3,7 +3,7 @@ * texts, with the form's other languages as `translations` (#135), flattened * in survey order with each variable's enclosing group path/label/appearance. Both DDI pipelines go through here (#69). */ -import { METADATA_ROW_TYPES } from '../conventions/metadata.js'; +import { METADATA_ROW_TYPES, isMetadataType } from '../conventions/metadata.js'; import { APPEARANCES } from '../generated/Appearances.js'; import { OTHER_APPLIES_TO, @@ -319,14 +319,59 @@ function pushCompanion( ...groupsOf(ctx), relevant, ...(Object.keys(translations).length ? { translations } : {}), + synthesized: true, }); } +/** + * A row without data (a metadata row like `start`, a matrix header): the + * codebook keeps its name, type cell, texts and place in `cdl:row` notes + * (#160). + */ +function pushRow(q: QuestionItem, ctx: GroupContext, state: ProjectState) { + const translations = variableTranslations( + { label: q.label, hint: q.hint }, + state.others, + ); + state.variables.push({ + name: q.name, + type: q.type, + row: q.rawType.trim().replace(/\s+/g, ' '), + label: pick(q.label, state.lang), + ...optionalText({ + hint: pick(q.hint, state.lang).trim(), + guidanceHint: '', + }), + ...(q.relevant.trim() ? { relevant: q.relevant.trim() } : {}), + ...(q.appearance ? { appearance: q.appearance } : {}), + ...(translations ? { translations } : {}), + group: ctx.path, + groupLabel: ctx.label, + groupAppearance: ctx.appearance, + listName: '', + vocab: '', + choices: [], + ...groupsOf(ctx), + }); +} + +/** A row with no data column: a metadata row, a matrix header. */ +function isRow(q: QuestionItem): boolean { + if (!q.name) return false; + return isMetadataType(q.type) || NO_DATA_APPEARANCES.has(q.appearance); +} + +/** An added other pair: from the type cell's `or_other`, or said to be there. */ +function otherOrigin(q: QuestionItem): 'shorthand' | 'added' { + return /\sor_other\s*$/.test(q.rawType) ? 'shorthand' : 'added'; +} + function pushQuestion( q: QuestionItem, ctx: GroupContext, state: ProjectState, ): void { + if (isRow(q)) return pushRow(q, ctx, state); if (!emitsVariable(q)) return; const { stdType, listName, vocab } = resolveType(q); const orOther = q.orOther && OTHER_TYPES.has(stdType); @@ -352,6 +397,7 @@ function pushQuestion( state.others, ); + const added = !!other && !state.authoredNames.has(q.name + OTHER_SUFFIX); state.variables.push({ name: q.name, type: stdType, @@ -367,6 +413,7 @@ function pushQuestion( ...groupsOf(ctx), ...logicOf(q, state.lang), ...(translations ? { translations } : {}), + ...(added ? { orOther: otherOrigin(q) } : {}), }); if (other) pushCompanion(q, stdType, other, ctx, state); diff --git a/src/ddi/notes.ts b/src/ddi/notes.ts index 034fd82..c2a5b47 100644 --- a/src/ddi/notes.ts +++ b/src/ddi/notes.ts @@ -9,11 +9,16 @@ * - **orphan** — a note with no data-carrying successor in its group (intro / * outro / group-trailing) emits as `` on ``. * - * Consecutive inline notes for the same variable join with a blank line. + * Consecutive inline notes for the same variable join with a blank line; + * their names are kept in order (`cdl:note_names`, #160). + * + * Rows without data (metadata rows like `start`, matrix headers: `row` set) + * carry none either. They and the orphan notes keep their place as a + * {@link Position} (`cdl:position`). */ import { joinTranslations, textsOf } from './translations.js'; -import { Translations, Variable } from './types.js'; +import { DdiGroup, Translations, Variable } from './types.js'; export interface ClassifiedNotes { /** Data-carrying variables in original order (notes removed). */ @@ -22,52 +27,139 @@ export interface ClassifiedNotes { inlinePreqtxt: Record; /** The same text in the form's other languages (#135). */ inlinePreqtxtTranslations: Record; + /** `variable.name` → the names of the notes in its `inlinePreqtxt`. */ + inlineNames: Record; /** Notes with no data-carrying successor in the same group. */ orphanNotes: Variable[]; + /** Rows without data: metadata rows, matrix headers. */ + rows: Variable[]; + /** + * Where each orphan note, metadata row and group without data is, in + * survey order (a group by its path, before what it holds). + */ + positions: Position[]; + /** Groups with no data question under them, each with its enclosing ones. */ + emptyGroups: DdiGroup[][]; +} + +/** An item's place: its group's path and the item before it there. */ +export interface Position { + name: string; + /** The enclosing group's path, `''` at the top. */ + in: string; + /** The name of the item before it in that group, `''` when first. */ + after: string; + /** A group without data, `name` its path. */ + group?: boolean; +} + +/** + * The place of `variables[index]`: the last item before it in its group, a + * nested group by its own name. An added `or_other` companion is not an item. + */ +function positionOf( + variables: Variable[], + index: number, + group = variables[index].group, + name = variables[index].name, +): Position { + const prefix = group ? `${group}/` : ''; + for (let i = index - 1; i >= 0; i--) { + const v = variables[i]; + if (v.synthesized) continue; + if (v.group === group) return { name, in: group, after: v.name }; + if (!v.group.startsWith(prefix)) break; + const after = v.group.slice(prefix.length).split('/')[0]; + return { name, in: group, after }; + } + return { name, in: group, after: '' }; } /** Split notes into inline (``) and orphan (``) buckets. */ export function classifyNotes(variables: Variable[]): ClassifiedNotes { const dataVars: Variable[] = []; const inline: Record = {}; - const orphan: Variable[] = []; - let pending: Variable[] = []; + /** Orphan notes and rows without data, by index in `variables`. */ + const placed: number[] = []; + let pending: number[] = []; - for (const v of variables) { + variables.forEach((v, i) => { + if (v.row !== undefined) { + placed.push(i); + return; + } if (v.type === 'note') { - pending.push(v); - continue; + pending.push(i); + return; } - // A data-carrying variable resolves pending notes: same-group notes attach // to it; different-group notes have no successor in scope → orphan. - const sameGroup = pending.filter((n) => n.group === v.group && n.label); - const diffGroup = pending.filter((n) => n.group !== v.group); - if (sameGroup.length) { - (inline[v.name] ??= []).push(...sameGroup); + for (const n of pending) { + const note = variables[n]; + if (note.group !== v.group) placed.push(n); + else if (note.label) (inline[v.name] ??= []).push(note); } - orphan.push(...diffGroup); pending = []; dataVars.push(v); - } - + }); // Anything left has no successor at all → study outro. - orphan.push(...pending); + placed.push(...pending); + placed.sort((a, b) => a - b); const inlinePreqtxt: Record = {}; const inlinePreqtxtTranslations: Record = {}; + const inlineNames: Record = {}; for (const [name, notes] of Object.entries(inline)) { inlinePreqtxt[name] = notes.map((n) => n.label).join('\n\n'); inlinePreqtxtTranslations[name] = joinTranslations( notes.map((n) => textsOf(n, 'label')), '\n\n', ); + inlineNames[name] = notes.map((n) => n.name); } + const kept = placed.map((i) => variables[i]); + const { positions, emptyGroups } = placements(variables, dataVars, placed); return { dataVars, inlinePreqtxt, inlinePreqtxtTranslations, - orphanNotes: orphan, + inlineNames, + orphanNotes: kept.filter((v) => v.type === 'note'), + rows: kept.filter((v) => v.row !== undefined), + positions, + emptyGroups, }; } + +/** + * The positions of the placed rows, each preceded by those of its groups + * that hold no data question: those groups have no other place. + */ +function placements( + variables: Variable[], + dataVars: Variable[], + placed: number[], +): Pick { + const withData = new Set(); + for (const v of dataVars) { + for (const g of v.groups ?? []) withData.add(g.path); + } + const positions: Position[] = []; + const emptyGroups: DdiGroup[][] = []; + for (const i of placed) { + const chain = variables[i].groups ?? []; + chain.forEach((g, depth) => { + if (withData.has(g.path)) return; + withData.add(g.path); + const parent = depth ? chain[depth - 1].path : ''; + positions.push({ + ...positionOf(variables, i, parent, g.path), + group: true, + }); + emptyGroups.push(chain.slice(0, depth + 1)); + }); + positions.push(positionOf(variables, i)); + } + return { positions, emptyGroups }; +} diff --git a/src/ddi/types.ts b/src/ddi/types.ts index feeb9db..d44196a 100644 --- a/src/ddi/types.ts +++ b/src/ddi/types.ts @@ -102,4 +102,18 @@ export interface Variable { parameters?: string; /** Its texts in the form's other languages, by language tag (#135). */ translations?: Record; + /** + * A select whose "other" answer and companion were added, not authored: a + * `cdl:or_other` note (#160). `shorthand` from the XLSForm type cell's + * `or_other`, `added` from a source that only says it has one + * (LimeSurvey's `other=Y`). + */ + orOther?: 'shorthand' | 'added'; + /** The `or_other` companion the projection added, not an authored row. */ + synthesized?: boolean; + /** + * A row with no data column (a metadata row, a matrix header): its type + * cell, a `cdl:row` note (#160). + */ + row?: string; } diff --git a/src/generated/conventions.json b/src/generated/conventions.json index 8ec8126..03de26d 100644 --- a/src/generated/conventions.json +++ b/src/generated/conventions.json @@ -35,7 +35,7 @@ "var", "varGrp" ], - "text": "the parameters not in a standard element, space-separated key=value" + "text": "the parameters not in a standard element, space-separated key=value; a range's start and end as authored (valrng/range has them too)" }, "hint": { "type": "cdl:hint", @@ -57,21 +57,113 @@ "stdyDscr" ], "subject": "the settings key", - "keys": [ - "style" + "text": "the value of every string or number setting DDI has no element for (id_string too: IDNo is form_id's), default_language as authored (codeBook/@xml:lang is also set without one: a LimeSurvey survey's language, a multilingual form's first)", + "standard": [ + "form_title", + "form_id", + "version" ] + }, + "list": { + "type": "cdl:list", + "on": [ + "var", + "varGrp" + ], + "text": "the choice list's name, when it isn't the question's own name (a grid member's: its grid's)" + }, + "or_other": { + "type": "cdl:or_other", + "on": [ + "varGrp[@type='other']" + ], + "text": "the pair's other choice and companion question were added, not authored: 'shorthand' from the XLSForm type cell's or_other, 'added' from a source that only says the question has one (LimeSurvey's other=Y), whose XLSForm writes the explicit pair" + }, + "note_names": { + "type": "cdl:note_names", + "on": [ + "var", + "varGrp" + ], + "text": "the names of the note rows whose texts the preQTxt (or a group's untyped notes) joins with a blank line, space-separated, in order" + }, + "row": { + "type": "cdl:row", + "on": [ + "stdyDscr" + ], + "subject": "the row's name", + "text": "the type cell of a row with no data column: a metadata row (convention:unregisteredRows metadataRowTypes: start, deviceid, …) or a matrix header (an appearance with carriesData false)" + }, + "row_label": { + "type": "cdl:row_label", + "on": [ + "stdyDscr" + ], + "subject": "the row's name", + "localized": true, + "text": "a cdl:row's label" + }, + "row_hint": { + "type": "cdl:row_hint", + "on": [ + "stdyDscr" + ], + "subject": "the name of a note row or cdl:row", + "localized": true, + "text": "its hint" + }, + "row_relevant": { + "type": "cdl:row_relevant", + "on": [ + "stdyDscr" + ], + "subject": "the name of a note row or cdl:row", + "text": "its relevant (XLSForm XPath)" + }, + "row_appearance": { + "type": "cdl:row_appearance", + "on": [ + "stdyDscr" + ], + "subject": "the name of a note row or cdl:row", + "text": "its appearance" + }, + "position": { + "type": "cdl:position", + "on": [ + "stdyDscr" + ], + "subject": "the name of an intro/outro note or a cdl:row, or the path of a group with no data question", + "text": "where it is: in= after=" + }, + "language": { + "type": "cdl:language", + "on": [ + "stdyDscr" + ], + "subject": "a language's BCP 47 tag", + "text": "the form's own name for it (its column suffix, 'Deutsch (de)'): every language of a form in several, in the form's order; of a form in one, when its name isn't the bare tag" + }, + "other_label": { + "type": "cdl:other_label", + "on": [ + "varGrp[@type='other']" + ], + "localized": true, + "text": "a select_multiple's authored other choice's label: the pair has no binary var for it" } }, "fields": { "ItemBase.name": { - "ddi": "var/@name; a group's path in varGrp/@name" + "ddi": "var/@name; a group's path in varGrp/@name; a note's in cdl:note_names or notes/@subject" }, "ItemBase.label": { "ddi": "qstn/qstnLit; a group's varGrp/txt" }, "ItemBase.hint": { "ddi": "qstn/postQTxt; a group's in a cdl:hint note (varGrp has no hint element)", - "note": "CDL uses postQTxt for the hint shown to respondents with the question, though the XSD describes it as text about what follows the question: DDI 2.5 has no hint element." + "note": "CDL uses postQTxt for the hint shown to respondents with the question, though the XSD describes it as text about what follows the question: DDI 2.5 has no hint element. A note row's hint is a cdl:row_hint." }, "ItemBase.relevant": { "ddi": "universe + cdl:relevant note (convention:logicMapping)" @@ -99,17 +191,17 @@ "ddi": "qstn/@responseDomainType, var/@intrvl, varFormat/@type, varFormat/@category (date, time), var/@dcml='0' (integer)" }, "QuestionItem.rawType": { - "loss": "Rebuilt from type, list and the other pattern; the or_other shorthand comes back as the explicit pair, which behaves the same." + "ddi": "type, list and the other pattern; the or_other shorthand: cdl:or_other" }, "QuestionItem.list": { - "loss": "The list name: questions with identical category sets come back sharing one list." + "cdlNote": "cdl:list", + "note": "Only when the list isn't named after the question (a grid member: after its grid); without it the list has that name. DDI that isn't CDL gets identical category sets as one list." }, "QuestionItem.file": { "ddi": "concept/@vocab (the file is .csv)" }, "QuestionItem.orOther": { - "ddi": "varGrp[@type='other'] (convention:other)", - "note": "Comes back as the explicit pair, see rawType." + "ddi": "varGrp[@type='other'] (convention:other), with cdl:or_other when it was the shorthand" }, "QuestionItem.guidanceHint": { "ddi": "qstn/ivuInstr" @@ -144,20 +236,20 @@ "note": "The choice sheet's other columns are not carried." }, "Instrument.languages": { - "ddi": "xml:lang siblings of every text (convention:languageTagging)" + "ddi": "xml:lang siblings of every text (convention:languageTagging); a language's own name in a cdl:language note" }, "Instrument.defaultLanguage": { "ddi": "codeBook/@xml:lang" }, "Instrument.settings": { - "ddi": "form_title: titl; form_id: IDNo; version: verStmt/version; default_language: codeBook/@xml:lang; style: stdyDscr/notes[@type='cdl:setting'][@subject='style']", - "note": "Other settings are not carried." + "ddi": "form_title: titl (+ parTitl); form_id: IDNo; version: verStmt/version; every other setting, default_language as authored too: stdyDscr/notes[@type='cdl:setting'][@subject=]", + "note": "Settings that are neither a string nor a number are not carried." }, "Instrument.lists": { "ddi": "catgry of each var that uses a list" }, "Instrument.body": { - "ddi": "dataDscr: varGrp and var, in survey order" + "ddi": "dataDscr: varGrp and var, in survey order; note rows in preQTxt (names: cdl:note_names) or as stdyDscr notes type='instruction'; rows without data as cdl:row; their fields in cdl:row_*; where the stdyDscr ones are: cdl:position" } } }, diff --git a/src/generated/conventions.ts b/src/generated/conventions.ts index 5786e0f..4e26b98 100644 --- a/src/generated/conventions.ts +++ b/src/generated/conventions.ts @@ -44,7 +44,7 @@ const conventions = { "var", "varGrp" ], - "text": "the parameters not in a standard element, space-separated key=value" + "text": "the parameters not in a standard element, space-separated key=value; a range's start and end as authored (valrng/range has them too)" }, "hint": { "type": "cdl:hint", @@ -66,21 +66,113 @@ const conventions = { "stdyDscr" ], "subject": "the settings key", - "keys": [ - "style" + "text": "the value of every string or number setting DDI has no element for (id_string too: IDNo is form_id's), default_language as authored (codeBook/@xml:lang is also set without one: a LimeSurvey survey's language, a multilingual form's first)", + "standard": [ + "form_title", + "form_id", + "version" ] + }, + "list": { + "type": "cdl:list", + "on": [ + "var", + "varGrp" + ], + "text": "the choice list's name, when it isn't the question's own name (a grid member's: its grid's)" + }, + "or_other": { + "type": "cdl:or_other", + "on": [ + "varGrp[@type='other']" + ], + "text": "the pair's other choice and companion question were added, not authored: 'shorthand' from the XLSForm type cell's or_other, 'added' from a source that only says the question has one (LimeSurvey's other=Y), whose XLSForm writes the explicit pair" + }, + "note_names": { + "type": "cdl:note_names", + "on": [ + "var", + "varGrp" + ], + "text": "the names of the note rows whose texts the preQTxt (or a group's untyped notes) joins with a blank line, space-separated, in order" + }, + "row": { + "type": "cdl:row", + "on": [ + "stdyDscr" + ], + "subject": "the row's name", + "text": "the type cell of a row with no data column: a metadata row (convention:unregisteredRows metadataRowTypes: start, deviceid, …) or a matrix header (an appearance with carriesData false)" + }, + "row_label": { + "type": "cdl:row_label", + "on": [ + "stdyDscr" + ], + "subject": "the row's name", + "localized": true, + "text": "a cdl:row's label" + }, + "row_hint": { + "type": "cdl:row_hint", + "on": [ + "stdyDscr" + ], + "subject": "the name of a note row or cdl:row", + "localized": true, + "text": "its hint" + }, + "row_relevant": { + "type": "cdl:row_relevant", + "on": [ + "stdyDscr" + ], + "subject": "the name of a note row or cdl:row", + "text": "its relevant (XLSForm XPath)" + }, + "row_appearance": { + "type": "cdl:row_appearance", + "on": [ + "stdyDscr" + ], + "subject": "the name of a note row or cdl:row", + "text": "its appearance" + }, + "position": { + "type": "cdl:position", + "on": [ + "stdyDscr" + ], + "subject": "the name of an intro/outro note or a cdl:row, or the path of a group with no data question", + "text": "where it is: in= after=" + }, + "language": { + "type": "cdl:language", + "on": [ + "stdyDscr" + ], + "subject": "a language's BCP 47 tag", + "text": "the form's own name for it (its column suffix, 'Deutsch (de)'): every language of a form in several, in the form's order; of a form in one, when its name isn't the bare tag" + }, + "other_label": { + "type": "cdl:other_label", + "on": [ + "varGrp[@type='other']" + ], + "localized": true, + "text": "a select_multiple's authored other choice's label: the pair has no binary var for it" } }, "fields": { "ItemBase.name": { - "ddi": "var/@name; a group's path in varGrp/@name" + "ddi": "var/@name; a group's path in varGrp/@name; a note's in cdl:note_names or notes/@subject" }, "ItemBase.label": { "ddi": "qstn/qstnLit; a group's varGrp/txt" }, "ItemBase.hint": { "ddi": "qstn/postQTxt; a group's in a cdl:hint note (varGrp has no hint element)", - "note": "CDL uses postQTxt for the hint shown to respondents with the question, though the XSD describes it as text about what follows the question: DDI 2.5 has no hint element." + "note": "CDL uses postQTxt for the hint shown to respondents with the question, though the XSD describes it as text about what follows the question: DDI 2.5 has no hint element. A note row's hint is a cdl:row_hint." }, "ItemBase.relevant": { "ddi": "universe + cdl:relevant note (convention:logicMapping)" @@ -108,17 +200,17 @@ const conventions = { "ddi": "qstn/@responseDomainType, var/@intrvl, varFormat/@type, varFormat/@category (date, time), var/@dcml='0' (integer)" }, "QuestionItem.rawType": { - "loss": "Rebuilt from type, list and the other pattern; the or_other shorthand comes back as the explicit pair, which behaves the same." + "ddi": "type, list and the other pattern; the or_other shorthand: cdl:or_other" }, "QuestionItem.list": { - "loss": "The list name: questions with identical category sets come back sharing one list." + "cdlNote": "cdl:list", + "note": "Only when the list isn't named after the question (a grid member: after its grid); without it the list has that name. DDI that isn't CDL gets identical category sets as one list." }, "QuestionItem.file": { "ddi": "concept/@vocab (the file is .csv)" }, "QuestionItem.orOther": { - "ddi": "varGrp[@type='other'] (convention:other)", - "note": "Comes back as the explicit pair, see rawType." + "ddi": "varGrp[@type='other'] (convention:other), with cdl:or_other when it was the shorthand" }, "QuestionItem.guidanceHint": { "ddi": "qstn/ivuInstr" @@ -153,20 +245,20 @@ const conventions = { "note": "The choice sheet's other columns are not carried." }, "Instrument.languages": { - "ddi": "xml:lang siblings of every text (convention:languageTagging)" + "ddi": "xml:lang siblings of every text (convention:languageTagging); a language's own name in a cdl:language note" }, "Instrument.defaultLanguage": { "ddi": "codeBook/@xml:lang" }, "Instrument.settings": { - "ddi": "form_title: titl; form_id: IDNo; version: verStmt/version; default_language: codeBook/@xml:lang; style: stdyDscr/notes[@type='cdl:setting'][@subject='style']", - "note": "Other settings are not carried." + "ddi": "form_title: titl (+ parTitl); form_id: IDNo; version: verStmt/version; every other setting, default_language as authored too: stdyDscr/notes[@type='cdl:setting'][@subject=]", + "note": "Settings that are neither a string nor a number are not carried." }, "Instrument.lists": { "ddi": "catgry of each var that uses a list" }, "Instrument.body": { - "ddi": "dataDscr: varGrp and var, in survey order" + "ddi": "dataDscr: varGrp and var, in survey order; note rows in preQTxt (names: cdl:note_names) or as stdyDscr notes type='instruction'; rows without data as cdl:row; their fields in cdl:row_*; where the stdyDscr ones are: cdl:position" } } }, diff --git a/src/instrument/fromDdi.ts b/src/instrument/fromDdi.ts index 089fdc0..c51c98f 100644 --- a/src/instrument/fromDdi.ts +++ b/src/instrument/fromDdi.ts @@ -19,7 +19,11 @@ import { } from '../diagnostics.js'; import { GRID_APPEARANCE } from '../conventions/grid.js'; import { EXCLUSIVE_RULE } from '../conventions/exclusive.js'; -import { OTHER_CODE, otherLabelFor } from '../conventions/other.js'; +import { + OTHER_CODE, + limesurveyOtherText, + otherLabelFor, +} from '../conventions/other.js'; import { fromFileTypeFor } from '../conventions/fromFile.js'; import { TYPE_MAPPINGS } from '../generated/TypeMappings.js'; import { parseParameters } from '../utils/parameters.js'; @@ -58,6 +62,10 @@ interface ReadState { /** Choice set (as a key) → its list's name. */ listByKey: Map; names: Set; + /** A CDL codebook: its lists are named (`cdl:list`), not deduplicated. */ + cdl: boolean; + /** Section / grid items by `varGrp/@name`, as the tree builds them. */ + groupItems: Map; onWarning?: WarningHandler; } @@ -155,14 +163,19 @@ function chainOf(id: string, state: ReadState): string[] { // ── choices ────────────────────────────────────────────────────────────── /** - * A choice set's list: an identical set already read shares its list (the - * list's own name is not in the DDI), else a new one named `preferred`. + * A choice set's list. In a CDL codebook it is `preferred`, the `cdl:list` + * note's name or the question's (#160). In other DDI an identical set already + * read shares its list, else it is a new one named `preferred`. */ function listFor( choices: InstrumentChoice[], preferred: string, state: ReadState, ): string { + if (state.cdl) { + state.lists[preferred] ??= choices; + return preferred; + } const key = JSON.stringify( choices.map((c) => [c.name, c.label, c.row[EXCLUSIVE_RULE.choicesColumn]]), ); @@ -279,11 +292,26 @@ function varType(v: XmlNode, state: ReadState): string { return textType(v); } -/** One plain `var` as a question. */ +/** The list a question names (`cdl:list`), else `named`. */ +function listName(node: XmlNode, named: string, state: ReadState): string { + return noteText(node, FIELDS.list.type, state) || named; +} + +/** The whole type cell: type, list or file, and the `or_other` shorthand. */ +function rawTypeOf(q: QuestionItem, shorthand = q.orOther): string { + const parts = [q.type, q.file || q.list, shorthand ? 'or_other' : '']; + return parts.filter(Boolean).join(' '); +} + +/** + * One plain `var` as a question. With `orOther` it is the shorthand's + * select, whose `other` answer was added, not in its list. + */ function plainQuestion( v: XmlNode, state: ReadState, - listName = v.attrs['name'] ?? '', + named = v.attrs['name'] ?? '', + orOther = false, ): QuestionItem { const q = emptyQuestion(v.attrs['name'] ?? v.attrs['ID'] ?? ''); readQstn(q, v, state); @@ -293,10 +321,14 @@ function plainQuestion( if (vocab) { q.file = `${vocab}.csv`; } else if (q.type === 'select_one' || q.type === 'select_multiple') { - q.list = listFor(categories(v, state), listName, state); + const choices = categories(v, state).filter( + (c) => !orOther || c.name !== OTHER_CODE, + ); + q.list = listFor(choices, listName(v, named, state), state); } + q.orOther = orOther; if (q.type === 'range') q.parameters = rangeParameters(v, q.parameters); - q.rawType = [q.type, q.file || q.list].filter(Boolean).join(' '); + q.rawType = rawTypeOf(q); return q; } @@ -327,6 +359,7 @@ function multiQuestion( q.hint = childTexts(qstn, 'postQTxt', state); q.guidanceHint = childTexts(qstn, 'ivuInstr', state); } + q.orOther = withOther && isShorthand(grp); const excl = exclusive(grp, state); const choices: InstrumentChoice[] = binaries.map((b) => { const code = (b.attrs['name'] ?? '').slice(name.length + 1); @@ -338,31 +371,115 @@ function multiQuestion( : {}, }; }); - if (withOther) { - // Its binary isn't written; the label is the convention's. + if (withOther && !q.orOther) { + // Its binary isn't written: the label is cdl:other_label's, else the convention's. + const own = texts(notesOf(grp, FIELDS.other_label.type), state); const label: Text = {}; for (const lang of state.languages) - label[lang] = otherLabelFor(lang || 'en'); + label[lang] = own[lang] ?? otherLabelFor(lang || 'en'); choices.push({ name: OTHER_CODE, label, row: {} }); } - q.list = listFor(choices, name, state); - q.rawType = `select_multiple ${q.list}`; + q.list = listFor(choices, listName(grp, name, state), state); + q.rawType = rawTypeOf(q); + return q; +} + +/** A semi-open pair whose other answer and companion were added (`cdl:or_other`). */ +function isShorthand(pair: XmlNode): boolean { + return notesOf(pair, FIELDS.or_other.type).length > 0; +} + +/** The type cell's `or_other`: not for an `added` pair (LimeSurvey's other=Y). */ +function inTypeCell(pair: XmlNode): boolean { + return notesOf(pair, FIELDS.or_other.type).some( + (n) => textContent(n).trim() !== 'added', + ); +} + +/** + * The shorthand's "other" text where it isn't LimeSurvey's own (its + * `other_replace_text`): the companion's label. + */ +function otherLabelOf(companion: XmlNode | undefined, state: ReadState): Text { + if (!companion) return {}; + const label = childTexts(childNamed(companion, 'qstn'), 'qstnLit', state); + const own: Text = {}; + for (const [lang, text] of Object.entries(label)) { + if (text !== limesurveyOtherText(lang || 'en') && text !== OTHER_CODE) { + own[lang] = text; + } + } + return own; +} + +/** + * The shorthand select's own "other" text, when it has one, and its type + * cell: `or_other` unless the pair was only said to be there (`added`). + */ +function withOtherLabel( + q: QuestionItem, + pair: XmlNode, + companion: XmlNode | undefined, + state: ReadState, +): QuestionItem { + const own = otherLabelOf(companion, state); + if (Object.keys(own).length) q.otherLabel = own; + q.rawType = rawTypeOf(q, inTypeCell(pair)); return q; } // ── notes ──────────────────────────────────────────────────────────────── -/** A note row, whose name the DDI doesn't keep. */ -function noteItem(label: Text, near: string, state: ReadState): QuestionItem { - let name = `${near}_note`; - for (let i = 2; state.names.has(name); i++) name = `${near}_note_${i}`; +/** A note row: `name`, or without one (not a CDL codebook) `_note`. */ +function noteItem( + label: Text, + near: string, + state: ReadState, + name = '', +): QuestionItem { + if (!name) { + name = `${near}_note`; + for (let i = 2; state.names.has(name); i++) name = `${near}_note_${i}`; + } state.names.add(name); return { ...emptyQuestion(name), type: 'note', rawType: 'note', label }; } -/** A lead-in note kept as `preQTxt` (standalone) or an untyped group note. */ -function leadIn(label: Text, near: string, state: ReadState): QuestionItem[] { - return Object.keys(label).length ? [noteItem(label, near, state)] : []; +/** + * Each language's lead-in text split back into the notes `names` lists: at + * the blank lines that joined them, else (a count that doesn't match) all of + * it the first note's. + */ +function splitNotes(label: Text, names: string[]): Text[] { + const out: Text[] = names.map(() => ({})); + for (const [lang, text] of Object.entries(label)) { + const parts = text.split('\n\n'); + if (parts.length === names.length) { + parts.forEach((part, i) => (out[i][lang] = part)); + } else { + out[0][lang] = text; + } + } + return out; +} + +/** + * A lead-in kept as `preQTxt` (standalone) or an untyped group note on `el`: + * the note rows `cdl:note_names` lists, else one note. + */ +function leadIn( + label: Text, + el: XmlNode, + near: string, + state: ReadState, +): QuestionItem[] { + if (!Object.keys(label).length) return []; + const names = ids(noteText(el, FIELDS.note_names.type, state)); + if (names.length < 2) return [noteItem(label, near, state, names[0])]; + return splitNotes(label, names) + .map((text, i) => ({ text, name: names[i] })) + .filter(({ text }) => Object.keys(text).length) + .map(({ text, name }) => noteItem(text, near, state, name)); } const untypedNotes = (node: XmlNode, state: ReadState) => @@ -388,11 +505,15 @@ function varsOf(grp: XmlNode, state: ReadState): XmlNode[] { } /** A `var`'s lead-in note (its `preQTxt`) and the question itself. */ -function withLeadIn(v: XmlNode, state: ReadState): QuestionItem[] { +function withLeadIn( + v: XmlNode, + state: ReadState, + orOther = false, +): QuestionItem[] { const note = childTexts(childNamed(v, 'qstn'), 'preQTxt', state); return [ - ...leadIn(note, v.attrs['name'] ?? '', state), - plainQuestion(v, state), + ...leadIn(note, v, v.attrs['name'] ?? '', state), + plainQuestion(v, state, undefined, orOther), ]; } @@ -415,9 +536,11 @@ function placeMulti(multiId: string, multi: XmlNode, state: ReadState): Placed { const pair = pairId ? state.groups.get(pairId) : undefined; const q = multiQuestion(pair ?? multi, varsOf(multi, state), !!pair, state); const lead = untypedNotes(pair ?? multi, state); - const items = [...leadIn(lead, q.name, state), q]; - if (pair) - items.push(...varsOf(pair, state).map((m) => plainQuestion(m, state))); + const items = [...leadIn(lead, pair ?? multi, q.name, state), q]; + const companions = pair ? varsOf(pair, state) : []; + // The shorthand's companion was added: the select's or_other says it. + if (q.orOther) withOtherLabel(q, pair!, companions[0], state); + else items.push(...companions.map((m) => plainQuestion(m, state))); return { items, anchor: pairId ?? multiId }; } @@ -441,6 +564,12 @@ function placeVar( seen.add(key); if (multiId) return placeMulti(multiId, state.groups.get(multiId)!, state); // A semi-open select_one: its `other` group holds the select and its text. + const [select, ...companions] = varsOf(owner, state); + if (select && isShorthand(owner)) { + const items = withLeadIn(select, state, true); + withOtherLabel(items[items.length - 1], owner, companions[0], state); + return { items, anchor: ownerId }; + } return { items: varsOf(owner, state).flatMap((m) => withLeadIn(m, state)), anchor: ownerId, @@ -458,9 +587,9 @@ function groupItem(id: string, state: ReadState): GroupItem { const grp = state.groups.get(id)!; const grid = grp.attrs['type'] === 'grid'; const lead = grid - ? leadIn(untypedNotes(grp, state), gridName(id, state), state) + ? leadIn(untypedNotes(grp, state), grp, gridName(id, state), state) : []; - return { + const item: GroupItem = { kind: 'group', name: gridName(id, state), label: childTexts(grp, 'txt', state), @@ -473,6 +602,8 @@ function groupItem(id: string, state: ReadState): GroupItem { children: lead, closed: true, }; + state.groupItems.set(grp.attrs['name'] ?? id, item); + return item; } /** Seat each placed question in its groups, opening them as they come. */ @@ -525,7 +656,6 @@ function readSettings( ['form_title', citationText(stdy, ['titlStmt', 'titl'])], ['form_id', citationText(stdy, ['titlStmt', 'IDNo'])], ['version', citationText(stdy, ['verStmt', 'version'])], - ['default_language', state.base], ]; const out: Record = Object.fromEntries( fields.filter(([, value]) => value && value !== 'Untitled'), @@ -543,31 +673,199 @@ function readSettings( const key = note.attrs['subject']; if (key) out[key] = textContent(note).trim(); } + // Kobo's older name for form_id, kept as authored. + if (out['id_string'] === out['form_id']) delete out['form_id']; return out; } -/** The study's notes (`type="instruction"`): notes with no question after them. */ -function orphanNotes( - root: XmlNode | undefined, +/** A note or data-less row the study describes, from its type cell. */ +function studyItem(name: string, typeCell: string, label: Text): QuestionItem { + const [type = 'note', second = '', third = ''] = typeCell.split(/\s+/); + const q = { ...emptyQuestion(name), type, rawType: typeCell, label }; + if (type.endsWith('_from_file')) q.file = second; + else if (second) q.list = second; + q.orOther = second === 'or_other' || third === 'or_other'; + return q; +} + +/** The study's notes of one `cdl:` type, by subject. */ +function bySubject(stdy: XmlNode, type: string): Map { + const out = new Map(); + for (const note of notesOf(stdy, type)) { + const subject = note.attrs['subject'] ?? ''; + out.set(subject, [...(out.get(subject) ?? []), note]); + } + return out; +} + +/** + * The study's rows with no question after them: its notes + * (`type="instruction"`, intros and outros) and rows without data + * (`cdl:row`, with `cdl:row_label`), by name. + */ +function studyRows( + stdy: XmlNode | undefined, state: ReadState, -): QuestionItem[] { - const stdy = root && childNamed(root, 'stdyDscr'); - if (!stdy) return []; - const bySubject = new Map(); - for (const note of childrenNamed(stdy, 'notes')) { - if (note.attrs['type'] !== 'instruction') continue; +): Map { + const rows = new Map(); + if (!stdy) return rows; + const labels = bySubject(stdy, FIELDS.row_label.type); + for (const [subject, notes] of bySubject(stdy, 'instruction')) { + rows.set(subject, studyItem(subject, 'note', texts(notes, state))); + } + for (const [subject, [note]] of bySubject(stdy, FIELDS.row.type)) { + if (!subject) continue; + const label = texts(labels.get(subject) ?? [], state); + rows.set(subject, studyItem(subject, textContent(note).trim(), label)); + } + for (const name of rows.keys()) state.names.add(name); + return rows; +} + +/** A section with no data question under it (its path `name`), not yet placed. */ +function emptyGroup(name: string, state: ReadState): GroupItem | undefined { + if (state.groupItems.has(name)) return undefined; + for (const [id, grp] of state.groups) { + if (grp.attrs['name'] === name && grp.attrs['type'] === 'section') { + return groupItem(id, state); + } + } + return undefined; +} + +/** A `cdl:position` note's `in=… after=…`. */ +function positionOf(note: XmlNode): { in: string; after: string } { + const values = Object.fromEntries( + textContent(note) + .trim() + .split(/\s+/) + .map((token) => { + const at = token.indexOf('='); + return at < 0 ? [token, ''] : [token.slice(0, at), token.slice(at + 1)]; + }), + ); + return { in: values['in'] ?? '', after: values['after'] ?? '' }; +} + +/** + * Seat the study's rows where their `cdl:position` says (#160): in their + * group after the item named, first when none is. A row without a position, + * or whose group isn't in the codebook, ends the survey. + */ +function placeStudyRows( + body: Item[], + stdy: XmlNode | undefined, + state: ReadState, +): void { + const rows = studyRows(stdy, state); + const positions = stdy ? notesOf(stdy, FIELDS.position.type) : []; + for (const note of positions) { const subject = note.attrs['subject'] ?? ''; - bySubject.set(subject, [...(bySubject.get(subject) ?? []), note]); + const item = rows.get(subject) ?? emptyGroup(subject, state); + if (!item) continue; + rows.delete(subject); + const at = positionOf(note); + const group = at.in ? state.groupItems.get(at.in) : undefined; + const into = group ? group.children : body; + if (at.in && !group) { + into.push(item); + continue; + } + const after = at.after ? into.findIndex((i) => i.name === at.after) : -1; + if (at.after && after < 0) into.push(item); + else into.splice(after + 1, 0, item); } - return [...bySubject].map(([subject, notes]) => { - state.names.add(subject); - return { - ...emptyQuestion(subject), - type: 'note', - rawType: 'note', - label: texts(notes, state), - }; - }); + body.push(...rows.values()); +} + +/** The base-language text of the first note of `name`, `''` if none. */ +function subjectText(notes: Map, name: string): string { + const note = notes.get(name)?.[0]; + return note ? textContent(note).trim() : ''; +} + +/** + * The hint, relevant and appearance of note rows and data-less rows + * (`cdl:row_hint`, `cdl:row_relevant`, `cdl:row_appearance`), by name: only + * those rows have them. + */ +function readRowFields( + items: Item[], + stdy: XmlNode | undefined, + state: ReadState, +): void { + if (!stdy) return; + const hints = bySubject(stdy, FIELDS.row_hint.type); + const relevants = bySubject(stdy, FIELDS.row_relevant.type); + const appearances = bySubject(stdy, FIELDS.row_appearance.type); + const visit = (list: Item[]) => { + for (const item of list) { + if (item.kind === 'group') { + visit(item.children); + continue; + } + const hint = hints.get(item.name); + if (hint) item.hint = texts(hint, state); + item.relevant ||= subjectText(relevants, item.name); + item.appearance ||= subjectText(appearances, item.name); + } + }; + visit(items); +} + +/** `cdl:language`: each language tag's name in the form (its column suffix). */ +function languageNames(stdy: XmlNode | undefined): Map { + const names = new Map(); + for (const note of stdy ? notesOf(stdy, FIELDS.language.type) : []) { + const tag = note.attrs['subject']; + const name = textContent(note).trim(); + if (tag && name) names.set(tag, name); + } + return names; +} + +/** A text's languages by the form's names for them. */ +function renamed(text: Text, names: Map): Text { + return Object.fromEntries( + Object.entries(text).map(([lang, value]) => [ + names.get(lang) ?? lang, + value, + ]), + ); +} + +/** Every text of an item (and its children) by the form's language names. */ +function renameItem(item: Item, names: Map): void { + item.label = renamed(item.label, names); + item.hint = renamed(item.hint, names); + if (item.kind === 'group') { + item.children.forEach((c) => renameItem(c, names)); + return; + } + item.guidanceHint = renamed(item.guidanceHint, names); + item.constraintMessage = renamed(item.constraintMessage, names); + if (item.otherLabel) item.otherLabel = renamed(item.otherLabel, names); +} + +/** + * The instrument's texts keyed by the form's language names (`cdl:language`) + * instead of the bare tags, as its columns were (`label::Deutsch (de)`). + */ +function renameLanguages( + instrument: Instrument, + names: Map, +): Instrument { + if (!names.size) return instrument; + instrument.body.forEach((item) => renameItem(item, names)); + for (const choices of Object.values(instrument.lists)) { + for (const c of choices) c.label = renamed(c.label, names); + } + const title = instrument.settings['form_title']; + if (title && typeof title === 'object') { + instrument.settings['form_title'] = renamed(title as Text, names); + } + instrument.languages = instrument.languages.map((l) => names.get(l) ?? l); + return instrument; } // ── provenance ─────────────────────────────────────────────────────────── @@ -638,17 +936,24 @@ export function instrumentFromDdi( lists: {}, listByKey: new Map(), names: new Set(), + cdl: false, + groupItems: new Map(), onWarning: options.onWarning, }; + const stdy = root && childNamed(root, 'stdyDscr'); + const names = languageNames(stdy); + // The form's languages in its order (cdl:language), else as they come. + if (names.size) state.languages = [...names.keys()]; const elements = dataElements(roots); indexStructure(elements, state); + state.cdl = isCdl(elements); if (!root && state.vars.size === 0) { throw new ConversionError( 'ddi-invalid', 'The input holds no and no .', ); } - if (!isCdl(elements)) warnNotCdl(options.onWarning); + if (!state.cdl) warnNotCdl(options.onWarning); for (const [, v] of state.vars) state.names.add(v.attrs['name'] ?? ''); const seen = new Set(); @@ -658,12 +963,17 @@ export function instrumentFromDdi( if (p) placed.push(p); } const settings = readSettings(root, state); - const body = [...buildTree(placed, state), ...orphanNotes(root, state)]; - return { - languages: languagesOf(state), - ...(base ? { defaultLanguage: base } : {}), - settings, - lists: state.lists, - body, - }; + const body = buildTree(placed, state); + placeStudyRows(body, stdy, state); + readRowFields(body, stdy, state); + return renameLanguages( + { + languages: languagesOf(state), + ...(base ? { defaultLanguage: base } : {}), + settings, + lists: state.lists, + body, + }, + names, + ); } diff --git a/src/instrument/fromLstsv.ts b/src/instrument/fromLstsv.ts index 3f96f01..1139c9e 100644 --- a/src/instrument/fromLstsv.ts +++ b/src/instrument/fromLstsv.ts @@ -11,6 +11,11 @@ */ import { OTHER_CODE, OTHER_SUFFIX } from '../conventions/other.js'; import { GRID_APPEARANCE } from '../conventions/grid.js'; +import { EXCLUSIVE_RULE } from '../conventions/exclusive.js'; +import type { APPEARANCES } from '../generated/Appearances.js'; + +/** The registry appearance for a group shown as one page. */ +const PAGE_APPEARANCE: keyof typeof APPEARANCES = 'field-list'; import { fromFileTypeFor, vocabFromCssClass } from '../conventions/fromFile.js'; import { resolveType } from './lstsvTypes.js'; import { @@ -181,6 +186,14 @@ function questionItem(row: Row, state: ParseState): QuestionItem { }; } +/** Whether a Q row's `exclude_all_others` lists `code`. */ +function excludes(row: Record, code: string): boolean { + const value = row[EXCLUSIVE_RULE.limesurveyAttribute]; + return (typeof value === 'string' ? value : '') + .split(EXCLUSIVE_RULE.limesurveySeparator) + .some((c) => c.trim() === code); +} + /** * An A/SQ row: an option of `owner`'s list. A multiple choice's defaults sit * on its SQ rows as `Y`. @@ -197,10 +210,18 @@ function addChoice( .filter(Boolean) .join(' '); } + const code = cell(row, 'name'); (state.lists[owner] ??= []).push({ - name: cell(row, 'name'), - label: text(state, `label:${cls}:${owner}:${cell(row, 'name')}`), - row, + name: code, + label: text(state, `label:${cls}:${owner}:${code}`), + // The question's exclude_all_others, on the choice as an XLSForm has it. + row: + question && excludes(question.row, code) + ? { + ...row, + [EXCLUSIVE_RULE.choicesColumn]: EXCLUSIVE_RULE.trueValues[0], + } + : row, }); } @@ -400,6 +421,18 @@ export interface LstsvParseOptions { onWarning?: WarningHandler; } +/** + * `format=G` shows each group as one page, which is what an XLSForm + * `field-list` group means: every group that isn't a grid is one. + */ +function markPages(items: Item[]): void { + for (const item of items) { + if (item.kind !== 'group') continue; + if (!item.appearance) item.appearance = PAGE_APPEARANCE; + markPages(item.children); + } +} + /** Parse LimeSurvey structure-TSV rows into an Instrument. */ export function instrumentFromLstsv( rows: Row[], @@ -437,15 +470,14 @@ export function instrumentFromLstsv( if (options.expressions) { reverseAllExpressions(body, state.lists, options.onWarning); } + const pages = cell(setting('format') ?? {}, 'text') === 'G'; + if (pages) markPages(body); return { languages: languages.length ? languages : [''], ...(base ? { defaultLanguage: base } : {}), settings: { ...(title ? { form_title: cell(title, 'text') } : {}), - ...(base ? { default_language: base } : {}), - ...(cell(setting('format') ?? {}, 'text') === 'G' - ? { style: 'pages' } - : {}), + ...(pages ? { style: 'pages' } : {}), }, lists: state.lists, body, diff --git a/src/pipelines/README.md b/src/pipelines/README.md index e28bd86..9b7c431 100644 --- a/src/pipelines/README.md +++ b/src/pipelines/README.md @@ -78,6 +78,10 @@ What a CDL codebook carries beyond the canonical `Variable` `qstn/backward`), else a typed note. `convention:ddiFields` ([`registry/conventions/ddiFields.jsonld`](../../registry/conventions/ddiFields.jsonld)) maps every model field, or names it a loss. +- **what a form needs to be rebuilt** (#160): list names (`cdl:list`), the + `or_other` shorthand (`cdl:or_other`), note rows' names, fields and places, + rows without data (`cdl:row`), every setting and the form's language names, + so a codebook's XLSForm converts to the same codebook. DDI formtransform didn't write (a hand-written seed study, another tool's codebook) has none of the `cdl:` notes. `ddi2xlsform` still converts it, as a diff --git a/src/pipelines/ddi2xlsform/README.md b/src/pipelines/ddi2xlsform/README.md index 4f3a468..ccfe57c 100644 --- a/src/pipelines/ddi2xlsform/README.md +++ b/src/pipelines/ddi2xlsform/README.md @@ -23,29 +23,31 @@ Standard DDI first, `cdl:` notes where DDI has no element | XLSForm | DDI | |---|---| -| type | `qstn/@responseDomainType`; `varFormat/@category` (date, time); `var/@dcml="0"` (integer); `valrng` without a constraint (range); `concept/@vocab` (`select_*_from_file`) | +| type | `qstn/@responseDomainType`; `varFormat/@category` (date, time); `var/@dcml="0"` (integer); `valrng` without a constraint (range); `concept/@vocab` (`select_*_from_file`); the `or_other` shorthand: `cdl:or_other` | | label / hint / guidance_hint | `qstnLit` / `postQTxt` / `ivuInstr`, every `xml:lang` | -| a note row before a question | its `preQTxt` (a grid member's is the grid's text) or its group's untyped `notes` | -| choices | `catgry` (`catValu`, `labl`); a select_multiple's binary `var`s | -| groups | `varGrp type="section"` / `"grid"`, nested by `@varGrp`; label `txt`, hint `cdl:hint` | +| a note row before a question | its `preQTxt` (a grid member's is the grid's text) or its group's untyped `notes`; the rows' names in `cdl:note_names` | +| a note row with no question after it in its group | `stdyDscr/notes[@type='instruction']`, placed by `cdl:position` | +| a note row's hint, relevant, appearance | `cdl:row_hint`, `cdl:row_relevant`, `cdl:row_appearance` on `stdyDscr` | +| rows without data (`start`, `deviceid`, …, a matrix header) | `cdl:row` (the type cell) and `cdl:row_label` on `stdyDscr`, placed by `cdl:position` | +| choices | `catgry` (`catValu`, `labl`); a select_multiple's binary `var`s; the list's name in `cdl:list` when it isn't the question's; a select_multiple pair's other label in `cdl:other_label` | +| groups | `varGrp type="section"` / `"grid"`, nested by `@varGrp`; label `txt`, hint `cdl:hint`; a group of notes only placed by `cdl:position` | | order | `var` order, `qstn/@seqNo` | | relevant, constraint, constraint_message, required | `cdl:relevant`, `cdl:constraint`, `cdl:constraint_message`, `cdl:required` | -| default, appearance, parameters | `cdl:default`, `cdl:appearance`, `cdl:parameters`; a range's `valrng/range` | +| default, appearance, parameters | `cdl:default`, `cdl:appearance`, `cdl:parameters` (a range's bounds also `valrng/range`) | | exclusive | `cdl:exclusive` on the select_multiple's `varGrp` | -| settings | `titl` (+ `parTitl` per language), `IDNo`, `verStmt/version`, `codeBook/@xml:lang`, `cdl:setting` | +| settings | `titl` (+ `parTitl` per language), `IDNo`, `verStmt/version`, `codeBook/@xml:lang`, every other setting a `cdl:setting` | +| language columns | `xml:lang`; the form's name for each (`label::Deutsch (de)`) in `cdl:language` | ## Input it accepts - **A codebook formtransform wrote** gives back its form, up to the losses - below. `tests/ts/contract/ddiRoundtrip.test.ts` checks every fixture on the - Instrument model, and `ddiRoundtripGenerated.test.ts` checks random forms - (fast-check). Every survey's output is pinned as `ddi2xlsform.json` and - validated with pyxform. + below. Its XLSForm converts back to the same codebook, and to the same + LimeSurvey TSV as the original form (see Tests). - **Any other DDI** is never refused. It is read as far as its standard elements go. Without `qstn/@seqNo` or any `cdl:` note it is not a CDL - codebook, and each field only CDL carries gets one `ddi-field-missing` - warning. An unknown `responseDomainType` is read as text - (`ddi-type-unknown`). + codebook: identical category sets come back as one list, and each field + only CDL carries gets one `ddi-field-missing` warning. An unknown + `responseDomainType` is read as text (`ddi-type-unknown`). - **A fragment**, i.e. a `` or bare `` / `` elements, is read the same way. A `varGrp` that refers to a `var` outside the fragment gets `ddi-reference-outside`. @@ -54,32 +56,34 @@ Standard DDI first, `cdl:` notes where DDI has no element ## Known losses -What a CDL codebook doesn't give back (the round-trip tests fold these out): +What a CDL codebook doesn't give back (`canonicalInstrument.ts` folds these +out): -1. **List names.** Identical category sets come back as one list, named - after the first question that uses it (a grid's after the grid). -2. **The `or_other` shorthand** comes back as the explicit pair (an `other` - choice plus a `_other` text question with its `relevant`), which - behaves the same. The `other` choice's label is what LimeSurvey shows - (`convention:other` `limesurveyOtherText`), or for a select_multiple the - convention's label. -3. **Note rows:** - - Their names are rebuilt as `_note`. - - Consecutive notes before one question come back as one note. - - A note with no question after it in its group (an intro or outro) comes - back at the end of the survey, under its own name. - - A note's hint is lost. -4. **Rows DDI has no variable for**: device and session metadata (`start`, - `end`, `deviceid`, …), matrix header rows, `calculate` and other - unregistered types. A group with none of its questions left is dropped. -5. **Language names.** Columns use the BCP 47 tag (`label::de`), and - `default_language` gets an English name (`German (de)`). -6. **Defaults are written as defaults.** A range's `start`/`end` equal to - the registry defaults (1, 10) aren't written back. A group without a label - comes back labelled with its name. `required` is `yes` or absent. -7. A `constraint_message` without a `constraint` is not in the DDI. -8. **Choice columns** other than `exclusive` (media, filters) and **settings** - other than `form_title`, `form_id`, `version`, `default_language` and - `style` are not carried. -9. **The title.** A codebook built with an explicit study title (`assetName`) - has that title, not `form_title`. +1. **Rows no registry type covers** (`calculate`, …) and **groups with + nothing in them**. +2. **Columns the model doesn't lift**: on the survey sheet (media, + `calculation`, `choice_filter`, …), on the choices sheet every column but + `exclusive`. Settings that are neither a string nor a number. +3. **A group without a label** comes back labelled with its name. +4. **A `constraint_message` without a `constraint`** is not in the DDI. +5. **Cell spellings**: `required` is `yes` or absent, a `guidance_hint` inside + `parameters` comes back as the `guidance_hint` column, whitespace in the + type cell and `parameters` is one space. +6. **Consecutive note rows** before a question are split back at the blank + lines that joined them. A note whose text has a blank line itself, or a + language only some of them have, gives the whole text to the first. + +## Tests + +- `tests/ts/contract/ddiRoundtrip.test.ts`, on every whole-survey fixture and + registry entity: XLSForm → DDI → Instrument and XLSForm → DDI → XLSForm give + the form back, and DDI → XLSForm → DDI gives the same codebook, byte for + byte. +- `tests/ts/contract/lstsvDdiRoundtrip.test.ts`, through LimeSurvey: a TSV → + DDI → Instrument is the TSV's own; XLSForm → LimeSurvey → DDI → XLSForm is + XLSForm → LimeSurvey → XLSForm; XLSForm → DDI → XLSForm → LimeSurvey is + XLSForm → LimeSurvey, byte for byte. +- `tests/ts/contract/ddiRoundtripGenerated.test.ts`: all of these on random + forms (fast-check, 200 per property; `FT_ROUNDTRIP_RUNS` for more). +- Every survey's output is pinned as `ddi2xlsform.json` and validated with + pyxform. diff --git a/src/pipelines/lstsv2ddi/index.ts b/src/pipelines/lstsv2ddi/index.ts index 5801942..13440d7 100644 --- a/src/pipelines/lstsv2ddi/index.ts +++ b/src/pipelines/lstsv2ddi/index.ts @@ -75,15 +75,18 @@ export function lstsvToDdiXml( (r) => r.class?.trim() === 'SL' && r.name?.trim() === 'surveyls_title', )?.text; - const { variables, language } = lstsvProjection(rows, onWarning); - const opts: BuildDdiOptions = { ...ddiOptions }; - if (!opts.assetName && title?.trim()) { - opts.settings = { form_title: title.trim(), ...opts.settings }; - } - // The untagged texts are the base language; say which (#135). - if (language) { - opts.settings = { default_language: language, ...opts.settings }; - } + const { variables, language, settings, languages } = lstsvProjection( + rows, + onWarning, + ); + // The untagged texts are the survey's language; say which (#135, #160). + const opts: BuildDdiOptions = { + language, + languageNames: Object.fromEntries(languages.map((l) => [l, l])), + ...ddiOptions, + }; + opts.settings = { ...settings, ...opts.settings }; + if (opts.assetName || !title?.trim()) delete opts.settings.form_title; return buildDdiCodebook(variables, opts).toDocument(); } diff --git a/src/pipelines/lstsv2ddi/toVariables.ts b/src/pipelines/lstsv2ddi/toVariables.ts index 494e296..932b2d9 100644 --- a/src/pipelines/lstsv2ddi/toVariables.ts +++ b/src/pipelines/lstsv2ddi/toVariables.ts @@ -27,11 +27,10 @@ export function lstsvToVariables(rows: Row[]): Variable[] { /** * {@link lstsvToVariables} with relevance and constraints reversed into XPath * for the DDI's logic (#151): one outside the dialect gets a warning and is - * left out, rather than failing the conversion. Plus, for a survey in several languages, its base - * language (the `language` S row, as a BCP 47 tag): the DDI declares it as - * `codeBook/@xml:lang` so the untagged texts say their language. A - * single-language survey's DDI stays undeclared, as an XLSForm's without - * `default_language` does. + * left out, rather than failing the conversion. Plus the survey's base + * language (the `language` S row, as a BCP 47 tag), which the DDI declares + * as `codeBook/@xml:lang` so the untagged texts say their language, and its + * settings (title, style). */ export function lstsvProjection( rows: Row[], @@ -39,6 +38,9 @@ export function lstsvProjection( ): { variables: Variable[]; language?: string; + settings: Record; + /** The survey's languages, base first. */ + languages: string[]; } { const instrument = instrumentFromLstsv(rows, { expressions: true, @@ -50,7 +52,8 @@ export function lstsvProjection( choicesFromInstrument(instrument), { onWarning }, ), - language: - instrument.languages.length > 1 ? instrument.defaultLanguage : undefined, + language: instrument.defaultLanguage, + settings: instrument.settings, + languages: instrument.languages.filter(Boolean), }; } diff --git a/src/pipelines/lstsv2xlsform/README.md b/src/pipelines/lstsv2xlsform/README.md index fa0fde0..8efe3cc 100644 --- a/src/pipelines/lstsv2xlsform/README.md +++ b/src/pipelines/lstsv2xlsform/README.md @@ -145,7 +145,8 @@ listed here must match exactly. tool produces contains a calculation question: there is nothing to reverse. 9. **Group appearance is read from the survey format.** The TSV stores no per-group appearance, only `format=G` (one page per group, from - `style: pages`). With it, every non-grid group comes back `field-list`; + `style: pages`). With it, the parser makes every non-grid group + `field-list`; without it, none does. A `field-list` group in a survey without `style: pages` is lost, and so is a `style: pages` group that wasn't one. 10. **`true()` / `false()` come back as `1` / `0`.** The forward writes EM diff --git a/src/pipelines/xlsform2ddi/index.ts b/src/pipelines/xlsform2ddi/index.ts index 96b5076..fb25f4f 100644 --- a/src/pipelines/xlsform2ddi/index.ts +++ b/src/pipelines/xlsform2ddi/index.ts @@ -34,17 +34,25 @@ function ddiLanguage( : undefined; } +/** The form's languages: tag → the form's own name (its column suffix). */ +function languageNames( + surveyRows: Record[], +): Record { + const names: Record = {}; + for (const key of instrumentFromXlsform(surveyRows).languages) { + const tag = key ? languageTagOf(key) : null; + if (tag) names[tag] ??= key; + } + return names; +} + /** * Without `default_language`, a form in several languages has the first as * its base: declare it, so the untagged DDI texts say their language (#135). */ -function multilingualBase( - surveyRows: Record[], -): string | undefined { - const tags = instrumentFromXlsform(surveyRows) - .languages.map((l) => (l ? languageTagOf(l) : null)) - .filter((t): t is string => !!t); - return new Set(tags).size > 1 ? tags[0] : undefined; +function multilingualBase(names: Record): string | undefined { + const tags = Object.keys(names); + return tags.length > 1 ? tags[0] : undefined; } export function buildDdiXml( @@ -66,16 +74,12 @@ export function buildDdiXml( language, onWarning, }); - const base = language ?? multilingualBase(surveyRows); - return buildDdiCodebook( - variables, - base && !language - ? { - ...options, - settings: { ...options.settings, default_language: base }, - } - : options, - ).toDocument(); + const names = languageNames(surveyRows); + return buildDdiCodebook(variables, { + languageNames: names, + language: multilingualBase(names), + ...options, + }).toDocument(); } export { diff --git a/src/xlsform/fromInstrument.ts b/src/xlsform/fromInstrument.ts index 167d990..e864165 100644 --- a/src/xlsform/fromInstrument.ts +++ b/src/xlsform/fromInstrument.ts @@ -1,8 +1,8 @@ /** * {@link Instrument} → XLSForm sheets (#69, phase 4): the emitter both * reverse paths share, `lstsv2xlsform` and `ddi2xlsform` (#154). It writes - * what the model holds; what a source can't carry (LimeSurvey's list names, - * a DDI's note names) the parsers fill in and their READMEs list. + * what the model holds; what a source can't carry (LimeSurvey's list names) + * the parsers fill in and their READMEs list. * * `relevant` / `constraint` are XPath already (the LimeSurvey parser reverses * EM, the DDI parser reads its `cdl:` notes). `calculation` is out of scope: @@ -11,7 +11,6 @@ import { defaultConfig } from '../config/types.js'; import type { SurveyRow, ChoiceRow, SettingsRow } from './types.js'; -import { APPEARANCES } from '../generated/Appearances.js'; import { EXCLUSIVE_RULE, isExclusive } from '../conventions/exclusive.js'; import { OTHER_CODE, @@ -22,9 +21,6 @@ import { } from '../conventions/other.js'; import { GRID_APPEARANCE } from '../conventions/grid.js'; -/** The registry appearance for a group shown as one page. */ -const PAGE_APPEARANCE: keyof typeof APPEARANCES = 'field-list'; - import { formatDefaultLanguage } from './languageNames.js'; import type { GroupItem, @@ -34,6 +30,7 @@ import type { Text, } from '../instrument/types.js'; import { htmlToMarkdown } from '../utils/markdownRenderer.js'; +import { languageTagOf } from '../utils/languageUtils.js'; type Row = Record; @@ -90,8 +87,26 @@ interface EmitCtx { choices: ChoiceRow[]; /** Lists already written: a list shared by several questions is one list. */ written: Set; - /** `format=G` (style: pages): each LimeSurvey group is one page. */ - pages: boolean; +} + +/** + * `default_language`: as the form had it (`Deutsch (de)`, `de`), else the + * base language's English name when it isn't the default (`German (de)`). + */ +function defaultLanguageCell( + instrument: Instrument, + baseLanguage: string, +): string { + const authored = instrument.settings['default_language']; + if ( + typeof authored === 'string' && + (languageTagOf(authored) === baseLanguage || !instrument.defaultLanguage) + ) { + return authored; + } + return baseLanguage === defaultConfig.defaults.language + ? '' + : formatDefaultLanguage(baseLanguage); } /** Render the settings sheet — non-default values only. */ @@ -100,9 +115,8 @@ function buildSettingsRow( baseLanguage: string, ): SettingsRow[] { const row: SettingsRow = {}; - if (baseLanguage !== defaultConfig.defaults.language) { - row.default_language = formatDefaultLanguage(baseLanguage); - } + const language = defaultLanguageCell(instrument, baseLanguage); + if (language) row.default_language = language; const title = instrument.settings['form_title']; if ( typeof title === 'string' && @@ -114,8 +128,8 @@ function buildSettingsRow( // A title per language, as the form's own `form_title` was. row.form_title = title as Record; } - for (const key of ['form_id', 'version', 'style']) { - const value = instrument.settings[key]; + for (const [key, value] of Object.entries(instrument.settings)) { + if (key === 'form_title' || key === 'default_language') continue; if ((typeof value === 'string' && value) || typeof value === 'number') { row[key] = String(value); } @@ -138,7 +152,6 @@ export function xlsformFromInstrument(instrument: Instrument): XlsformOutput { survey: [], choices: [], written: new Set(), - pages: instrument.settings['style'] === 'pages', }; const groups = instrument.body.filter((i) => i.kind === 'group'); const content = instrument.body.filter((i) => !isMessageNote(i)); @@ -205,9 +218,7 @@ function emitGroup(group: GroupItem, ctx: EmitCtx): void { }; const hint = ctx.label(group.hint); if (hint) row.hint = htmlLabel(hint); - // A page per group is what XLSForm's field-list group means. - const appearance = group.appearance || (ctx.pages ? PAGE_APPEARANCE : ''); - if (appearance) row.appearance = appearance; + if (group.appearance) row.appearance = group.appearance; if (group.relevant) row.relevant = group.relevant; ctx.survey.push(row); emitItems(group.children, ctx); @@ -224,9 +235,11 @@ function emitQuestion(q: QuestionItem, siblings: Item[], ctx: EmitCtx): void { return; } const source = q.row as Row; + // Without the shorthand in its type (LimeSurvey's other=Y), the explicit pair. + const expand = q.orOther && !/\sor_other\s*$/.test(q.rawType); if (q.list && !ctx.written.has(q.list)) { emitChoiceList(q.list, ctx, exclusiveCodes(source)); - if (q.orOther) { + if (expand) { ctx.choices.push({ list_name: q.list, name: OTHER_CODE, @@ -237,7 +250,7 @@ function emitQuestion(q: QuestionItem, siblings: Item[], ctx: EmitCtx): void { ctx.survey.push(questionRow(q, ctx)); const companion = `${q.name}${OTHER_SUFFIX}`; - if (q.orOther && !siblings.some((s) => s.name === companion)) { + if (expand && !siblings.some((s) => s.name === companion)) { const label = ctx.label(q.otherLabel) || perLanguageOtherLabel(ctx.languages); ctx.survey.push({ diff --git a/tests/fixtures/surveys/all_types_survey/ddi.xml b/tests/fixtures/surveys/all_types_survey/ddi.xml index 275700d..82b3d27 100644 --- a/tests/fixtures/surveys/all_types_survey/ddi.xml +++ b/tests/fixtures/surveys/all_types_survey/ddi.xml @@ -12,6 +12,27 @@ Im nächsten Abschnitt bitten wir Sie um Ihr ausführliches Feedback. Bitte nehmen Sie sich Zeit. Vielen Dank für Ihre Teilnahme an dieser Umfrage! + start + end + today + deviceid + username + hidden + audit + select_one proficiency + Wie schätzen Sie Ihre Kenntnisse ein? + label + in= after= + in= after=start + in= after=end + in= after=today + in= after=deviceid + in= after=username + in= after=hidden1 + in=basic_types after=q_select_multi + in=matrix_wrapper/matrix_inner after= + in= after=features + de @@ -52,14 +73,18 @@ An welchen Tracks haben Sie teilgenommen? An welchen Tracks haben Sie teilgenommen? + topics Welcher Track hat Ihnen am besten gefallen? Welcher Track hat Ihnen am besten gefallen? + shorthand Welche Tracks würden Sie erneut besuchen? Welche Tracks würden Sie erneut besuchen? + topics + shorthand Welche Tracks würden Sie erneut besuchen? @@ -135,6 +160,7 @@ Würden Sie diesen Workshop weiterempfehlen? + yesno @@ -213,6 +239,7 @@ Wie zufrieden waren Sie mit der Organisation insgesamt? likert + satisfaction @@ -229,6 +256,7 @@ Bevorzugst du Dropdown? minimal + yesno @@ -253,6 +281,7 @@ Python list-nolabel + proficiency @@ -277,6 +306,7 @@ JavaScript list-nolabel + proficiency @@ -301,6 +331,7 @@ SQL list-nolabel + proficiency @@ -324,6 +355,7 @@ Welcher Track hat Ihnen am besten gefallen? + topics diff --git a/tests/fixtures/surveys/all_types_survey/ddi2xlsform.json b/tests/fixtures/surveys/all_types_survey/ddi2xlsform.json index 5f16b39..edb08a4 100644 --- a/tests/fixtures/surveys/all_types_survey/ddi2xlsform.json +++ b/tests/fixtures/surveys/all_types_survey/ddi2xlsform.json @@ -1,5 +1,40 @@ { "survey": [ + { + "type": "start", + "name": "start", + "label": "" + }, + { + "type": "end", + "name": "end", + "label": "" + }, + { + "type": "today", + "name": "today", + "label": "" + }, + { + "type": "deviceid", + "name": "deviceid", + "label": "" + }, + { + "type": "username", + "name": "username", + "label": "" + }, + { + "type": "hidden", + "name": "hidden1", + "label": "" + }, + { + "type": "audit", + "name": "audit", + "label": "" + }, { "type": "text", "name": "orphan_before", @@ -46,15 +81,20 @@ "label": "Um wie viel Uhr sind Sie angekommen?" }, { - "type": "select_one q_select_one", + "type": "select_one yesno", "name": "q_select_one", "label": "Würden Sie diesen Workshop weiterempfehlen?" }, { - "type": "select_multiple q_select_multi", + "type": "select_multiple topics", "name": "q_select_multi", "label": "An welchen Tracks haben Sie teilgenommen?" }, + { + "type": "note", + "name": "q_note", + "label": "Im nächsten Abschnitt bitten wir Sie um Ihr ausführliches Feedback. Bitte nehmen Sie sich Zeit." + }, { "type": "end_group" }, @@ -76,13 +116,13 @@ "appearance": "multiline" }, { - "type": "select_one q_likert", + "type": "select_one satisfaction", "name": "q_likert", "label": "Wie zufrieden waren Sie mit der Organisation insgesamt?", "appearance": "likert" }, { - "type": "select_one q_select_one", + "type": "select_one yesno", "name": "q_minimal", "label": "Bevorzugst du Dropdown?", "appearance": "minimal" @@ -102,19 +142,25 @@ "appearance": "field-list" }, { - "type": "select_one skill_python", + "type": "select_one proficiency", + "name": "matrix_header", + "label": "Wie schätzen Sie Ihre Kenntnisse ein?", + "appearance": "label" + }, + { + "type": "select_one proficiency", "name": "skill_python", "label": "Python", "appearance": "list-nolabel" }, { - "type": "select_one skill_python", + "type": "select_one proficiency", "name": "skill_js", "label": "JavaScript", "appearance": "list-nolabel" }, { - "type": "select_one skill_python", + "type": "select_one proficiency", "name": "skill_sql", "label": "SQL", "appearance": "list-nolabel" @@ -131,27 +177,15 @@ "label": "Weitere Fragen" }, { - "type": "select_one q_sel1_other", + "type": "select_one topics or_other", "name": "q_sel1_other", "label": "Welcher Track hat Ihnen am besten gefallen?" }, { - "type": "text", - "name": "q_sel1_other_other", - "label": "Sonstiges:", - "relevant": "${q_sel1_other} = 'other'" - }, - { - "type": "select_multiple q_selm_other", + "type": "select_multiple topics or_other", "name": "q_selm_other", "label": "Welche Tracks würden Sie erneut besuchen?" }, - { - "type": "text", - "name": "q_selm_other_other", - "label": "Sonstiges:", - "relevant": "selected(${q_selm_other}, 'other')" - }, { "type": "end_group" }, @@ -203,11 +237,6 @@ { "type": "end_group" }, - { - "type": "note", - "name": "q_note", - "label": "Im nächsten Abschnitt bitten wir Sie um Ihr ausführliches Feedback. Bitte nehmen Sie sich Zeit." - }, { "type": "note", "name": "orphan_after", @@ -216,109 +245,69 @@ ], "choices": [ { - "list_name": "q_select_one", + "list_name": "yesno", "name": "yes", "label": "Ja" }, { - "list_name": "q_select_one", + "list_name": "yesno", "name": "no", "label": "Nein" }, { - "list_name": "q_select_multi", + "list_name": "topics", "name": "red", "label": "Frontend-Entwicklung" }, { - "list_name": "q_select_multi", + "list_name": "topics", "name": "blue", "label": "Backend-Entwicklung" }, { - "list_name": "q_select_multi", + "list_name": "topics", "name": "green", "label": "Data Science" }, { - "list_name": "q_likert", + "list_name": "satisfaction", "name": "low", "label": "Schlecht" }, { - "list_name": "q_likert", + "list_name": "satisfaction", "name": "mid", "label": "Mittelmäßig" }, { - "list_name": "q_likert", + "list_name": "satisfaction", "name": "high", "label": "Ausgezeichnet" }, { - "list_name": "skill_python", + "list_name": "proficiency", "name": "none", "label": "Keine" }, { - "list_name": "skill_python", + "list_name": "proficiency", "name": "basic", "label": "Grundkenntnisse" }, { - "list_name": "skill_python", + "list_name": "proficiency", "name": "adv", "label": "Fortgeschritten" }, { - "list_name": "skill_python", + "list_name": "proficiency", "name": "exp", "label": "Experte" - }, - { - "list_name": "q_sel1_other", - "name": "red", - "label": "Frontend-Entwicklung" - }, - { - "list_name": "q_sel1_other", - "name": "blue", - "label": "Backend-Entwicklung" - }, - { - "list_name": "q_sel1_other", - "name": "green", - "label": "Data Science" - }, - { - "list_name": "q_sel1_other", - "name": "other", - "label": "Sonstiges:" - }, - { - "list_name": "q_selm_other", - "name": "red", - "label": "Frontend-Entwicklung" - }, - { - "list_name": "q_selm_other", - "name": "blue", - "label": "Backend-Entwicklung" - }, - { - "list_name": "q_selm_other", - "name": "green", - "label": "Data Science" - }, - { - "list_name": "q_selm_other", - "name": "other", - "label": "Sonstiges" } ], "settings": [ { - "default_language": "German (de)", + "default_language": "de", "form_title": "all_types_survey", "form_id": "all_types" } diff --git a/tests/fixtures/surveys/appearances_survey/ddi.xml b/tests/fixtures/surveys/appearances_survey/ddi.xml index 0a6a13b..66e436d 100644 --- a/tests/fixtures/surveys/appearances_survey/ddi.xml +++ b/tests/fixtures/surveys/appearances_survey/ddi.xml @@ -9,6 +9,11 @@ 2020-01-01 + select_one oft + Nutzung + label + in=matrix after= + de @@ -50,6 +55,7 @@ Wo wohnen Sie? minimal + land @@ -78,6 +84,7 @@ Wie zufrieden sind Sie? likert + zust @@ -94,6 +101,7 @@ Bus list-nolabel + oft @@ -110,6 +118,7 @@ Bahn list-nolabel + oft @@ -126,6 +135,7 @@ … der Preis? + oft @@ -142,6 +152,7 @@ … der Takt? + oft diff --git a/tests/fixtures/surveys/appearances_survey/ddi2xlsform.json b/tests/fixtures/surveys/appearances_survey/ddi2xlsform.json index 1b5db4b..2f976b4 100644 --- a/tests/fixtures/surveys/appearances_survey/ddi2xlsform.json +++ b/tests/fixtures/surveys/appearances_survey/ddi2xlsform.json @@ -7,7 +7,7 @@ "appearance": "field-list" }, { - "type": "select_one wohnort", + "type": "select_one land", "name": "wohnort", "label": "Wo wohnen Sie?", "appearance": "minimal" @@ -19,7 +19,7 @@ "appearance": "multiline" }, { - "type": "select_one zufrieden", + "type": "select_one zust", "name": "zufrieden", "label": "Wie zufrieden sind Sie?", "appearance": "likert" @@ -34,13 +34,19 @@ "appearance": "field-list" }, { - "type": "select_one bus", + "type": "select_one oft", + "name": "kopf", + "label": "Nutzung", + "appearance": "label" + }, + { + "type": "select_one oft", "name": "bus", "label": "Bus", "appearance": "list-nolabel" }, { - "type": "select_one bus", + "type": "select_one oft", "name": "bahn", "label": "Bahn", "appearance": "list-nolabel" @@ -55,12 +61,12 @@ "appearance": "table-list" }, { - "type": "select_one bus", + "type": "select_one oft", "name": "preis", "label": "… der Preis?" }, { - "type": "select_one bus", + "type": "select_one oft", "name": "takt", "label": "… der Takt?" }, @@ -80,44 +86,44 @@ ], "choices": [ { - "list_name": "wohnort", + "list_name": "land", "name": "stadt", "label": "Stadt" }, { - "list_name": "wohnort", + "list_name": "land", "name": "dorf", "label": "Land" }, { - "list_name": "zufrieden", + "list_name": "zust", "name": "1", "label": "Gar nicht" }, { - "list_name": "zufrieden", + "list_name": "zust", "name": "2", "label": "Teils" }, { - "list_name": "zufrieden", + "list_name": "zust", "name": "3", "label": "Sehr" }, { - "list_name": "bus", + "list_name": "oft", "name": "nie", "label": "Nie" }, { - "list_name": "bus", + "list_name": "oft", "name": "oft", "label": "Oft" } ], "settings": [ { - "default_language": "German (de)", + "default_language": "de", "form_title": "appearances_survey" } ] diff --git a/tests/fixtures/surveys/basic_survey/ddi.xml b/tests/fixtures/surveys/basic_survey/ddi.xml index e9b20bc..ac0ffa6 100644 --- a/tests/fixtures/surveys/basic_survey/ddi.xml +++ b/tests/fixtures/surveys/basic_survey/ddi.xml @@ -11,6 +11,9 @@ Thank you for your participation! + ${consent} = 'yes' + in= after=consent + en @@ -54,6 +57,7 @@ Do you consent to participate? yes + yes_no diff --git a/tests/fixtures/surveys/basic_survey/ddi2xlsform.json b/tests/fixtures/surveys/basic_survey/ddi2xlsform.json index c9dbf6f..d7d7400 100644 --- a/tests/fixtures/surveys/basic_survey/ddi2xlsform.json +++ b/tests/fixtures/surveys/basic_survey/ddi2xlsform.json @@ -13,7 +13,7 @@ "required": "yes" }, { - "type": "select_one consent", + "type": "select_one yes_no", "name": "consent", "label": "Do you consent to participate?", "required": "yes" @@ -21,23 +21,25 @@ { "type": "note", "name": "thank_you", - "label": "Thank you for your participation!" + "label": "Thank you for your participation!", + "relevant": "${consent} = 'yes'" } ], "choices": [ { - "list_name": "consent", + "list_name": "yes_no", "name": "yes", "label": "Yes" }, { - "list_name": "consent", + "list_name": "yes_no", "name": "no", "label": "No" } ], "settings": [ { + "default_language": "en", "form_title": "basic_survey", "form_id": "basic_survey" } diff --git a/tests/fixtures/surveys/bilingual_survey/ddi.xml b/tests/fixtures/surveys/bilingual_survey/ddi.xml index 62663a5..8277305 100644 --- a/tests/fixtures/surveys/bilingual_survey/ddi.xml +++ b/tests/fixtures/surveys/bilingual_survey/ddi.xml @@ -12,6 +12,10 @@ Willkommen zur Befragung. Welcome to the survey. + in= after= + Deutsch (de) + de + en @@ -43,6 +47,7 @@ Wie haben Sie von uns erfahren? How did you hear about us? Wie haben Sie von uns erfahren? + shorthand @@ -56,6 +61,7 @@ Was ist Ihr Beruf? + person_note @@ -172,6 +178,7 @@ … dem Parlament? + skala @@ -197,6 +204,7 @@ … der Polizei? + skala diff --git a/tests/fixtures/surveys/bilingual_survey/ddi2xlsform.json b/tests/fixtures/surveys/bilingual_survey/ddi2xlsform.json index c3f58d7..ef14fd6 100644 --- a/tests/fixtures/surveys/bilingual_survey/ddi2xlsform.json +++ b/tests/fixtures/surveys/bilingual_survey/ddi2xlsform.json @@ -1,5 +1,13 @@ { "survey": [ + { + "type": "note", + "name": "intro", + "label": { + "de": "Willkommen zur Befragung.", + "en": "Welcome to the survey." + } + }, { "type": "begin_group", "name": "person", @@ -10,7 +18,7 @@ }, { "type": "note", - "name": "beruf_note", + "name": "person_note", "label": { "de": "Alle Angaben sind freiwillig.", "en": "All answers are optional." @@ -48,22 +56,13 @@ } }, { - "type": "select_one quelle", + "type": "select_one quelle or_other", "name": "quelle", "label": { "de": "Wie haben Sie von uns erfahren?", "en": "How did you hear about us?" } }, - { - "type": "text", - "name": "quelle_other", - "label": { - "de": "Sonstiges:", - "en": "Other:" - }, - "relevant": "${quelle} = 'other'" - }, { "type": "end_group" }, @@ -77,7 +76,7 @@ "appearance": "table-list" }, { - "type": "select_one vertrauen", + "type": "select_one skala", "name": "parlament", "label": { "de": "… dem Parlament?", @@ -89,7 +88,7 @@ } }, { - "type": "select_one vertrauen", + "type": "select_one skala", "name": "polizei", "label": { "de": "… der Polizei?", @@ -98,14 +97,6 @@ }, { "type": "end_group" - }, - { - "type": "note", - "name": "intro", - "label": { - "de": "Willkommen zur Befragung.", - "en": "Welcome to the survey." - } } ], "choices": [ @@ -165,15 +156,7 @@ } }, { - "list_name": "quelle", - "name": "other", - "label": { - "de": "Sonstiges:", - "en": "Other:" - } - }, - { - "list_name": "vertrauen", + "list_name": "skala", "name": "1", "label": { "de": "gar nicht", @@ -181,7 +164,7 @@ } }, { - "list_name": "vertrauen", + "list_name": "skala", "name": "2", "label": { "de": "etwas", @@ -189,7 +172,7 @@ } }, { - "list_name": "vertrauen", + "list_name": "skala", "name": "3", "label": { "de": "sehr", @@ -199,7 +182,7 @@ ], "settings": [ { - "default_language": "German (de)", + "default_language": "Deutsch (de)", "form_title": "bilingual_survey", "form_id": "bilingual_survey" } diff --git a/tests/fixtures/surveys/complex_survey/ddi.xml b/tests/fixtures/surveys/complex_survey/ddi.xml index ed4e7fc..cfa7ff0 100644 --- a/tests/fixtures/surveys/complex_survey/ddi.xml +++ b/tests/fixtures/surveys/complex_survey/ddi.xml @@ -10,6 +10,7 @@ 2020-01-01 + en @@ -36,6 +37,7 @@ What are your favorite colors? What are your favorite colors? Only if “Age” ≥ 18 + colors @@ -155,6 +157,7 @@ How satisfied are you? + satisfaction diff --git a/tests/fixtures/surveys/complex_survey/ddi2xlsform.json b/tests/fixtures/surveys/complex_survey/ddi2xlsform.json index 61d0f4b..2ac8cc5 100644 --- a/tests/fixtures/surveys/complex_survey/ddi2xlsform.json +++ b/tests/fixtures/surveys/complex_survey/ddi2xlsform.json @@ -33,13 +33,13 @@ "relevant": "${age} >= 18" }, { - "type": "select_multiple favorite_colors", + "type": "select_multiple colors", "name": "favorite_colors", "label": "What are your favorite colors?", "hint": "Select all that apply" }, { - "type": "select_one satisfaction_level", + "type": "select_one satisfaction", "name": "satisfaction_level", "label": "How satisfied are you?" }, @@ -76,38 +76,39 @@ "label": "Other" }, { - "list_name": "favorite_colors", + "list_name": "colors", "name": "red", "label": "Red" }, { - "list_name": "favorite_colors", + "list_name": "colors", "name": "blue", "label": "Blue" }, { - "list_name": "favorite_colors", + "list_name": "colors", "name": "green", "label": "Green" }, { - "list_name": "favorite_colors", + "list_name": "colors", "name": "yellow", "label": "Yellow" }, { - "list_name": "satisfaction_level", + "list_name": "satisfaction", "name": "happ", "label": "Happy" }, { - "list_name": "satisfaction_level", + "list_name": "satisfaction", "name": "unhp", "label": "Unhappy" } ], "settings": [ { + "default_language": "en", "form_title": "complex_survey", "form_id": "complex_survey" } diff --git a/tests/fixtures/surveys/complex_xpath_survey/ddi.xml b/tests/fixtures/surveys/complex_xpath_survey/ddi.xml index 044b916..7102832 100644 --- a/tests/fixtures/surveys/complex_xpath_survey/ddi.xml +++ b/tests/fixtures/surveys/complex_xpath_survey/ddi.xml @@ -10,6 +10,7 @@ 2020-01-01 + en diff --git a/tests/fixtures/surveys/complex_xpath_survey/ddi2xlsform.json b/tests/fixtures/surveys/complex_xpath_survey/ddi2xlsform.json index 6a56a31..437ecef 100644 --- a/tests/fixtures/surveys/complex_xpath_survey/ddi2xlsform.json +++ b/tests/fixtures/surveys/complex_xpath_survey/ddi2xlsform.json @@ -58,6 +58,7 @@ "choices": [], "settings": [ { + "default_language": "en", "form_title": "complex_xpath_survey", "form_id": "complex_xpath_survey" } diff --git a/tests/fixtures/surveys/grid_survey/ddi.xml b/tests/fixtures/surveys/grid_survey/ddi.xml index f878e61..07d8514 100644 --- a/tests/fixtures/surveys/grid_survey/ddi.xml +++ b/tests/fixtures/surveys/grid_survey/ddi.xml @@ -10,6 +10,8 @@ Einige Fragen zu Ihrem Vertrauen. + in= after= + de @@ -30,6 +32,7 @@ Welche Medien nutzen Sie? Welche Medien nutzen Sie? Nun zu Ihrer Mediennutzung. + medienhinweis @@ -52,6 +55,7 @@ … dem Parlament? + skala @@ -72,6 +76,7 @@ … der Polizei? + skala @@ -111,6 +116,7 @@ Wie alt sind Sie? + alterhinweis diff --git a/tests/fixtures/surveys/grid_survey/ddi2xlsform.json b/tests/fixtures/surveys/grid_survey/ddi2xlsform.json index b37bcf5..c375243 100644 --- a/tests/fixtures/surveys/grid_survey/ddi2xlsform.json +++ b/tests/fixtures/surveys/grid_survey/ddi2xlsform.json @@ -1,5 +1,10 @@ { "survey": [ + { + "type": "note", + "name": "intro", + "label": "Einige Fragen zu Ihrem Vertrauen." + }, { "type": "begin_group", "name": "vertrauen", @@ -7,14 +12,14 @@ "appearance": "table-list" }, { - "type": "select_one vertrauen", + "type": "select_one skala", "name": "parlament", "label": "… dem Parlament?", "hint": "Bundestag", "guidance_hint": "Skala vorlesen" }, { - "type": "select_one vertrauen", + "type": "select_one skala", "name": "polizei", "label": "… der Polizei?" }, @@ -23,7 +28,7 @@ }, { "type": "note", - "name": "medien_note", + "name": "medienhinweis", "label": "Nun zu Ihrer Mediennutzung." }, { @@ -34,7 +39,7 @@ }, { "type": "note", - "name": "alter_note", + "name": "alterhinweis", "label": "Zuletzt eine Angabe zur Person." }, { @@ -42,26 +47,21 @@ "name": "alter", "label": "Wie alt sind Sie?", "hint": "In Jahren" - }, - { - "type": "note", - "name": "intro", - "label": "Einige Fragen zu Ihrem Vertrauen." } ], "choices": [ { - "list_name": "vertrauen", + "list_name": "skala", "name": "1", "label": "Gar nicht" }, { - "list_name": "vertrauen", + "list_name": "skala", "name": "2", "label": "Etwas" }, { - "list_name": "vertrauen", + "list_name": "skala", "name": "3", "label": "Voll" }, @@ -78,7 +78,7 @@ ], "settings": [ { - "default_language": "German (de)", + "default_language": "de", "form_title": "grid_survey" } ] diff --git a/tests/fixtures/surveys/hints_survey/ddi.xml b/tests/fixtures/surveys/hints_survey/ddi.xml index d6cff78..cfa03a6 100644 --- a/tests/fixtures/surveys/hints_survey/ddi.xml +++ b/tests/fixtures/surveys/hints_survey/ddi.xml @@ -10,6 +10,7 @@ 2020-01-01 + de @@ -34,6 +35,7 @@ Wie viele Personen leben in Ihrem Haushalt? + intro @@ -59,6 +61,7 @@ Sind Sie ehrenamtlich tätig? + janein diff --git a/tests/fixtures/surveys/hints_survey/ddi2xlsform.json b/tests/fixtures/surveys/hints_survey/ddi2xlsform.json index 7c05fb1..099fec9 100644 --- a/tests/fixtures/surveys/hints_survey/ddi2xlsform.json +++ b/tests/fixtures/surveys/hints_survey/ddi2xlsform.json @@ -2,7 +2,7 @@ "survey": [ { "type": "note", - "name": "personen_note", + "name": "intro", "label": "Die folgenden Fragen betreffen Ihren Haushalt." }, { @@ -19,7 +19,7 @@ "guidance_hint": "Bei Unklarheit nach der Tätigkeit fragen." }, { - "type": "select_one ehrenamt", + "type": "select_one janein", "name": "ehrenamt", "label": "Sind Sie ehrenamtlich tätig?", "guidance_hint": "Auch gelegentliches Engagement zählt." @@ -33,12 +33,12 @@ ], "choices": [ { - "list_name": "ehrenamt", + "list_name": "janein", "name": "ja", "label": "Ja" }, { - "list_name": "ehrenamt", + "list_name": "janein", "name": "nein", "label": "Nein" }, @@ -55,7 +55,7 @@ ], "settings": [ { - "default_language": "German (de)", + "default_language": "de", "form_title": "hints_survey", "form_id": "hints" } diff --git a/tests/fixtures/surveys/multilingual_survey/ddi.xml b/tests/fixtures/surveys/multilingual_survey/ddi.xml index bd1f25b..83e746c 100644 --- a/tests/fixtures/surveys/multilingual_survey/ddi.xml +++ b/tests/fixtures/surveys/multilingual_survey/ddi.xml @@ -10,6 +10,9 @@ 2020-01-01 + en + en + es diff --git a/tests/fixtures/surveys/multilingual_survey/ddi2xlsform.json b/tests/fixtures/surveys/multilingual_survey/ddi2xlsform.json index 99fe2b9..8b7232a 100644 --- a/tests/fixtures/surveys/multilingual_survey/ddi2xlsform.json +++ b/tests/fixtures/surveys/multilingual_survey/ddi2xlsform.json @@ -100,6 +100,7 @@ ], "settings": [ { + "default_language": "en", "form_title": "multilingual_survey", "form_id": "multilingual_survey" } diff --git a/tests/fixtures/surveys/multipage_survey/ddi.xml b/tests/fixtures/surveys/multipage_survey/ddi.xml index 4cd46da..0851b42 100644 --- a/tests/fixtures/surveys/multipage_survey/ddi.xml +++ b/tests/fixtures/surveys/multipage_survey/ddi.xml @@ -11,6 +11,8 @@ Thank you! + in=page2 after=favcolor + en pages @@ -66,6 +68,7 @@ Favourite colour + colors diff --git a/tests/fixtures/surveys/multipage_survey/ddi2xlsform.json b/tests/fixtures/surveys/multipage_survey/ddi2xlsform.json index 9754a1c..4587973 100644 --- a/tests/fixtures/surveys/multipage_survey/ddi2xlsform.json +++ b/tests/fixtures/surveys/multipage_survey/ddi2xlsform.json @@ -26,38 +26,39 @@ "appearance": "field-list" }, { - "type": "select_one favcolor", + "type": "select_one colors", "name": "favcolor", "label": "Favourite colour" }, - { - "type": "end_group" - }, { "type": "note", "name": "thankyou", "label": "Thank you!" + }, + { + "type": "end_group" } ], "choices": [ { - "list_name": "favcolor", + "list_name": "colors", "name": "red", "label": "Red" }, { - "list_name": "favcolor", + "list_name": "colors", "name": "blue", "label": "Blue" }, { - "list_name": "favcolor", + "list_name": "colors", "name": "green", "label": "Green" } ], "settings": [ { + "default_language": "en", "form_title": "multipage_survey", "form_id": "multipage", "style": "pages" diff --git a/tests/fixtures/surveys/settings_survey/ddi.xml b/tests/fixtures/surveys/settings_survey/ddi.xml index af26309..3e19f91 100644 --- a/tests/fixtures/surveys/settings_survey/ddi.xml +++ b/tests/fixtures/surveys/settings_survey/ddi.xml @@ -12,6 +12,9 @@ **Welcome** to the survey! **Thank you** for participating! + in= after= + in= after=main + en @@ -50,6 +53,7 @@ What is your _favorite_ color? + colors diff --git a/tests/fixtures/surveys/settings_survey/ddi2xlsform.json b/tests/fixtures/surveys/settings_survey/ddi2xlsform.json index 515b262..ea4f1c3 100644 --- a/tests/fixtures/surveys/settings_survey/ddi2xlsform.json +++ b/tests/fixtures/surveys/settings_survey/ddi2xlsform.json @@ -1,12 +1,17 @@ { "survey": [ + { + "type": "note", + "name": "welcome", + "label": "**Welcome** to the survey!" + }, { "type": "begin_group", "name": "main", "label": "Main Questions" }, { - "type": "select_one fav_color", + "type": "select_one colors", "name": "fav_color", "label": "What is your _favorite_ color?" }, @@ -24,11 +29,6 @@ { "type": "end_group" }, - { - "type": "note", - "name": "welcome", - "label": "**Welcome** to the survey!" - }, { "type": "note", "name": "end", @@ -37,23 +37,24 @@ ], "choices": [ { - "list_name": "fav_color", + "list_name": "colors", "name": "red", "label": "Red" }, { - "list_name": "fav_color", + "list_name": "colors", "name": "blue", "label": "Blue" }, { - "list_name": "fav_color", + "list_name": "colors", "name": "other", "label": "Other" } ], "settings": [ { + "default_language": "en", "form_title": "settings_survey", "form_id": "settings_survey" } diff --git a/tests/fixtures/surveys/testA/ddi.xml b/tests/fixtures/surveys/testA/ddi.xml index 7638d96..5356a99 100644 --- a/tests/fixtures/surveys/testA/ddi.xml +++ b/tests/fixtures/surveys/testA/ddi.xml @@ -15,6 +15,11 @@ Hey! Wir möchten gerne wissen, wie TestWerk für dich ist und was es dir bringt. Deine Antworten sind anonym und helfen uns, unser Angebot zu verbessern. Es gibt keine richtigen oder falschen Antworten - sag einfach deine Meinung! Vielen Dank! Deine Meinung hilft uns sehr weiter! + in= after= + in=intro after= + in= after=demografie + Deutsch + testwerk_wirkung_2025 @@ -48,6 +53,10 @@ demografie demografie + + intro + intro + Welche Bereiche nutzt du überhaupt? (Mehrfachauswahl möglich) Welche Bereiche nutzt du überhaupt? (Mehrfachauswahl möglich) @@ -244,6 +253,7 @@ yes likert + likert5 @@ -273,6 +283,7 @@ yes likert + likert5 @@ -302,6 +313,7 @@ yes likert + likert5 @@ -331,6 +343,7 @@ yes likert + likert5 @@ -360,6 +373,7 @@ yes likert + likert5 @@ -390,6 +404,8 @@ yes likert + likert5 + beruf_note @@ -419,6 +435,7 @@ yes likert + likert5 @@ -433,7 +450,7 @@ Ca. Wie viel Prozent dieser Veränderung geht auf TestWerk zurück? ${beruf_post} != ${beruf_pre} - step=5 + start=0 end=100 step=5 @@ -509,6 +526,7 @@ yes likert + nps diff --git a/tests/fixtures/surveys/testA/ddi2xlsform.json b/tests/fixtures/surveys/testA/ddi2xlsform.json index 4226fdf..76f8c2f 100644 --- a/tests/fixtures/surveys/testA/ddi2xlsform.json +++ b/tests/fixtures/surveys/testA/ddi2xlsform.json @@ -1,5 +1,18 @@ { "survey": [ + { + "type": "begin_group", + "name": "intro", + "label": "intro" + }, + { + "type": "note", + "name": "intro_text", + "label": "Hey! Wir möchten gerne wissen, wie TestWerk für dich ist und was es dir bringt. Deine Antworten sind anonym und helfen uns, unser Angebot zu verbessern. Es gibt keine richtigen oder falschen Antworten - sag einfach deine Meinung!" + }, + { + "type": "end_group" + }, { "type": "begin_group", "name": "teilnahme", @@ -50,35 +63,35 @@ "hint": "Bitte gib an, wie sehr du zustimmst (1 = stimme gar nicht zu, 5 = stimme voll zu):" }, { - "type": "select_one wohlfuehlen", + "type": "select_one likert5", "name": "wohlfuehlen", "label": "Ich fühle mich bei TestWerk wohl und akzeptiert.", "required": "yes", "appearance": "likert" }, { - "type": "select_one wohlfuehlen", + "type": "select_one likert5", "name": "freundschaften", "label": "Ich habe bei TestWerk Freund*innen gefunden.", "required": "yes", "appearance": "likert" }, { - "type": "select_one wohlfuehlen", + "type": "select_one likert5", "name": "eigene_ideen", "label": "Ich traue mir zu, eigene Ideen und Projekte umzusetzen.", "required": "yes", "appearance": "likert" }, { - "type": "select_one wohlfuehlen", + "type": "select_one likert5", "name": "problemloesung", "label": "Wenn bei einem Projekt etwas nicht klappt, finde ich eine Lösung.", "required": "yes", "appearance": "likert" }, { - "type": "select_one wohlfuehlen", + "type": "select_one likert5", "name": "skills_gelernt", "label": "Ich habe bei TestWerk handwerkliche Fähigkeiten gelernt.", "required": "yes", @@ -86,18 +99,18 @@ }, { "type": "note", - "name": "beruf_pre_note", + "name": "beruf_note", "label": "Jetzt denk mal an deine berufliche Zukunft:" }, { - "type": "select_one wohlfuehlen", + "type": "select_one likert5", "name": "beruf_pre", "label": "Bevor ich zu TestWerk kam, hatte ich eine klare Vorstellung von meiner beruflichen Zukunft.", "required": "yes", "appearance": "likert" }, { - "type": "select_one wohlfuehlen", + "type": "select_one likert5", "name": "beruf_post", "label": "Aktuell habe ich eine klare Vorstellung von meiner beruflichen Zukunft.", "required": "yes", @@ -143,7 +156,7 @@ "label": "empfehlung" }, { - "type": "select_one nps_score", + "type": "select_one nps", "name": "nps_score", "label": "Würdest du TestWerk Freund*innen empfehlen?", "hint": "0 = auf keinen Fall, 10 = auf jeden Fall", @@ -177,11 +190,6 @@ { "type": "end_group" }, - { - "type": "note", - "name": "intro_text", - "label": "Hey! Wir möchten gerne wissen, wie TestWerk für dich ist und was es dir bringt. Deine Antworten sind anonym und helfen uns, unser Angebot zu verbessern. Es gibt keine richtigen oder falschen Antworten - sag einfach deine Meinung!" - }, { "type": "note", "name": "danke", @@ -295,82 +303,82 @@ "label": "Ja, viel" }, { - "list_name": "wohlfuehlen", + "list_name": "likert5", "name": "1", "label": "Stimme gar nicht zu" }, { - "list_name": "wohlfuehlen", + "list_name": "likert5", "name": "2", "label": "Stimme eher nicht zu" }, { - "list_name": "wohlfuehlen", + "list_name": "likert5", "name": "3", "label": "Weder noch" }, { - "list_name": "wohlfuehlen", + "list_name": "likert5", "name": "4", "label": "Stimme eher zu" }, { - "list_name": "wohlfuehlen", + "list_name": "likert5", "name": "5", "label": "Stimme voll zu" }, { - "list_name": "nps_score", + "list_name": "nps", "name": "0", "label": "0" }, { - "list_name": "nps_score", + "list_name": "nps", "name": "1", "label": "1" }, { - "list_name": "nps_score", + "list_name": "nps", "name": "2", "label": "2" }, { - "list_name": "nps_score", + "list_name": "nps", "name": "3", "label": "3" }, { - "list_name": "nps_score", + "list_name": "nps", "name": "4", "label": "4" }, { - "list_name": "nps_score", + "list_name": "nps", "name": "5", "label": "5" }, { - "list_name": "nps_score", + "list_name": "nps", "name": "6", "label": "6" }, { - "list_name": "nps_score", + "list_name": "nps", "name": "7", "label": "7" }, { - "list_name": "nps_score", + "list_name": "nps", "name": "8", "label": "8" }, { - "list_name": "nps_score", + "list_name": "nps", "name": "9", "label": "9" }, { - "list_name": "nps_score", + "list_name": "nps", "name": "10", "label": "10" }, @@ -422,9 +430,10 @@ ], "settings": [ { + "default_language": "Deutsch", "form_title": "testA", - "form_id": "testwerk_wirkung_2025", - "version": "2.2" + "version": "2.2", + "id_string": "testwerk_wirkung_2025" } ] } diff --git a/tests/fixtures/surveys/testB/ddi.xml b/tests/fixtures/surveys/testB/ddi.xml index 25ce766..6693c01 100644 --- a/tests/fixtures/surveys/testB/ddi.xml +++ b/tests/fixtures/surveys/testB/ddi.xml @@ -16,7 +16,50 @@ Hello! Disclaimer Disclaimer + start + end + select_one iy6st81 + Bitte bewerte Deine Erfahrung mit den folgenden Technologien und Tools: + Please rate your experience with the following technologies and tools: + Hier kannst du mehr darüber lesen, wie wir die verschiedenen Erfahrungsstufen definieren: https://docs.correlaid.org/project-manual/data4good-projects#experience-scale + Read more about what the different experience labels mean for us: https://docs.correlaid.org/project-manual/data4good-projects#experience-scale + label + select_one xe9mo32 + Bitte bewerte Deine Erfahrung mit den folgenden Techniken. + Please rate your experience with the following techniques: + Hier kannst du mehr darüber lesen, wie wir die verschiedenen Erfahrungsstufen definieren: https://docs.correlaid.org/project-manual/data4good-projects#experience-scale + Read more about what the different experience labels mean for us: https://docs.correlaid.org/project-manual/data4good-projects#experience-scale + label + select_one dm1uu76 + Bitte bewerte Deine Erfahrung mit den folgenden Themen: + Please rate your experience with the following topics: + Hier kannst du mehr darüber lesen, wie wir die verschiedenen Erfahrungsstufen definieren: https://docs.correlaid.org/project-manual/data4good-projects#experience-scale + Read more about what the different experience labels mean for us: https://docs.correlaid.org/project-manual/data4good-projects#experience-scale + label + Hier kannst Du Dich auf das Projekt auf das Projekt Alpha bewerben. Bitte beachte, dass du dir pro Woche ca. 4h Zeit für das Projekt nehmen können solltest. + +Für die Umfrage benötigst Du 5 bis 10 Minuten. Mach Dir also keine Sorgen, falls Du auch mal Anfänger*in ankreuzt, du musst nicht alles schon perfekt können, um zum Projekt beitragen zu können. Das trifft vor allem auf das Thema Wirkungsmessung zu. Unsere Projektkoordinator*innen kontaktieren Dich dann schnellstmöglich. + Here you can apply for projects. Please note that you should be able to devote approximately 4 hours per week to the project. + +The survey will take 5 to 10 minutes to complete. So don't worry if you check the “beginner” box—you don't have to be an expert to contribute to the project. This applies in particular to the topic of impact measurement. Our project coordinators will then contact you as soon as possible. + In der folgenden Umfrage fragen wir Daten zu Dir und Deiner bisherigen Berufserfahrung ab. Dabei erheben wir Deinen Namen und Deine Kontaktdaten, damit unsere Projektmanager*innen Dich kontaktieren können. Nach erfolgreicher Ausschreibung werden diese Daten auch mit dem Projektteam innerhalb unserer Organisation geteilt. + +Bei einer Umfrage hast Du gemäß der DSGVO das Recht auf Auskunft sowie Löschung Deiner personenbezogenen Daten. Du kannst diese Einwilligungserklärung jederzeit widerrufen. Schreib hierzu einfach eine E-Mail an info@correlaid.org. Nach erfolgtem Widerruf werden Deine Daten gelöscht. + In the following survey, we ask for information about you and your previous professional experience. We collect your name and contact details so that our project managers can contact you. Once you have been successfully selected, this data will also be shared with the project team within our organization. + +In accordance with the GDPR, you have the right to access and delete your personal data in a survey. You can revoke this declaration of consent at any time. Simply send an email to info@correlaid.org. Your data will be deleted after revocation. + in= after= + in= after=start + in= after=end + in=group_gi4rv46 after= + in=group_gi4rv46 after=Hallo + in=group_lt58n55/rating_technologies_tools after= + in=group_lt58n55/rating_techniques after= + in=group_lt58n55/rating_topics after= + Deutsch (de) theme-grid no-text-transform + de + en @@ -61,11 +104,16 @@ About you Über dich + + Hallo! + Hallo! + Für welche Projekte möchtest Du dich bewerben? Which project(s) do you want to apply for? Für welche Projekte möchtest Du dich bewerben? yes + mb5co98 In welcher Rolle siehst du dich im Projekt Projekt Alpha? @@ -75,6 +123,7 @@ Only if “Which project(s) do you want to apply for?” includes Project Alpha selected(${project_id}, 'project-alpha') yes + cl-role-project-alpha In welcher Rolle siehst du dich im Projekt Beta? @@ -84,6 +133,7 @@ Only if “Which project(s) do you want to apply for?” includes Project Beta selected(${project_id}, 'project-beta') yes + cl-role-project-beta In welcher Rolle siehst du dich im Projekt Gamma? @@ -93,6 +143,7 @@ Only if “Which project(s) do you want to apply for?” includes Project Gamma selected(${project_id}, 'project-gamma') yes + cl-role-project-gamma @@ -403,6 +454,7 @@ selected(${project_role_project-alpha}, 'role-survey-design') yes list-nolabel + iy6st81 @@ -437,6 +489,7 @@ selected(${project_role_project-alpha}, 'role-data-analysis') yes list-nolabel + iy6st81 @@ -471,6 +524,7 @@ selected(${project_role_project-alpha}, 'role-data-analysis') yes list-nolabel + iy6st81 @@ -505,6 +559,7 @@ selected(${project_role_project-beta}, 'role-visualization') yes list-nolabel + iy6st81 @@ -539,6 +594,7 @@ selected(${project_role_project-beta}, 'role-visualization') yes list-nolabel + iy6st81 @@ -573,6 +629,7 @@ selected(${project_role_project-beta}, 'role-data-engineering') yes list-nolabel + iy6st81 @@ -607,6 +664,7 @@ selected(${project_role_project-beta}, 'role-data-engineering') yes list-nolabel + iy6st81 @@ -641,6 +699,7 @@ selected(${project_role_project-gamma}, 'role-machine-learning') yes list-nolabel + iy6st81 @@ -675,6 +734,7 @@ selected(${project_role_project-gamma}, 'role-machine-learning') yes list-nolabel + iy6st81 @@ -720,6 +780,7 @@ selected(${project_role_project-alpha}, 'role-survey-design') yes list-nolabel + xe9mo32 @@ -754,6 +815,7 @@ selected(${project_role_project-alpha}, 'role-survey-design') yes list-nolabel + xe9mo32 @@ -788,6 +850,7 @@ selected(${project_role_project-alpha}, 'role-survey-design') yes list-nolabel + xe9mo32 @@ -822,6 +885,7 @@ selected(${project_role_project-alpha}, 'role-data-analysis') yes list-nolabel + xe9mo32 @@ -856,6 +920,7 @@ selected(${project_role_project-alpha}, 'role-data-analysis') yes list-nolabel + xe9mo32 @@ -890,6 +955,7 @@ selected(${project_role_project-beta}, 'role-visualization') yes list-nolabel + xe9mo32 @@ -924,6 +990,7 @@ selected(${project_role_project-beta}, 'role-data-engineering') yes list-nolabel + xe9mo32 @@ -958,6 +1025,7 @@ selected(${project_role_project-beta}, 'role-data-engineering') yes list-nolabel + xe9mo32 @@ -992,6 +1060,7 @@ selected(${project_role_project-gamma}, 'role-project-management') yes list-nolabel + xe9mo32 @@ -1026,6 +1095,7 @@ selected(${project_role_project-gamma}, 'role-machine-learning') yes list-nolabel + xe9mo32 @@ -1060,6 +1130,7 @@ selected(${project_role_project-alpha}, 'role-survey-design') yes list-nolabel + dm1uu76 @@ -1094,6 +1165,7 @@ selected(${project_role_project-alpha}, 'role-survey-design') yes list-nolabel + dm1uu76 @@ -1128,6 +1200,7 @@ selected(${project_role_project-alpha}, 'role-survey-design') yes list-nolabel + dm1uu76 @@ -1162,6 +1235,7 @@ selected(${project_role_project-alpha}, 'role-survey-design') yes list-nolabel + dm1uu76 @@ -1196,6 +1270,7 @@ selected(${project_role_project-alpha}, 'role-survey-design') yes list-nolabel + dm1uu76 @@ -1230,6 +1305,7 @@ selected(${project_role_project-alpha}, 'role-survey-design') yes list-nolabel + dm1uu76 @@ -1280,6 +1356,7 @@ Hast du dich in der Vergangenheit bereits auf CorrelAid-Projekte beworben? yes + il2wc73 @@ -1359,6 +1436,7 @@ Was ist dein Geschlecht? yes + iy7os66 @@ -1390,6 +1468,7 @@ Einwilligung in die Datenschutzerklärung yes + mj8ty33 diff --git a/tests/fixtures/surveys/testB/ddi2xlsform.json b/tests/fixtures/surveys/testB/ddi2xlsform.json index 549c410..6336021 100644 --- a/tests/fixtures/surveys/testB/ddi2xlsform.json +++ b/tests/fixtures/surveys/testB/ddi2xlsform.json @@ -1,7 +1,51 @@ { "survey": [ { - "type": "select_multiple project_id", + "type": "start", + "name": "start", + "label": "" + }, + { + "type": "end", + "name": "end", + "label": "" + }, + { + "type": "begin_group", + "name": "group_gi4rv46", + "label": { + "de": "Hallo!" + } + }, + { + "type": "note", + "name": "Hallo", + "label": { + "de": "Hallo!", + "en": "Hello!" + }, + "hint": { + "de": "Hier kannst Du Dich auf das Projekt auf das Projekt Alpha bewerben. Bitte beachte, dass du dir pro Woche ca. 4h Zeit für das Projekt nehmen können solltest. \n\nFür die Umfrage benötigst Du 5 bis 10 Minuten. Mach Dir also keine Sorgen, falls Du auch mal Anfänger*in ankreuzt, du musst nicht alles schon perfekt können, um zum Projekt beitragen zu können. Das trifft vor allem auf das Thema Wirkungsmessung zu. Unsere Projektkoordinator*innen kontaktieren Dich dann schnellstmöglich.", + "en": "Here you can apply for projects. Please note that you should be able to devote approximately 4 hours per week to the project. \n\nThe survey will take 5 to 10 minutes to complete. So don't worry if you check the “beginner” box—you don't have to be an expert to contribute to the project. This applies in particular to the topic of impact measurement. Our project coordinators will then contact you as soon as possible." + } + }, + { + "type": "note", + "name": "Disclaimer", + "label": { + "de": "Disclaimer", + "en": "Disclaimer" + }, + "hint": { + "de": "In der folgenden Umfrage fragen wir Daten zu Dir und Deiner bisherigen Berufserfahrung ab. Dabei erheben wir Deinen Namen und Deine Kontaktdaten, damit unsere Projektmanager*innen Dich kontaktieren können. Nach erfolgreicher Ausschreibung werden diese Daten auch mit dem Projektteam innerhalb unserer Organisation geteilt.\n\nBei einer Umfrage hast Du gemäß der DSGVO das Recht auf Auskunft sowie Löschung Deiner personenbezogenen Daten. Du kannst diese Einwilligungserklärung jederzeit widerrufen. Schreib hierzu einfach eine E-Mail an info@correlaid.org. Nach erfolgtem Widerruf werden Deine Daten gelöscht.", + "en": "In the following survey, we ask for information about you and your previous professional experience. We collect your name and contact details so that our project managers can contact you. Once you have been successfully selected, this data will also be shared with the project team within our organization.\n\nIn accordance with the GDPR, you have the right to access and delete your personal data in a survey. You can revoke this declaration of consent at any time. Simply send an email to info@correlaid.org. Your data will be deleted after revocation." + } + }, + { + "type": "end_group" + }, + { + "type": "select_multiple mb5co98", "name": "project_id", "label": { "de": "Für welche Projekte möchtest Du dich bewerben?", @@ -10,7 +54,7 @@ "required": "yes" }, { - "type": "select_multiple project_role_project-alpha", + "type": "select_multiple cl-role-project-alpha", "name": "project_role_project-alpha", "label": { "de": "In welcher Rolle siehst du dich im Projekt Projekt Alpha?", @@ -20,7 +64,7 @@ "relevant": "selected(${project_id}, 'project-alpha')" }, { - "type": "select_multiple project_role_project-beta", + "type": "select_multiple cl-role-project-beta", "name": "project_role_project-beta", "label": { "de": "In welcher Rolle siehst du dich im Projekt Beta?", @@ -30,7 +74,7 @@ "relevant": "selected(${project_id}, 'project-beta')" }, { - "type": "select_multiple project_role_project-gamma", + "type": "select_multiple cl-role-project-gamma", "name": "project_role_project-gamma", "label": { "de": "In welcher Rolle siehst du dich im Projekt Gamma?", @@ -56,7 +100,20 @@ } }, { - "type": "select_one sosci_survey", + "type": "select_one iy6st81", + "name": "rating_technologies_tools_header", + "label": { + "de": "Bitte bewerte Deine Erfahrung mit den folgenden Technologien und Tools:", + "en": "Please rate your experience with the following technologies and tools:" + }, + "hint": { + "de": "Hier kannst du mehr darüber lesen, wie wir die verschiedenen Erfahrungsstufen definieren: https://docs.correlaid.org/project-manual/data4good-projects#experience-scale", + "en": "Read more about what the different experience labels mean for us: https://docs.correlaid.org/project-manual/data4good-projects#experience-scale" + }, + "appearance": "label" + }, + { + "type": "select_one iy6st81", "name": "sosci_survey", "label": { "de": "SoSci Survey", @@ -67,7 +124,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-survey-design')" }, { - "type": "select_one sosci_survey", + "type": "select_one iy6st81", "name": "python", "label": { "de": "Python", @@ -78,7 +135,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-data-analysis')" }, { - "type": "select_one sosci_survey", + "type": "select_one iy6st81", "name": "rstats", "label": { "de": "R", @@ -89,7 +146,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-data-analysis')" }, { - "type": "select_one sosci_survey", + "type": "select_one iy6st81", "name": "powerbi", "label": { "de": "Power BI", @@ -100,7 +157,7 @@ "relevant": "selected(${project_role_project-beta}, 'role-visualization')" }, { - "type": "select_one sosci_survey", + "type": "select_one iy6st81", "name": "excel", "label": { "de": "Microsoft Excel", @@ -111,7 +168,7 @@ "relevant": "selected(${project_role_project-beta}, 'role-visualization')" }, { - "type": "select_one sosci_survey", + "type": "select_one iy6st81", "name": "sql", "label": { "de": "SQL", @@ -122,7 +179,7 @@ "relevant": "selected(${project_role_project-beta}, 'role-data-engineering')" }, { - "type": "select_one sosci_survey", + "type": "select_one iy6st81", "name": "git", "label": { "de": "Git", @@ -133,7 +190,7 @@ "relevant": "selected(${project_role_project-beta}, 'role-data-engineering')" }, { - "type": "select_one sosci_survey", + "type": "select_one iy6st81", "name": "jupyter", "label": { "de": "Jupyter Notebooks", @@ -144,7 +201,7 @@ "relevant": "selected(${project_role_project-gamma}, 'role-machine-learning')" }, { - "type": "select_one sosci_survey", + "type": "select_one iy6st81", "name": "tensorflow", "label": { "de": "TensorFlow / PyTorch", @@ -180,7 +237,20 @@ "appearance": "field-list" }, { - "type": "select_one sosci_survey", + "type": "select_one xe9mo32", + "name": "rating_techniques_header", + "label": { + "de": "Bitte bewerte Deine Erfahrung mit den folgenden Techniken.", + "en": "Please rate your experience with the following techniques:" + }, + "hint": { + "de": "Hier kannst du mehr darüber lesen, wie wir die verschiedenen Erfahrungsstufen definieren: https://docs.correlaid.org/project-manual/data4good-projects#experience-scale", + "en": "Read more about what the different experience labels mean for us: https://docs.correlaid.org/project-manual/data4good-projects#experience-scale" + }, + "appearance": "label" + }, + { + "type": "select_one xe9mo32", "name": "survey_design", "label": { "de": "Entwicklung von Fragebögen", @@ -191,7 +261,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-survey-design')" }, { - "type": "select_one sosci_survey", + "type": "select_one xe9mo32", "name": "indicator_development", "label": { "de": "Entwicklung von Indikatoren", @@ -202,7 +272,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-survey-design')" }, { - "type": "select_one sosci_survey", + "type": "select_one xe9mo32", "name": "data_collection", "label": { "de": "Datenerhebung", @@ -213,7 +283,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-survey-design')" }, { - "type": "select_one sosci_survey", + "type": "select_one xe9mo32", "name": "data_cleaning", "label": { "de": "Datenbereinigung", @@ -224,7 +294,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-data-analysis')" }, { - "type": "select_one sosci_survey", + "type": "select_one xe9mo32", "name": "descriptive_statistics", "label": { "de": "Deskriptive Statistik", @@ -235,7 +305,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-data-analysis')" }, { - "type": "select_one sosci_survey", + "type": "select_one xe9mo32", "name": "data_visualization", "label": { "de": "Datenvisualisierung", @@ -246,7 +316,7 @@ "relevant": "selected(${project_role_project-beta}, 'role-visualization')" }, { - "type": "select_one sosci_survey", + "type": "select_one xe9mo32", "name": "data_engineering", "label": { "de": "Data Engineering", @@ -257,7 +327,7 @@ "relevant": "selected(${project_role_project-beta}, 'role-data-engineering')" }, { - "type": "select_one sosci_survey", + "type": "select_one xe9mo32", "name": "automation", "label": { "de": "Automatisierung", @@ -268,7 +338,7 @@ "relevant": "selected(${project_role_project-beta}, 'role-data-engineering')" }, { - "type": "select_one sosci_survey", + "type": "select_one xe9mo32", "name": "projectplanning", "label": { "de": "Projektplanung", @@ -279,7 +349,7 @@ "relevant": "selected(${project_role_project-gamma}, 'role-project-management')" }, { - "type": "select_one sosci_survey", + "type": "select_one xe9mo32", "name": "mlmodeling", "label": { "de": "ML-Modellierung", @@ -302,7 +372,20 @@ "appearance": "field-list" }, { - "type": "select_one sosci_survey", + "type": "select_one dm1uu76", + "name": "rating_topics_header", + "label": { + "de": "Bitte bewerte Deine Erfahrung mit den folgenden Themen:", + "en": "Please rate your experience with the following topics:" + }, + "hint": { + "de": "Hier kannst du mehr darüber lesen, wie wir die verschiedenen Erfahrungsstufen definieren: https://docs.correlaid.org/project-manual/data4good-projects#experience-scale", + "en": "Read more about what the different experience labels mean for us: https://docs.correlaid.org/project-manual/data4good-projects#experience-scale" + }, + "appearance": "label" + }, + { + "type": "select_one dm1uu76", "name": "wirkungsmessung", "label": { "de": "Wirkungsmessung", @@ -313,7 +396,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-survey-design')" }, { - "type": "select_one sosci_survey", + "type": "select_one dm1uu76", "name": "research_design", "label": { "de": "Research Design", @@ -324,7 +407,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-survey-design')" }, { - "type": "select_one sosci_survey", + "type": "select_one dm1uu76", "name": "survey_research", "label": { "de": "Umfrageforschung", @@ -335,7 +418,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-survey-design')" }, { - "type": "select_one sosci_survey", + "type": "select_one dm1uu76", "name": "data_protection", "label": { "de": "Datenschutz", @@ -346,7 +429,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-survey-design')" }, { - "type": "select_one sosci_survey", + "type": "select_one dm1uu76", "name": "resilience_mental_health", "label": { "de": "Resilienz und Mentale Gesundheit", @@ -357,7 +440,7 @@ "relevant": "selected(${project_role_project-alpha}, 'role-survey-design')" }, { - "type": "select_one sosci_survey", + "type": "select_one dm1uu76", "name": "education_research", "label": { "de": "Bildungsforschung o.ä.", @@ -407,7 +490,7 @@ "appearance": "multiline" }, { - "type": "select_one past_applications", + "type": "select_one il2wc73", "name": "past_applications", "label": { "de": "Hast du dich in der Vergangenheit bereits auf CorrelAid-Projekte beworben?", @@ -478,7 +561,7 @@ "constraint": "regex(., '^[\\w-\\.]+@([\\w-]+\\.)+[\\w-]{2,}$')" }, { - "type": "select_one gender", + "type": "select_one iy7os66", "name": "gender", "label": { "de": "Was ist dein Geschlecht?", @@ -508,7 +591,7 @@ "type": "end_group" }, { - "type": "select_one consent_privacy_policy", + "type": "select_one mj8ty33", "name": "consent_privacy_policy", "label": { "de": "Einwilligung in die Datenschutzerklärung", @@ -519,27 +602,11 @@ "en": "I have read the disclaimer and hereby give my consent to the collection, processing and use of my personal data (gender and personal skills) for the team selection process and for internal and external reporting as well as presentation purposes. In addition, I expressly grant permission to use my contact details (name and e-mail) to contact me and to share them with the project team if the application is successful." }, "required": "yes" - }, - { - "type": "note", - "name": "Hallo", - "label": { - "de": "Hallo!", - "en": "Hello!" - } - }, - { - "type": "note", - "name": "Disclaimer", - "label": { - "de": "Disclaimer", - "en": "Disclaimer" - } } ], "choices": [ { - "list_name": "project_id", + "list_name": "mb5co98", "name": "project-alpha", "label": { "de": "Projekt Alpha", @@ -547,7 +614,7 @@ } }, { - "list_name": "project_id", + "list_name": "mb5co98", "name": "project-beta", "label": { "de": "Projekt Beta", @@ -555,7 +622,7 @@ } }, { - "list_name": "project_id", + "list_name": "mb5co98", "name": "project-gamma", "label": { "de": "Projekt Gamma", @@ -563,7 +630,7 @@ } }, { - "list_name": "project_role_project-alpha", + "list_name": "cl-role-project-alpha", "name": "role-survey-design", "label": { "de": "Umfragedesign und Datenerhebung", @@ -571,7 +638,7 @@ } }, { - "list_name": "project_role_project-alpha", + "list_name": "cl-role-project-alpha", "name": "role-data-analysis", "label": { "de": "Datenanalyse", @@ -579,7 +646,7 @@ } }, { - "list_name": "project_role_project-alpha", + "list_name": "cl-role-project-alpha", "name": "team_coordinator", "label": { "de": "Teamkoordinator:in", @@ -587,7 +654,7 @@ } }, { - "list_name": "project_role_project-alpha", + "list_name": "cl-role-project-alpha", "name": "team_trainee", "label": { "de": "Team Trainee", @@ -595,7 +662,7 @@ } }, { - "list_name": "project_role_project-beta", + "list_name": "cl-role-project-beta", "name": "role-visualization", "label": { "de": "Datenvisualisierung", @@ -603,7 +670,7 @@ } }, { - "list_name": "project_role_project-beta", + "list_name": "cl-role-project-beta", "name": "role-data-engineering", "label": { "de": "Data Engineering", @@ -611,7 +678,7 @@ } }, { - "list_name": "project_role_project-beta", + "list_name": "cl-role-project-beta", "name": "team_coordinator", "label": { "de": "Teamkoordinator:in", @@ -619,7 +686,7 @@ } }, { - "list_name": "project_role_project-beta", + "list_name": "cl-role-project-beta", "name": "team_trainee", "label": { "de": "Team Trainee", @@ -627,7 +694,7 @@ } }, { - "list_name": "project_role_project-gamma", + "list_name": "cl-role-project-gamma", "name": "role-project-management", "label": { "de": "Projektmanagement", @@ -635,7 +702,7 @@ } }, { - "list_name": "project_role_project-gamma", + "list_name": "cl-role-project-gamma", "name": "role-machine-learning", "label": { "de": "Machine Learning", @@ -643,7 +710,7 @@ } }, { - "list_name": "project_role_project-gamma", + "list_name": "cl-role-project-gamma", "name": "team_coordinator", "label": { "de": "Teamkoordinator:in", @@ -651,7 +718,7 @@ } }, { - "list_name": "project_role_project-gamma", + "list_name": "cl-role-project-gamma", "name": "team_trainee", "label": { "de": "Team Trainee", @@ -659,7 +726,71 @@ } }, { - "list_name": "sosci_survey", + "list_name": "iy6st81", + "name": "beginner", + "label": { + "de": "Anfänger:in", + "en": "Beginner" + } + }, + { + "list_name": "iy6st81", + "name": "user", + "label": { + "de": "Benutzer:in", + "en": "User" + } + }, + { + "list_name": "iy6st81", + "name": "advanced", + "label": { + "de": "Fortgeschrittene:r", + "en": "Advanced" + } + }, + { + "list_name": "iy6st81", + "name": "expert", + "label": { + "de": "Expert:in", + "en": "Expert" + } + }, + { + "list_name": "xe9mo32", + "name": "beginner", + "label": { + "de": "Anfänger:in", + "en": "Beginner" + } + }, + { + "list_name": "xe9mo32", + "name": "user", + "label": { + "de": "Benutzer:in", + "en": "User" + } + }, + { + "list_name": "xe9mo32", + "name": "advanced", + "label": { + "de": "Fortgeschrittene:r", + "en": "Advanced" + } + }, + { + "list_name": "xe9mo32", + "name": "expert", + "label": { + "de": "Expert:in", + "en": "Expert" + } + }, + { + "list_name": "dm1uu76", "name": "beginner", "label": { "de": "Anfänger:in", @@ -667,7 +798,7 @@ } }, { - "list_name": "sosci_survey", + "list_name": "dm1uu76", "name": "user", "label": { "de": "Benutzer:in", @@ -675,7 +806,7 @@ } }, { - "list_name": "sosci_survey", + "list_name": "dm1uu76", "name": "advanced", "label": { "de": "Fortgeschrittene:r", @@ -683,7 +814,7 @@ } }, { - "list_name": "sosci_survey", + "list_name": "dm1uu76", "name": "expert", "label": { "de": "Expert:in", @@ -691,7 +822,7 @@ } }, { - "list_name": "past_applications", + "list_name": "il2wc73", "name": "successful", "label": { "de": "Ja, und ich war mindestens einmal erfolgreich / Teil eines Projektteams.", @@ -699,7 +830,7 @@ } }, { - "list_name": "past_applications", + "list_name": "il2wc73", "name": "not_successful", "label": { "de": "Ja, aber ich war nicht erfolgreich / ich wurde bis jetzt immer abgelehnt.", @@ -707,7 +838,7 @@ } }, { - "list_name": "past_applications", + "list_name": "il2wc73", "name": "first_application", "label": { "de": "Nein, das ist meine erste Bewerbung.", @@ -715,7 +846,7 @@ } }, { - "list_name": "gender", + "list_name": "iy7os66", "name": "female", "label": { "de": "Weiblich", @@ -723,7 +854,7 @@ } }, { - "list_name": "gender", + "list_name": "iy7os66", "name": "male", "label": { "de": "Männlich", @@ -731,7 +862,7 @@ } }, { - "list_name": "gender", + "list_name": "iy7os66", "name": "non_binary", "label": { "de": "Nicht-binär / non-binary", @@ -739,7 +870,7 @@ } }, { - "list_name": "gender", + "list_name": "iy7os66", "name": "not_disclosed", "label": { "de": "Das möchte ich nicht angeben", @@ -747,7 +878,7 @@ } }, { - "list_name": "gender", + "list_name": "iy7os66", "name": "self_identification", "label": { "de": "Mein Geschlecht ist (eigene Angabe):", @@ -755,7 +886,7 @@ } }, { - "list_name": "consent_privacy_policy", + "list_name": "mj8ty33", "name": "yes", "label": { "de": "Ja", @@ -765,7 +896,7 @@ ], "settings": [ { - "default_language": "German (de)", + "default_language": "Deutsch (de)", "form_title": "testB", "version": "vw6VTbhmVCMEtuN7LhcmjP", "style": "theme-grid no-text-transform" diff --git a/tests/fixtures/surveys/validation_relevance_survey/ddi.xml b/tests/fixtures/surveys/validation_relevance_survey/ddi.xml index 6285f3b..cc7d4a5 100644 --- a/tests/fixtures/surveys/validation_relevance_survey/ddi.xml +++ b/tests/fixtures/surveys/validation_relevance_survey/ddi.xml @@ -9,6 +9,7 @@ 2020-01-01 + en @@ -30,6 +31,7 @@ . != '' Username cannot be empty + welcome @@ -69,6 +71,7 @@ Do you consent? + yes_no diff --git a/tests/fixtures/surveys/validation_relevance_survey/ddi2xlsform.json b/tests/fixtures/surveys/validation_relevance_survey/ddi2xlsform.json index 862fc66..2e1f0c9 100644 --- a/tests/fixtures/surveys/validation_relevance_survey/ddi2xlsform.json +++ b/tests/fixtures/surveys/validation_relevance_survey/ddi2xlsform.json @@ -2,7 +2,7 @@ "survey": [ { "type": "note", - "name": "username_note", + "name": "welcome", "label": "Validation and Relevance Test Survey" }, { @@ -27,7 +27,7 @@ "constraint": ". > 0 and . < 1000" }, { - "type": "select_one consent", + "type": "select_one yes_no", "name": "consent", "label": "Do you consent?" }, @@ -79,18 +79,19 @@ ], "choices": [ { - "list_name": "consent", + "list_name": "yes_no", "name": "yes", "label": "Yes" }, { - "list_name": "consent", + "list_name": "yes_no", "name": "no", "label": "No" } ], "settings": [ { + "default_language": "en", "form_title": "validation_relevance_survey" } ] diff --git a/tests/ts/contract/canonicalInstrument.ts b/tests/ts/contract/canonicalInstrument.ts index 5aa0242..88c328c 100644 --- a/tests/ts/contract/canonicalInstrument.ts +++ b/tests/ts/contract/canonicalInstrument.ts @@ -1,21 +1,15 @@ /** - * An {@link Instrument} in the form the DDI round trip compares (#154): what - * a DDI codebook can give back, with the documented losses of - * `src/pipelines/ddi2xlsform/README.md` folded out. Two instruments that are - * the same form have equal canonical forms. + * An {@link Instrument} in the form the DDI round trips compare (#154, #160): + * what a DDI codebook can give back, with the losses + * `src/pipelines/ddi2xlsform/README.md` lists folded out. Two instruments + * that are the same form have equal canonical forms. */ import { APPEARANCES } from '../../../src/generated/Appearances.js'; import { TYPE_MAP } from '../../../src/generated/DdiMappings.js'; -import { TYPE_MAPPINGS } from '../../../src/generated/TypeMappings.js'; -import { METADATA_ROW_TYPES } from '../../../src/conventions/metadata.js'; -import { - OTHER_CODE, - OTHER_SUFFIX, - otherCompanionRelevance, -} from '../../../src/conventions/other.js'; +import { isMetadataType } from '../../../src/conventions/metadata.js'; +import { OTHER_CODE } from '../../../src/conventions/other.js'; import { isExclusive } from '../../../src/conventions/exclusive.js'; import { languageTagOf } from '../../../src/utils/languageUtils.js'; -import { parseParameters } from '../../../src/utils/parameters.js'; import type { Instrument, Item, @@ -30,14 +24,16 @@ const NO_DATA = new Set( .filter(([, a]) => a.carriesData === false) .map(([name]) => name), ); -const METADATA = new Set(METADATA_ROW_TYPES); const ALIASES: Record = { int: 'integer', string: 'text' }; interface Ctx { instrument: Instrument; - /** Language key → the tag compared (`''` for a one-language form). */ + /** Language key → the key compared. */ lang: (key: string) => string; - notes: string[]; + /** The lists the kept questions use. */ + lists: Set; + /** Lists of a select_multiple with an explicit other pair. */ + multiOther: Set; } function text(t: Text | undefined, ctx: Ctx): Record { @@ -49,164 +45,152 @@ function text(t: Text | undefined, ctx: Ctx): Record { return out; } -/** Parameters DDI keeps: no guidance_hint (it's ivuInstr), no default range bounds. */ +/** Parameters without guidance_hint: DDI has it as ivuInstr. */ function parameters(q: QuestionItem): string { - const values = parseParameters( - q.parameters - .split(';') - .filter((p) => !/^\s*guidance_hint\s*=/.test(p)) - .join(' '), - ); - const defaults = TYPE_MAPPINGS[q.type]?.parameters ?? {}; - return Object.entries(values) - .filter(([k, v]) => defaults[k] !== v) - .map(([k, v]) => `${k}=${v}`) - .sort() + return q.parameters + .split(';') + .filter((p) => !/^\s*guidance_hint\s*=/.test(p)) + .join(' ') + .split(/[\s,]+/) + .filter(Boolean) .join(' '); } -/** Whether DDI has a variable for it. */ -function emitted(q: QuestionItem): boolean { +/** Whether the DDI has it: a variable, a note or a row without data. */ +function kept(q: QuestionItem): boolean { const type = ALIASES[q.type] ?? q.type; - return ( - !METADATA.has(type) && - !NO_DATA.has(q.appearance) && - (type in TYPE_MAP || type === 'note') - ); -} - -function choices(q: QuestionItem, ctx: Ctx): Canon[] { - const list = q.list ? (ctx.instrument.lists[q.list] ?? []) : []; - const out: Canon[] = list.map((c) => - // The "other" answer's text is LimeSurvey's or the convention's. - c.name === OTHER_CODE - ? { name: c.name } - : { - name: c.name, - label: text(c.label, ctx), - ...(isExclusive(c.row) ? { exclusive: true } : {}), - }, - ); - if (q.orOther && !out.some((c) => c.name === OTHER_CODE)) { - out.push({ name: OTHER_CODE }); - } - return out; + if (isMetadataType(type) || NO_DATA.has(q.appearance)) return !!q.name; + return type in TYPE_MAP || type === 'note'; } function question(q: QuestionItem, ctx: Ctx): Canon { const type = ALIASES[q.type] ?? q.type; + if (q.list) ctx.lists.add(q.list); const hasConstraint = !!q.constraint.trim(); return { name: q.name, type, + list: q.list, + file: q.file, + orOther: q.orOther, label: text(q.label, ctx), hint: text(q.hint, ctx), guidance: text(q.guidanceHint, ctx), relevant: q.relevant.trim(), constraint: q.constraint.trim(), + // A message without a constraint is not in the DDI. constraintMessage: hasConstraint ? text(q.constraintMessage, ctx) : {}, required: q.required, default: q.default, appearance: q.appearance, parameters: parameters(q), - file: q.file, - choices: choices(q, ctx), - }; -} - -/** The or_other shorthand's companion, as the explicit pair writes it. */ -function companion(q: QuestionItem): Canon { - return { - name: q.name + OTHER_SUFFIX, - type: 'text', - relevant: otherCompanionRelevance(q.type, q.name), }; } -/** A companion question compares by name, type and condition: its text is LimeSurvey's. */ -function isCompanion(name: string, all: Set): boolean { - return ( - name.endsWith(OTHER_SUFFIX) && all.has(name.slice(0, -OTHER_SUFFIX.length)) - ); -} - -function items(list: Item[], ctx: Ctx, all: Set): Canon[] { +function items(list: Item[], ctx: Ctx): Canon[] { const out: Canon[] = []; for (const item of list) { if (item.kind === 'group') { - const children = items(item.children, ctx, all); - if (!children.length) continue; + // A group with nothing in it is not carried. + if (!item.children.length) continue; const label = text(item.label, ctx); out.push({ group: item.name, - label: Object.keys(label).length ? label : { '': item.name }, + // A group without a label is labelled with its name. + label: Object.keys(label).length ? label : text({ '': item.name }, ctx), hint: text(item.hint, ctx), relevant: item.relevant.trim(), appearance: item.appearance, - children, + children: items(item.children, ctx), }); + } else if (!kept(item)) { continue; - } - if (!emitted(item)) continue; - if (item.type === 'note') { - // Note rows come back without their names, merged, orphans at the end. - const t = text(item.label, ctx); - for (const part of Object.values(t).join('\n\n').split(/\n\n/)) { - if (part.trim()) ctx.notes.push(part.trim()); - } - continue; - } - const q = question(item, ctx); - if (isCompanion(item.name, all)) { - out.push({ name: q.name, type: q.type, relevant: q.relevant }); + } else if (isMetadataType(ALIASES[item.type] ?? item.type)) { + out.push({ metadata: item.name, type: item.type }); + } else if (item.type === 'note') { + out.push({ + note: item.name, + label: text(item.label, ctx), + hint: text(item.hint, ctx), + relevant: item.relevant.trim(), + appearance: item.appearance, + }); } else { - out.push(q); - } - if (item.orOther && !all.has(item.name + OTHER_SUFFIX)) { - out.push(companion(item)); + out.push(question(item, ctx)); } } return out; } -function names(list: Item[], into = new Set()): Set { +/** Mark the lists of explicit select_multiple pairs (their other label is the convention's). */ +function multiOthers(list: Item[], into: Set, names: Set) { + for (const item of list) { + if (item.kind === 'group') multiOthers(item.children, into, names); + else if (item.type === 'select_multiple' && item.list) { + if (names.has(`${item.name}_other`)) into.add(item.list); + } + } +} + +function allNames(list: Item[], into = new Set()): Set { for (const item of list) { into.add(item.name); - if (item.kind === 'group') names(item.children, into); + if (item.kind === 'group') allNames(item.children, into); } return into; } -/** The canonical form of an instrument for the DDI round trip. */ +function lists(ctx: Ctx): Canon { + const out: Canon = {}; + for (const name of [...ctx.lists].sort()) { + out[name] = (ctx.instrument.lists[name] ?? []).map((c) => + c.name === OTHER_CODE && ctx.multiOther.has(name) + ? { name: c.name } + : { + name: c.name, + label: text(c.label, ctx), + ...(isExclusive(c.row) ? { exclusive: true } : {}), + }, + ); + } + return out; +} + +function settings(instrument: Instrument, ctx: Ctx): Canon { + const out: Canon = {}; + for (const [key, value] of Object.entries(instrument.settings)) { + if (typeof value === 'string' || typeof value === 'number') { + if (String(value).trim()) out[key] = String(value).trim(); + } else if (key === 'form_title' && value && typeof value === 'object') { + out[key] = text(value as Text, ctx); + } + } + return out; +} + +/** The canonical form of an instrument for the DDI round trips. */ export function canonical(instrument: Instrument): Canon { - const tags = instrument.languages.map((l) => (l ? languageTagOf(l) : null)); - const single = new Set(tags.filter(Boolean)).size <= 1; - // An untagged column of a multilingual form is its base language's text. - const base = - (instrument.defaultLanguage && languageTagOf(instrument.defaultLanguage)) || - tags.find(Boolean) || - ''; + const keys = instrument.languages.filter(Boolean); + const tags = keys.map((k) => languageTagOf(k)); + const multilingual = new Set(tags.filter(Boolean)).size > 1; + // A multilingual form's base: default_language, else its first language. + const baseTag = + instrument.defaultLanguage ?? (multilingual ? tags[0] : null) ?? ''; + const baseKey = keys.find((k) => languageTagOf(k) === baseTag) ?? baseTag; const ctx: Ctx = { instrument, - lang: (key) => (single ? '' : key ? (languageTagOf(key) ?? key) : base), - notes: [], + // One language: every text is it. Several: an untagged column is the base's. + lang: (key) => (!multilingual ? '' : key || baseKey), + lists: new Set(), + multiOther: new Set(), }; - const body = items(instrument.body, ctx, names(instrument.body)); - const s = instrument.settings; - const setting = (k: string): unknown => - typeof s[k] === 'string' || typeof s[k] === 'number' - ? String(s[k]) - : s[k] && typeof s[k] === 'object' - ? text(s[k] as Text, ctx) - : ''; + multiOthers(instrument.body, ctx.multiOther, allNames(instrument.body)); + const body = items(instrument.body, ctx); return { - settings: { - form_title: setting('form_title'), - form_id: setting('form_id') || setting('id_string'), - version: setting('version'), - style: setting('style'), - }, + base: baseTag, + languages: multilingual ? keys : [], + settings: settings(instrument, ctx), + lists: lists(ctx), body, - notes: ctx.notes.sort(), }; } diff --git a/tests/ts/contract/ddiRoundtrip.test.ts b/tests/ts/contract/ddiRoundtrip.test.ts index e36f661..b82fd28 100644 --- a/tests/ts/contract/ddiRoundtrip.test.ts +++ b/tests/ts/contract/ddiRoundtrip.test.ts @@ -2,81 +2,21 @@ * The DDI round trip (#154): XLSForm → DDI → Instrument gives back the * XLSForm's Instrument, compared on the model (`canonicalInstrument.ts`), for * every whole-survey fixture and registry entity, and through the XLSForm - * sheets `ddi2xlsform` writes. + * sheets `ddi2xlsform` writes. The codebook of those sheets is the codebook + * they were read from (#160). The round trips through LimeSurvey are + * `lstsvDdiRoundtrip.test.ts`. */ import * as fs from 'node:fs'; import * as path from 'node:path'; -import { fileURLToPath } from 'node:url'; import { describe, test, expect } from 'vitest'; import { buildDdiXml } from '../../../src/pipelines/xlsform2ddi/index.js'; import { instrumentFromDdi } from '../../../src/instrument/fromDdi.js'; import { instrumentFromXlsform } from '../../../src/instrument/fromXlsform.js'; -import { XLSLoader } from '../../../src/xlsform/loader.js'; import { ddiToXlsform } from '../../../src/pipelines/ddi2xlsform/index.js'; import { canonical } from './canonicalInstrument.js'; - -type Row = Record; - -const ROOT = path.resolve( - path.dirname(fileURLToPath(import.meta.url)), - '../../..', -); - -interface Case { - name: string; - survey: Row[]; - choices: Row[]; - settings: Row[]; -} - -function load(name: string, dir: string, json: string): Case | null { - if (fs.existsSync(json)) { - const f = JSON.parse(fs.readFileSync(json, 'utf-8')) as Record< - string, - Row[] - >; - return { - name, - survey: f.survey ?? [], - choices: f.choices ?? [], - settings: f.settings ?? [], - }; - } - const xlsx = path.join(dir, 'xlsform.xlsx'); - if (!fs.existsSync(xlsx)) return null; - const data = XLSLoader.parseXLSData(fs.readFileSync(xlsx), { - skipValidation: true, - }); - return { - name, - survey: data.surveyData as unknown as Row[], - choices: data.choicesData as unknown as Row[], - settings: data.settingsData as unknown as Row[], - }; -} - -function cases(): Case[] { - const out: Case[] = []; - const surveys = path.join(ROOT, 'tests/fixtures/surveys'); - for (const name of fs.readdirSync(surveys).sort()) { - const dir = path.join(surveys, name); - const c = load(name, dir, path.join(dir, 'xlsform.json')); - if (c) out.push(c); - } - const entities = path.join(ROOT, 'registry/entities'); - for (const name of fs.readdirSync(entities).sort()) { - const dir = path.join(entities, name); - const c = load( - `entity:${name}`, - dir, - path.join(dir, 'fixtures/xlsform.json'), - ); - if (c) out.push(c); - } - return out; -} +import { cases, ROOT } from './roundtripCases.js'; const CASES = cases(); @@ -114,6 +54,22 @@ describe('XLSForm → DDI → XLSForm sheets', () => { }); }); +describe('DDI → XLSForm → DDI gives the same codebook', () => { + test.each(CASES)('$name', (c) => { + const xml = buildDdiXml(c.survey, c.choices, { + prodDate: '2020-01-01', + settings: c.settings[0] ?? {}, + }); + const sheets = ddiToXlsform(xml); + expect( + buildDdiXml(sheets.survey, sheets.choices, { + prodDate: '2020-01-01', + settings: sheets.settings[0] ?? {}, + }), + ).toBe(xml); + }); +}); + describe('the comparison is not blind', () => { const c = CASES.find((x) => x.name === 'validation_relevance_survey')!; const xml = buildDdiXml(c.survey, c.choices, { prodDate: '2020-01-01' }); diff --git a/tests/ts/contract/ddiRoundtripGenerated.test.ts b/tests/ts/contract/ddiRoundtripGenerated.test.ts index daeeaa0..4bca1b8 100644 --- a/tests/ts/contract/ddiRoundtripGenerated.test.ts +++ b/tests/ts/contract/ddiRoundtripGenerated.test.ts @@ -1,19 +1,38 @@ /** - * The DDI round trip on generated forms (#154): random valid XLSForms from - * the registry's types (groups, grids, languages, logic, fields) survive - * XLSForm → DDI → Instrument, compared as `ddiRoundtrip.test.ts` does. + * The DDI round trips on generated forms (#154, #160): random valid + * XLSForms from the registry's types, with groups, grids, matrices, + * languages, logic, fields, named and shared lists, `or_other` and explicit + * other pairs, runs of notes, groups of notes only, metadata rows and + * settings. Compared as `ddiRoundtrip.test.ts` and + * `lstsvDdiRoundtrip.test.ts` do. */ import { describe, test, expect } from 'vitest'; import fc from 'fast-check'; import { buildDdiXml } from '../../../src/pipelines/xlsform2ddi/index.js'; +import { ddiToXlsform } from '../../../src/pipelines/ddi2xlsform/index.js'; +import { XLSFormToTSVConverter } from '../../../src/pipelines/xlsform2lstsv/index.js'; +import { lstsvToXlsform } from '../../../src/pipelines/lstsv2xlsform/index.js'; +import { lstsvToDdiXml } from '../../../src/pipelines/lstsv2ddi/index.js'; import { instrumentFromDdi } from '../../../src/instrument/fromDdi.js'; +import { instrumentFromLstsv } from '../../../src/instrument/fromLstsv.js'; import { instrumentFromXlsform } from '../../../src/instrument/fromXlsform.js'; +import { parseLstsv } from '../../../src/lstsv/parser.js'; +import type { + ChoiceRow, + SettingsRow, + SurveyRow, +} from '../../../src/xlsform/types.js'; import { canonical } from './canonicalInstrument.js'; type Row = Record; const LANGS = ['Deutsch (de)', 'English (en)']; +/** Runs per property; `FT_ROUNDTRIP_RUNS=5000` for a longer search. */ +const RUNS = Number(process.env['FT_ROUNDTRIP_RUNS'] ?? 200); +/** The LimeSurvey paths convert twice more. */ +const LS_RUNS = Math.ceil(RUNS / 2); +const quiet = () => {}; /** A text that survives XML and isn't read as HTML. */ const word = fc @@ -29,6 +48,8 @@ interface Texts { const texts: fc.Arbitrary = fc.record({ one: word, de: word, en: word }); const SIMPLE = ['text', 'integer', 'decimal', 'date', 'time', 'range']; +const METADATA = ['start', 'end', 'today', 'deviceid']; +const RANGE_PARAMETERS = ['start=0 end=20 step=2', 'start=1 end=10', 'step=1']; type Question = | { @@ -40,41 +61,73 @@ type Question = ref: boolean; constraint: boolean; appearance: boolean; - params: boolean; + params?: string; } | { kind: 'select'; multiple: boolean; t: Texts; choices: Texts[]; - orOther: boolean; + /** Name the list apart from the question, or share the last one. */ + list: 'own' | 'named' | 'shared'; + other: 'none' | 'shorthand' | 'pair'; exclusive: boolean; ref: boolean; } - | { kind: 'note'; t: Texts }; + | { kind: 'note'; t: Texts; hint?: Texts; ref: boolean } + | { kind: 'metadata'; type: string }; const question: fc.Arbitrary = fc.oneof( + { + weight: 3, + arbitrary: fc.record({ + kind: fc.constant('simple' as const), + type: fc.constantFrom(...SIMPLE), + t: texts, + hint: fc.option(texts, { nil: undefined }), + required: fc.boolean(), + ref: fc.boolean(), + constraint: fc.boolean(), + appearance: fc.boolean(), + params: fc.option(fc.constantFrom(...RANGE_PARAMETERS), { + nil: undefined, + }), + }), + }, + { + weight: 3, + arbitrary: fc.record({ + kind: fc.constant('select' as const), + multiple: fc.boolean(), + t: texts, + choices: fc.array(texts, { minLength: 1, maxLength: 4 }), + list: fc.constantFrom( + 'own' as const, + 'named' as const, + 'shared' as const, + ), + other: fc.constantFrom( + 'none' as const, + 'shorthand' as const, + 'pair' as const, + ), + exclusive: fc.boolean(), + ref: fc.boolean(), + }), + }, + { + weight: 2, + arbitrary: fc.record({ + kind: fc.constant('note' as const), + t: texts, + hint: fc.option(texts, { nil: undefined }), + ref: fc.boolean(), + }), + }, fc.record({ - kind: fc.constant('simple' as const), - type: fc.constantFrom(...SIMPLE), - t: texts, - hint: fc.option(texts, { nil: undefined }), - required: fc.boolean(), - ref: fc.boolean(), - constraint: fc.boolean(), - appearance: fc.boolean(), - params: fc.boolean(), - }), - fc.record({ - kind: fc.constant('select' as const), - multiple: fc.boolean(), - t: texts, - choices: fc.array(texts, { minLength: 1, maxLength: 4 }), - orOther: fc.boolean(), - exclusive: fc.boolean(), - ref: fc.boolean(), + kind: fc.constant('metadata' as const), + type: fc.constantFrom(...METADATA), }), - fc.record({ kind: fc.constant('note' as const), t: texts }), ); type Block = @@ -86,22 +139,32 @@ type Block = inner: Question[]; nested: Question[]; } - | { kind: 'grid'; t: Texts; choices: Texts[]; rows: Texts[] }; + | { kind: 'grid'; t: Texts; choices: Texts[]; rows: Texts[] } + | { kind: 'matrix'; t: Texts; choices: Texts[]; rows: Texts[] }; const block: fc.Arbitrary = fc.oneof( { - weight: 3, + weight: 4, arbitrary: question.map((q) => ({ kind: 'question' as const, q })), }, + { + weight: 2, + arbitrary: fc.record({ + kind: fc.constant('group' as const), + t: texts, + relevant: fc.boolean(), + inner: fc.array(question, { minLength: 1, maxLength: 3 }), + nested: fc.array(question, { maxLength: 2 }), + }), + }, fc.record({ - kind: fc.constant('group' as const), + kind: fc.constant('grid' as const), t: texts, - relevant: fc.boolean(), - inner: fc.array(question, { minLength: 1, maxLength: 3 }), - nested: fc.array(question, { maxLength: 2 }), + choices: fc.array(texts, { minLength: 1, maxLength: 3 }), + rows: fc.array(texts, { minLength: 1, maxLength: 3 }), }), fc.record({ - kind: fc.constant('grid' as const), + kind: fc.constant('matrix' as const), t: texts, choices: fc.array(texts, { minLength: 1, maxLength: 3 }), rows: fc.array(texts, { minLength: 1, maxLength: 3 }), @@ -110,14 +173,23 @@ const block: fc.Arbitrary = fc.oneof( const form = fc.record({ multilingual: fc.boolean(), + settings: fc.record({ + form_title: fc.option(word, { nil: undefined }), + form_id: fc.option(fc.constantFrom('f1', 'survey_2'), { nil: undefined }), + version: fc.option(fc.constantFrom('1', '2024-01'), { nil: undefined }), + style: fc.option(fc.constant('pages'), { nil: undefined }), + }), blocks: fc.array(block, { minLength: 1, maxLength: 6 }), }); +type Form = typeof form extends fc.Arbitrary ? F : never; + /** Serialize a generated form to XLSForm rows. */ -function sheets(f: fc.RecordValue<{ multilingual: boolean; blocks: Block[] }>) { +function sheets(f: Form) { const survey: Row[] = []; const choices: Row[] = []; let n = 0; + let lastList = ''; const numbers: string[] = []; const answered: string[] = []; const put = (row: Row, column: string, t: Texts) => { @@ -135,32 +207,98 @@ function sheets(f: fc.RecordValue<{ multilingual: boolean; blocks: Block[] }>) { } choices.push(row); }); + const condition = () => + numbers.length ? `\${${numbers[0]}} > 3` : undefined; + const addSelect = ( + q: Extract, + name: string, + ) => { + const row: Row = { name }; + put(row, 'label', q.t); + let listName = q.list === 'named' ? `l_${name}` : name; + if (q.list === 'shared' && lastList) listName = lastList; + else { + list(listName, q.choices, q.multiple && q.exclusive); + if (q.other === 'pair') { + const other: Row = { list_name: listName, name: 'other' }; + put(other, 'label', q.t); + choices.push(other); + } + lastList = listName; + } + const base = q.multiple ? 'select_multiple' : 'select_one'; + const shorthand = q.other === 'shorthand' ? ' or_other' : ''; + row.type = `${base} ${listName}${shorthand}`; + if (q.ref && answered.length) row.relevant = `\${${answered[0]}} != ''`; + survey.push(row); + if (q.other === 'pair' && !(q.list === 'shared' && lastList !== listName)) { + const companion: Row = { + type: 'text', + name: `${name}_other`, + relevant: q.multiple + ? `selected(\${${name}}, 'other')` + : `\${${name}} = 'other'`, + }; + put(companion, 'label', q.t); + survey.push(companion); + } + }; const addQuestion = (q: Question) => { const name = `q${++n}`; + if (q.kind === 'metadata') { + survey.push({ type: q.type, name }); + return; + } + if (q.kind === 'select') { + addSelect(q, name); + answered.push(name); + return; + } const row: Row = { name }; put(row, 'label', q.t); if (q.kind === 'note') { row.type = 'note'; - } else if (q.kind === 'select') { - list(name, q.choices, q.multiple && q.exclusive); - row.type = `${q.multiple ? 'select_multiple' : 'select_one'} ${name}${q.orOther ? ' or_other' : ''}`; - if (q.ref && answered.length) row.relevant = `\${${answered[0]}} != ''`; - } else { - row.type = q.type; if (q.hint) put(row, 'hint', q.hint); - if (q.required) row.required = 'yes'; - if (q.ref && numbers.length) row.relevant = `\${${numbers[0]}} > 3`; - if (q.constraint && q.type === 'integer') { - row.constraint = '. >= 1 and . <= 10'; - put(row, 'constraint_message', q.t); - } - if (q.appearance && q.type === 'text') row.appearance = 'multiline'; - if (q.params && q.type === 'range') - row.parameters = 'start=0 end=20 step=2'; - if (q.type === 'integer' || q.type === 'decimal') numbers.push(name); + const c = q.ref ? condition() : undefined; + if (c) row.relevant = c; + survey.push(row); + return; + } + row.type = q.type; + if (q.hint) put(row, 'hint', q.hint); + if (q.required) row.required = 'yes'; + const c = q.ref ? condition() : undefined; + if (c) row.relevant = c; + if (q.constraint && q.type === 'integer') { + row.constraint = '. >= 1 and . <= 10'; + put(row, 'constraint_message', q.t); } + if (q.appearance && q.type === 'text') row.appearance = 'multiline'; + if (q.params && q.type === 'range') row.parameters = q.params; + if (q.type === 'integer' || q.type === 'decimal') numbers.push(name); survey.push(row); - if (q.kind !== 'note') answered.push(name); + answered.push(name); + }; + const addRows = ( + b: Extract, + name: string, + ) => { + list(name, b.choices, false); + if (b.kind === 'matrix') { + const header: Row = { + type: `select_one ${name}`, + name: `h${++n}`, + appearance: 'label', + }; + put(header, 'label', b.t); + survey.push(header); + } + for (const r of b.rows) { + const member: Row = { type: `select_one ${name}`, name: `q${++n}` }; + put(member, 'label', r); + if (b.kind === 'matrix') member.appearance = 'list-nolabel'; + survey.push(member); + } }; for (const b of f.blocks) { if (b.kind === 'question') addQuestion(b.q); @@ -181,38 +319,134 @@ function sheets(f: fc.RecordValue<{ multilingual: boolean; blocks: Block[] }>) { survey.push({ type: 'end_group' }); } else { const name = `grid${++n}`; - const row: Row = { type: 'begin_group', name, appearance: 'table-list' }; + const row: Row = { + type: 'begin_group', + name, + appearance: b.kind === 'grid' ? 'table-list' : 'field-list', + }; put(row, 'label', b.t); survey.push(row); - list(name, b.choices, false); - for (const r of b.rows) { - const member: Row = { type: `select_one ${name}`, name: `q${++n}` }; - put(member, 'label', r); - survey.push(member); - } + addRows(b, name); survey.push({ type: 'end_group' }); } } - const settings: Row[] = f.multilingual - ? [{ default_language: LANGS[0] }] - : []; - return { survey, choices, settings }; + const settings: Row = Object.fromEntries( + Object.entries(f.settings).filter(([, v]) => v !== undefined), + ); + if (f.multilingual) settings.default_language = LANGS[0]; + return { + survey, + choices, + settings: Object.keys(settings).length ? [settings] : [], + }; +} + +type Sheets = ReturnType; + +function codebook(s: Pick) { + return buildDdiXml(s.survey, s.choices, { + prodDate: '2020-01-01', + settings: s.settings[0] ?? {}, + }); +} + +const model = (s: { + survey: unknown[]; + choices: unknown[]; + settings: unknown[]; +}) => + canonical( + instrumentFromXlsform( + s.survey as Row[], + s.choices as Row[], + s.settings as Row[], + ), + ); + +/** The form's LimeSurvey TSV, or `null` for one LimeSurvey can't hold. */ +async function toTsv(s: { + survey: unknown[]; + choices: unknown[]; + settings: unknown[]; +}): Promise { + try { + return await new XLSFormToTSVConverter().convert( + s.survey as SurveyRow[], + s.choices as ChoiceRow[], + s.settings as SettingsRow[], + ); + } catch { + return null; + } } describe('generated forms', () => { test('XLSForm → DDI → Instrument gives the form back', () => { fc.assert( fc.property(form, (f) => { - const { survey, choices, settings } = sheets(f); - const xml = buildDdiXml(survey, choices, { + const s = sheets(f); + expect(canonical(instrumentFromDdi(codebook(s)))).toEqual(model(s)); + }), + { numRuns: RUNS }, + ); + }); + + test('XLSForm → DDI → XLSForm gives the form back', () => { + fc.assert( + fc.property(form, (f) => { + const s = sheets(f); + expect(model(ddiToXlsform(codebook(s)))).toEqual(model(s)); + }), + { numRuns: RUNS }, + ); + }); + + test('DDI → XLSForm → DDI gives the same codebook', () => { + fc.assert( + fc.property(form, (f) => { + const xml = codebook(sheets(f)); + expect(codebook(ddiToXlsform(xml))).toBe(xml); + }), + { numRuns: RUNS }, + ); + }); + + test('XLSForm → DDI → XLSForm → LimeSurvey is XLSForm → LimeSurvey', async () => { + let converted = 0; + await fc.assert( + fc.asyncProperty(form, async (f) => { + const s = sheets(f); + const tsv = await toTsv(s); + if (tsv === null) return; + converted++; + expect(await toTsv(ddiToXlsform(codebook(s)))).toBe(tsv); + }), + { numRuns: LS_RUNS }, + ); + expect(converted).toBeGreaterThan(LS_RUNS / 2); + }); + + test('XLSForm → LimeSurvey → DDI → XLSForm is XLSForm → LimeSurvey → XLSForm', async () => { + await fc.assert( + fc.asyncProperty(form, async (f) => { + const tsv = await toTsv(sheets(f)); + if (tsv === null) return; + const xml = lstsvToDdiXml(tsv, { prodDate: '2020-01-01', - settings: settings[0] ?? {}, + onWarning: quiet, }); + expect(model(ddiToXlsform(xml))).toEqual(model(lstsvToXlsform(tsv))); + // And the TSV's own model, straight from the codebook. expect(canonical(instrumentFromDdi(xml))).toEqual( - canonical(instrumentFromXlsform(survey, choices, settings)), + canonical( + instrumentFromLstsv(parseLstsv(tsv), { + expressions: true, + onWarning: quiet, + }), + ), ); }), - { numRuns: 200 }, + { numRuns: LS_RUNS }, ); }); }); diff --git a/tests/ts/contract/koboRealExports.test.ts b/tests/ts/contract/koboRealExports.test.ts index ea40b3c..3428497 100644 --- a/tests/ts/contract/koboRealExports.test.ts +++ b/tests/ts/contract/koboRealExports.test.ts @@ -137,7 +137,8 @@ describe.each(surveys)('Kobo real exports → DDI data: %s', (name) => { if (isKoboMeta(key) || key === '__version__') continue; const q = key.split('/').pop() as string; const v = variables.find((x) => x.name === q); - if (!v) { + if (!v || v.row !== undefined) { + // A metadata row is described (cdl:row), it has no column. expect(METADATA_TYPES, `${i}:${key}`).toContain(typeOf(q)); } else if (v.type === 'select_multiple') { const picked = String(value).split(' '); diff --git a/tests/ts/contract/lstsv2ddiRoundtrip.test.ts b/tests/ts/contract/lstsv2ddiRoundtrip.test.ts index 3c013ff..5b7a24e 100644 --- a/tests/ts/contract/lstsv2ddiRoundtrip.test.ts +++ b/tests/ts/contract/lstsv2ddiRoundtrip.test.ts @@ -7,7 +7,8 @@ * - grids (emitted as a LimeSurvey array F, reversed back to a grid varGrp), * - external vocabularies (via the `cdlvocab-` cssclass hint), and * - the semi-open `_other` pattern (rebuilt from the native `other=Y` flag + - * the per-language `convention:other` label). + * the per-language `convention:other` label), + * up to what a TSV doesn't hold ({@link withoutLimeSurveyLosses}). */ import * as fs from 'node:fs'; import * as path from 'node:path'; @@ -40,10 +41,46 @@ function scrub(xml: string): string { * `section` (#152); the XLSForm had none. */ function withoutWrapperGroup(xml: string): string { - return xml.replace( - /\n\s*]*>[\s\S]*?<\/varGrp>/, - '', - ); + return xml + .replace( + /\n\s*]*>[\s\S]*?<\/varGrp>/, + '', + ) + .replace( + /\n\s*[^<]*<\/notes>/, + '', + ) + .replace(/(]*>in=)Questions /g, '$1 ') + .replace(/\s*<\/dataDscr>/, ''); +} + +/** + * What a TSV doesn't hold (`lstsv2xlsform/README.md`, #160): the list names + * (`cdl:list`), the form's name for its language (`cdl:setting` + * default_language), whether an other pair was authored or added + * (`cdl:or_other`, and so an authored other label, `cdl:other_label`). And a TSV always has a language, so its codebook + * declares one (`codeBook/@xml:lang`) and writes the universe prose in it + * where the XLSForm's has none. + */ +function withoutLimeSurveyLosses(xml: string): string { + const out = xml + .replace( + /\n\s*]*>[^<]*<\/notes>/g, + '', + ) + .replace( + /\n\s*[^<]*<\/notes>/, + '', + ); + return out; +} + +/** The universe prose and language of a codebook that declares none. */ +function asUndeclared(xml: string, committed: string): string { + if (/]*xml:lang=/.test(committed)) return xml; + return xml + .replace(/(]*) xml:lang="[^"]*"/, '$1') + .replace(/[^<]*<\/universe>/g, ''); } interface Case { @@ -78,6 +115,8 @@ describe('lstsv2ddi round-trip', () => { const committed = fs.readFileSync(path.join(dir, 'ddi.xml'), 'utf-8'); // xlsform2ddi's snapshots use the folder name as the study title. const observed = lstsvToDdiXml(tsv, { assetName: id }); - expect(withoutWrapperGroup(scrub(observed))).toBe(scrub(committed)); + const normalize = (xml: string) => + asUndeclared(withoutLimeSurveyLosses(scrub(xml)), committed); + expect(normalize(withoutWrapperGroup(observed))).toBe(normalize(committed)); }); }); diff --git a/tests/ts/contract/lstsvDdiRoundtrip.test.ts b/tests/ts/contract/lstsvDdiRoundtrip.test.ts new file mode 100644 index 0000000..c8b2d60 --- /dev/null +++ b/tests/ts/contract/lstsvDdiRoundtrip.test.ts @@ -0,0 +1,100 @@ +/** + * The DDI round trips through LimeSurvey (#160), on every fixture form + * (`roundtripCases.ts`), compared on the model (`canonicalInstrument.ts`): + * + * - A LimeSurvey TSV → DDI → Instrument is the TSV's own Instrument: the + * codebook loses nothing a TSV holds. + * - XLSForm → LimeSurvey → DDI → XLSForm is XLSForm → LimeSurvey → XLSForm: + * going through DDI adds no loss to LimeSurvey's. + * - XLSForm → DDI → XLSForm → LimeSurvey is XLSForm → LimeSurvey, to the + * byte: the XLSForm DDI gives back converts as the original does. + * + * `lstsv2ddiRoundtrip.test.ts` compares the codebooks of both paths as text. + */ +import { describe, test, expect } from 'vitest'; + +import { XLSFormToTSVConverter } from '../../../src/pipelines/xlsform2lstsv/index.js'; +import { lstsvToXlsform } from '../../../src/pipelines/lstsv2xlsform/index.js'; +import { lstsvToDdiXml } from '../../../src/pipelines/lstsv2ddi/index.js'; +import { buildDdiXml } from '../../../src/pipelines/xlsform2ddi/index.js'; +import { ddiToXlsform } from '../../../src/pipelines/ddi2xlsform/index.js'; +import { instrumentFromDdi } from '../../../src/instrument/fromDdi.js'; +import { instrumentFromLstsv } from '../../../src/instrument/fromLstsv.js'; +import { instrumentFromXlsform } from '../../../src/instrument/fromXlsform.js'; +import { parseLstsv } from '../../../src/lstsv/parser.js'; +import type { XlsformOutput } from '../../../src/xlsform/fromInstrument.js'; +import type { + ChoiceRow, + SettingsRow, + SurveyRow, +} from '../../../src/xlsform/types.js'; +import { canonical } from './canonicalInstrument.js'; +import { cases, type Case } from './roundtripCases.js'; + +const CASES = cases(); +const WITH_TSV = CASES.filter((c) => c.tsv); +const quiet = () => {}; + +/** The form's LimeSurvey TSV, or `null` for a form LimeSurvey can't hold. */ +async function toTsv(sheets: { + survey: unknown[]; + choices: unknown[]; + settings: unknown[]; +}): Promise { + try { + return await new XLSFormToTSVConverter().convert( + sheets.survey as SurveyRow[], + sheets.choices as ChoiceRow[], + sheets.settings as SettingsRow[], + ); + } catch { + return null; + } +} + +const fromSheets = (s: XlsformOutput) => + canonical(instrumentFromXlsform(s.survey, s.choices, s.settings)); + +function codebook(c: Pick): string { + return buildDdiXml(c.survey, c.choices, { + prodDate: '2020-01-01', + settings: c.settings[0] ?? {}, + }); +} + +describe('LimeSurvey TSV → DDI → Instrument', () => { + test('there are TSVs', () => { + expect(WITH_TSV.length).toBeGreaterThan(20); + }); + + test.each(WITH_TSV)('$name', (c) => { + const tsv = c.tsv!; + const own = instrumentFromLstsv(parseLstsv(tsv), { + expressions: true, + onWarning: quiet, + }); + const back = instrumentFromDdi( + lstsvToDdiXml(tsv, { prodDate: '2020-01-01', onWarning: quiet }), + ); + expect(canonical(back)).toEqual(canonical(own)); + }); +}); + +describe('XLSForm → LimeSurvey → DDI → XLSForm', () => { + test.each(CASES)('$name', async (c) => { + const tsv = await toTsv(c); + if (tsv === null) return; + const viaDdi = ddiToXlsform( + lstsvToDdiXml(tsv, { prodDate: '2020-01-01', onWarning: quiet }), + ); + expect(fromSheets(viaDdi)).toEqual(fromSheets(lstsvToXlsform(tsv))); + }); +}); + +describe('XLSForm → DDI → XLSForm → LimeSurvey', () => { + test.each(CASES)('$name', async (c) => { + const tsv = await toTsv(c); + if (tsv === null) return; + expect(await toTsv(ddiToXlsform(codebook(c)))).toBe(tsv); + }); +}); diff --git a/tests/ts/contract/roundtripCases.ts b/tests/ts/contract/roundtripCases.ts new file mode 100644 index 0000000..a967217 --- /dev/null +++ b/tests/ts/contract/roundtripCases.ts @@ -0,0 +1,92 @@ +/** + * The forms the round-trip tests run on (#154, #160): every whole-survey + * fixture and every registry entity's example, as XLSForm sheets, with its + * committed LimeSurvey TSV where there is one. + */ +import * as fs from 'node:fs'; +import * as path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +import { XLSLoader } from '../../../src/xlsform/loader.js'; + +type Row = Record; + +export const ROOT = path.resolve( + path.dirname(fileURLToPath(import.meta.url)), + '../../..', +); + +export interface Case { + name: string; + survey: Row[]; + choices: Row[]; + settings: Row[]; + /** The committed `tsv.tsv`, if any. */ + tsv?: string; +} + +function readTsv(file: string): string | undefined { + return fs.existsSync(file) ? fs.readFileSync(file, 'utf-8') : undefined; +} + +function load( + name: string, + dir: string, + json: string, + tsv: string, +): Case | null { + const committed = readTsv(tsv); + if (fs.existsSync(json)) { + const f = JSON.parse(fs.readFileSync(json, 'utf-8')) as Record< + string, + Row[] + >; + return { + name, + survey: f.survey ?? [], + choices: f.choices ?? [], + settings: f.settings ?? [], + tsv: committed, + }; + } + const xlsx = path.join(dir, 'xlsform.xlsx'); + if (!fs.existsSync(xlsx)) return null; + const data = XLSLoader.parseXLSData(fs.readFileSync(xlsx), { + skipValidation: true, + }); + return { + name, + survey: data.surveyData as unknown as Row[], + choices: data.choicesData as unknown as Row[], + settings: data.settingsData as unknown as Row[], + tsv: committed, + }; +} + +/** Every fixture form: surveys first, then `entity:`. */ +export function cases(): Case[] { + const out: Case[] = []; + const surveys = path.join(ROOT, 'tests/fixtures/surveys'); + for (const name of fs.readdirSync(surveys).sort()) { + const dir = path.join(surveys, name); + const c = load( + name, + dir, + path.join(dir, 'xlsform.json'), + path.join(dir, 'tsv.tsv'), + ); + if (c) out.push(c); + } + const entities = path.join(ROOT, 'registry/entities'); + for (const name of fs.readdirSync(entities).sort()) { + const dir = path.join(entities, name); + const c = load( + `entity:${name}`, + dir, + path.join(dir, 'fixtures/xlsform.json'), + path.join(dir, 'tsv.tsv'), + ); + if (c) out.push(c); + } + return out; +} diff --git a/tests/ts/unit/ddi/fields.test.ts b/tests/ts/unit/ddi/fields.test.ts index a52fdaa..b2da947 100644 --- a/tests/ts/unit/ddi/fields.test.ts +++ b/tests/ts/unit/ddi/fields.test.ts @@ -53,7 +53,7 @@ describe('standard DDI', () => { expect(varXml(xml, 'x')).not.toContain('dcml'); }); - test("a range's start and end are its valrng, the rest a note", () => { + test("a range's start and end are its valrng, and as authored a note", () => { const xml = ddi([ { type: 'range', @@ -65,7 +65,10 @@ describe('standard DDI', () => { ]); const r = varXml(xml, 'r'); expect(r).toContain(''); - expect(r).toContain('step=5'); + // A bound equal to the default is told apart from none (#160). + expect(r).toContain( + 'start=0 end=100 step=5', + ); // The registry's defaults, as pyxform's. expect(varXml(xml, 'plain')).toContain(''); expect(varXml(xml, 'plain')).not.toContain('cdl:parameters'); diff --git a/tests/ts/unit/ddi/fromDdi.test.ts b/tests/ts/unit/ddi/fromDdi.test.ts index 4f5f5b2..15f2ad4 100644 --- a/tests/ts/unit/ddi/fromDdi.test.ts +++ b/tests/ts/unit/ddi/fromDdi.test.ts @@ -122,15 +122,29 @@ describe('DDI formtransform did not write', () => { }); describe('choice lists', () => { - test('identical category sets share one list', () => { - const cat = `yYes`; - const v = (n: string, s: number) => - `${n}${cat}`; - const { instrument } = read(v('a', 1) + v('b', 2)); - expect(instrument.body.map((q) => (q as QuestionItem).list)).toEqual([ - 'a', - 'a', - ]); - expect(Object.keys(instrument.lists)).toEqual(['a']); + const cat = `yYes`; + const v = (n: string, extra = '') => + `${n}${cat}${extra}`; + const lists = (xml: string) => { + const { instrument } = read(xml); + return { + used: instrument.body.map((q) => (q as QuestionItem).list), + names: Object.keys(instrument.lists), + }; + }; + + test('in DDI not from CDL, identical category sets share one list', () => { + expect(lists(v('a') + v('b'))).toEqual({ + used: ['a', 'a'], + names: ['a'], + }); + }); + + test("in a CDL codebook, a list is cdl:list's, else the question's (#160)", () => { + const named = 'yn'; + expect(lists(v('a', named) + v('b', named) + v('c'))).toEqual({ + used: ['yn', 'yn', 'c'], + names: ['yn', 'c'], + }); }); }); diff --git a/tests/ts/unit/ddi/multilingual.test.ts b/tests/ts/unit/ddi/multilingual.test.ts index 28345d7..b26321d 100644 --- a/tests/ts/unit/ddi/multilingual.test.ts +++ b/tests/ts/unit/ddi/multilingual.test.ts @@ -197,13 +197,13 @@ describe('lstsv → multilingual DDI', () => { expect(xml).toContain('Occupation?'); }); - test('a single-language survey declares no language', () => { + test('a single-language survey declares its language (#160)', () => { const tsv = [ header, line('S', '', 'language', '1', 'de', '', '', ''), line('G', '1', 'Gruppe', '1', '', '', 'de', ''), line('Q', 'S', 'job', '1', 'Beruf?', '', 'de', 'N'), ].join('\n'); - expect(rootLang(lstsvToDdiXml(tsv, OPTS))).toBeUndefined(); + expect(rootLang(lstsvToDdiXml(tsv, OPTS))).toBe('de'); }); }); diff --git a/tests/ts/unit/ddi/structure.test.ts b/tests/ts/unit/ddi/structure.test.ts index 378dcdf..fe34156 100644 --- a/tests/ts/unit/ddi/structure.test.ts +++ b/tests/ts/unit/ddi/structure.test.ts @@ -65,8 +65,15 @@ describe('groups', () => { ); }); - test('a group without a variable under it has no varGrp', () => { - expect(xml).not.toContain('VG_empty'); + test('a group of notes only is a varGrp, placed by cdl:position (#160)', () => { + expect(varGrp(xml, 'VG_empty')).toContain('type="section"'); + expect(varGrp(xml, 'VG_empty')).not.toMatch(/ var=/); + expect(xml).toContain( + 'in= after=outer', + ); + expect(xml).toContain( + 'in=empty after=', + ); }); test('top-level questions are in no group', () => { diff --git a/tests/ts/unit/instrument/fromLstsv.test.ts b/tests/ts/unit/instrument/fromLstsv.test.ts index 32bdff5..b0a8d45 100644 --- a/tests/ts/unit/instrument/fromLstsv.test.ts +++ b/tests/ts/unit/instrument/fromLstsv.test.ts @@ -68,10 +68,9 @@ describe('instrumentFromLstsv', () => { ), ); expect(ins.languages).toEqual(['de', 'en']); - expect(ins.settings).toMatchObject({ - default_language: 'de', - style: 'pages', - }); + expect(ins.defaultLanguage).toBe('de'); + // The bare tag is the model's defaultLanguage, not an authored setting. + expect(ins.settings).toEqual({ style: 'pages' }); const [g] = ins.body as GroupItem[]; expect(g.label).toEqual({ de: 'Seite', en: 'Page' }); expect(g.children[0]).toMatchObject({ diff --git a/tests/ts/unit/pipelines/xlsform2ddi/codebook.test.ts b/tests/ts/unit/pipelines/xlsform2ddi/codebook.test.ts index 899962e..e4069e0 100644 --- a/tests/ts/unit/pipelines/xlsform2ddi/codebook.test.ts +++ b/tests/ts/unit/pipelines/xlsform2ddi/codebook.test.ts @@ -265,7 +265,10 @@ describe('notes', () => { { type: 'integer', name: 'q', label: 'Q' }, ]); expect(xml).toContain('Context'); - expect(xml).not.toContain(']*>[^<]*<\/notes>/g)).toEqual([ + 'n', + ]); }); }); diff --git a/tests/validation/test_registry_schematron_conformance.py b/tests/validation/test_registry_schematron_conformance.py index 2c86d08..4af2e21 100644 --- a/tests/validation/test_registry_schematron_conformance.py +++ b/tests/validation/test_registry_schematron_conformance.py @@ -336,6 +336,17 @@ def mutate(xml: str) -> str | None: _dup_first("notes"), "more than one note of one cdl: type in one language", ), + # The study's notes about a row (#160): a subject, one per type, subject and language. + ( + "type:note", + _sub1(r'(type="cdl:position") subject="hinweis"', r"\1"), + "A cdl:position note needs a subject", + ), + ( + "type:note", + _sub1(r'(]*>[^<]*)', r"\1\1"), + "more than one note of one cdl: type about one subject", + ), ]