From e768378cadae0296bc1f30d8ac10d485c1257726 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 10:55:56 -0500 Subject: [PATCH 01/41] chore: open 0.7.0-alpha.1 on develop --- plugins/dw/.claude-plugin/plugin.json | 2 +- pyproject.toml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/plugins/dw/.claude-plugin/plugin.json b/plugins/dw/.claude-plugin/plugin.json index e245189d..ad9715b9 100644 --- a/plugins/dw/.claude-plugin/plugin.json +++ b/plugins/dw/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "dw", "description": "Compose MiniMax H3 video, MiniMax Music 3 and LTX-2.5 workflows over a dw MCP server, and cut a multi-episode series from them: which template fits which shape, the hard rules, cost, and how to judge the output. Prompt format comes from the vendors' own guides.", - "version": "0.6.0", + "version": "0.7.0-alpha.1", "author": { "name": "Don Kackman" }, diff --git a/pyproject.toml b/pyproject.toml index 74661108..50a01dcc 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -8,7 +8,7 @@ name = "diffusers-workflow" # runtime (static in TOML because importing dw at build time would drag # torch into the build environment). Release tags must match it - see # docs/RELEASING.md -version = "0.6.0" +version = "0.7.0-alpha.1" description = "Declarative workflow engine and web UI for Hugging Face Diffusers" readme = "README.md" requires-python = ">=3.10" From b420f6b535cc3a1d82a5787523d2e7cf848d6325 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 12:14:41 -0500 Subject: [PATCH 02/41] docs(stabilization): Phase 4 plan, stage 4a hot zone Phase 4 staged 4a-4d: carried code, guardrails in dw, seam map and context diet, gate 4 (harness stage C before FREEZE is deleted). Don's rulings 2026-10-01: concat refuses an unrated track, 1,100-line ceiling with a warn band, CLAUDE.md only shrinks after the diet, release decided at the gate; the diet's budget is the root file (<= 150 lines). Co-Authored-By: Claude Opus 5.5 --- docs/stabilization/ROADMAP.md | 2 +- docs/stabilization/hot-zone.txt | 46 ++- .../phase-4-surveys/carried-items.md | 255 ++++++++++++++ docs/stabilization/phase-4.md | 310 ++++++++++++++++++ 4 files changed, 611 insertions(+), 2 deletions(-) create mode 100644 docs/stabilization/phase-4-surveys/carried-items.md create mode 100644 docs/stabilization/phase-4.md diff --git a/docs/stabilization/ROADMAP.md b/docs/stabilization/ROADMAP.md index 2e7c6bc2..6f36d13f 100644 --- a/docs/stabilization/ROADMAP.md +++ b/docs/stabilization/ROADMAP.md @@ -14,7 +14,7 @@ phase works on is what the earlier phases leave behind. | 1 | Metrics v2 first (see below); remove the REPL; one prepare pipeline; one admission service; `dw.run` becomes a thin client of `dw.serve` | Validation sees the definition the run sees; the server admits a request once (one `Workflow`, one expansion); every entry point reaches the worker through the server; ratchets re-baselined | [phase-1.md](phase-1.md) | done 2026-09-28 (`stabilization-gate-1`) | | 2 | Seams in place: `references.py`, validation context + check registry, shared task rules, step cache, typed worker protocol | `validation_errors` is a registry loop; no prefix literals outside `references.py` | [phase-2.md](phase-2.md) (staged: 2a-2d) | done 2026-09-30 (`stabilization-gate-2`) | | 3 | Structural moves: `app.py` routers + services, `LibraryPath`, split `result.py` / `pipeline.py`, one media + dsp module | No module over 1,000 lines, no function over 150; suite and lem smoke green | [phase-3.md](phase-3.md) (staged: 3a-3e) | done 2026-10-01 (`stabilization-gate-3`) | -| 4 | Context diet (CLAUDE.md <= 250 lines total) and guardrails installed | Guardrails live in dw CI and the harness; freeze lifted | written at gate 3 | - | +| 4 | Carried fixes; guardrails installed; context diet (root CLAUDE.md <= 150 lines, every CLAUDE.md triaged; was "<= 250 total", Don 2026-10-01) | Guardrails live in dw CI and the harness; freeze lifted | [phase-4.md](phase-4.md) (staged: 4a-4d) | 4a in progress | ## Metrics diff --git a/docs/stabilization/hot-zone.txt b/docs/stabilization/hot-zone.txt index 73b72948..c15fd484 100644 --- a/docs/stabilization/hot-zone.txt +++ b/docs/stabilization/hot-zone.txt @@ -2,6 +2,50 @@ # The harness implementer must not change these; a field bug whose fix # needs one is labelled `stabilization` and handed to owner:don. # One path (or directory ending in /) per line; '#' starts a comment. -# Stage 3e merged (2026-10-01); Phase 3 complete pending gate 3. +# Stage 4a (carried code) live from 2026-10-01; plan: docs/stabilization/phase-4.md +dw/workflow.py +dw/workflow_run.py +dw/validation.py +dw/library.py +dw/realize.py +dw/worker.py +dw/kernel_availability.py +dw/workflow_schema.json +dw/references.py +dw/tasks/audio_utils.py +dw/tasks/joins.py +dw/tasks/concat_videos.py +dw/tasks/dissolve_videos.py +dw/server/admission.py +dw/server/routes/library.py +dw/server/routes/jobs.py +dw/server/routes/assets.py +dw/server/catalog.py +dw/server/enhancers.py +dw/server/exports.py +dw/server/jobs.py +dw/server/outputs.py +dw/adapter_compatibility.py +dw/argument_media.py +dw/arguments.py +dw/assets.py +dw/content_types.py +dw/elision.py +dw/for_each.py +dw/locations.py +dw/plan.py +dw/probe_paths.py +dw/prompts.py +dw/reference_limits.py +dw/reference_names.py +dw/runs.py +dw/shots.py +dw/step_value_checks.py +dw/subfolders.py +dw/variable_constraints.py +dw/video_extensions.py +dw/vram_estimate.py +dw/server/catalog_shape.py scripts/arch_metrics.py +scripts/surface_snapshot.py docs/stabilization/ diff --git a/docs/stabilization/phase-4-surveys/carried-items.md b/docs/stabilization/phase-4-surveys/carried-items.md new file mode 100644 index 00000000..df989e83 --- /dev/null +++ b/docs/stabilization/phase-4-surveys/carried-items.md @@ -0,0 +1,255 @@ +# Phase 4 survey: dw-stabilization worktree (read-only) + +Paths relative to /Users/don/src/dkackman/dw-stabilization. community_pipelines ignored. Line numbers are as of HEAD e768378c. + +## Helper semantics to keep in mind +- `is_ref(kind, value)`: isinstance str + startswith; accepts a str or a tuple. +- `ref_name(kind, value)`: returns the remainder or None; single prefix only; does NOT `.strip()`. Many hand sites do `.removeprefix(X).strip()`, so the replacement is `ref_name(...)` + `.strip()` (or add a stripping variant). Call this out in the plan. +- `make_ref(kind, name)`: single prefix only. +- Tuples: SUBSTITUTED=(VARIABLE,ITEM), UNRESOLVED=(VARIABLE,ITEM,PREVIOUS_RESULT,GATHER), DEFERRED=UNRESOLVED+(ASSET,OUTPUT,PROMPT,CONSTANT,BUILTIN). `references.py` has no stripping helper and no CONSTRAINT tuple. + +## Part A. Prefix spellings outside dw/references.py + +### A3. Module-level alias constants (16 lines, 13 modules). Each is the root cause of several A1/A2 sites. +| file:line | code | replacement | +|---|---|---| +| dw/assets.py:30 | `ASSET_PREFIX = references.ASSET` | delete; importers (reference_names, server/exports, server/outputs, server/admission) use `references.ASSET` / helpers | +| dw/runs.py:57 | `OUTPUT_PREFIX = references.OUTPUT` | delete; importers: reference_names, realize, server/exports, server/outputs | +| dw/prompts.py:28 | `PROMPT_PREFIX = references.PROMPT` | delete; importers: arguments, realize, server/catalog, server/admission, reference_names | +| dw/realize.py:42 | `BUILTIN_PREFIX = references.BUILTIN` | delete; importer: plan.py:29 | +| dw/realize.py:43 | `VARIABLE_PREFIX = references.VARIABLE` | delete; importers: plan.py:30, server/jobs.py:46 | +| dw/prompts.py:33-40 | `RESERVED_TEXT_PREFIXES = (PREVIOUS_RESULT, VARIABLE, CONSTANT, ASSET, OUTPUT, PROMPT_PREFIX)` | no helper fits: a bespoke 6-tuple (it is not DEFERRED, which also holds ITEM/GATHER/BUILTIN). Either add a named tuple to references.py (e.g. RESERVED_TEXT) or keep and use is_ref(RESERVED_TEXT_PREFIXES, text). Also joined into messages at prompts.py:159 and server/routes/library.py:573 | +| dw/shots.py:42 | `SHOT_REFERENCE_PREFIX = f"{ref_prefixes.PREVIOUS_RESULT}shot@"` | no helper fits: a composite prefix plus the literal "shot@" (cf. MEMBER_SEPARATOR "@" in references.py). Use `is_ref(PREVIOUS_RESULT, r)` and `ref_name(...).startswith("shot@")`, or `make_ref(PREVIOUS_RESULT, "shot@")` | +| dw/reference_names.py:35 | `_UNRESOLVED_PREFIXES = references.UNRESOLVED` | `references.UNRESOLVED` | +| dw/video_extensions.py:30 | `_UNRESOLVED_PREFIXES = references.UNRESOLVED` | same | +| dw/probe_paths.py:21 | `UNRESOLVED_PREFIXES = references.UNRESOLVED` (exported in `__all__` line 66) | same; the public export means checking importers (grep found none outside the module) | +| dw/reference_limits.py:49 | `_UNRESOLVED_PREFIXES = references.UNRESOLVED` | same | +| dw/adapter_compatibility.py:58 | `_UNRESOLVED_PREFIXES = references.UNRESOLVED` | same | +| dw/subfolders.py:31 | `_UNRESOLVED_PREFIXES = references.SUBSTITUTED` | `references.SUBSTITUTED` | +| dw/content_types.py:73 | `_UNRESOLVED_PREFIXES = references.SUBSTITUTED` | same | +| dw/kernel_availability.py:36 | `_UNRESOLVED_PREFIXES = references.SUBSTITUTED` | same | +| dw/step_value_checks.py:49 | `_UNRESOLVED_PREFIXES = references.SUBSTITUTED` | same | + +The name `_UNRESOLVED_PREFIXES` is misleading: five of the eight modules bind it to SUBSTITUTED, three to UNRESOLVED. +No module rebuilds `(VARIABLE, ITEM)` or UNRESOLVED inline; the duplication is through aliases only. + +Inline ad-hoc tuples passed to is_ref (legit use of the helper, but none matches a named tuple; 5 sites): +- dw/argument_media.py:138 and :294 `references.is_ref((references.PREVIOUS_RESULT, references.VARIABLE), x)` +- dw/arguments.py:935 same pair +- dw/video_extensions.py:46 `(CONSTANT, PROMPT)` +- dw/server/catalog_shape.py:171 `(VARIABLE, GATHER)` +A named tuple (e.g. LAZY_MEDIA = (PREVIOUS_RESULT, VARIABLE)) would remove the repetition; optional. + +### A1 + A2. startswith / removeprefix / slice / replace by hand +"->" is the replacement. `S` = SUBSTITUTED, `U` = UNRESOLVED. +Many are startswith + removeprefix/slice pairs of one logical site; pairs are marked (pair). + +startswith sites (31 lines): +- dw/plan.py:186 `if isinstance(reference, str) and reference.startswith(VARIABLE_PREFIX):` + :187 `name = reference.removeprefix(VARIABLE_PREFIX)` (pair) -> `name = ref_name(VARIABLE, reference)`; `if name is not None` +- dw/plan.py:230 `if isinstance(written_seed, str) and written_seed.startswith(VARIABLE_PREFIX):` + :231 removeprefix (pair) -> ref_name +- dw/plan.py:652 `if isinstance(path, str) and not path.startswith(BUILTIN_PREFIX):` -> `not is_ref(BUILTIN, path)` (note: `path` must be str: is_ref already checks) +- dw/realize.py:104 `...definition_seed.startswith(VARIABLE_PREFIX):` + :105 `seed_variable = definition_seed.removeprefix(VARIABLE_PREFIX)` (pair) -> ref_name +- dw/realize.py:152 `if string.startswith(PROMPT_PREFIX):` -> `is_ref(PROMPT, string)` +- dw/realize.py:218 `...not path.startswith(BUILTIN_PREFIX):` -> `not is_ref(BUILTIN, path)` +- dw/realize.py:121 `if value.startswith(prefix) and value not in found:` inside `strings_with_prefix(tree, prefix)`: generic over a parameter, not a spelling; -> `is_ref(prefix, value)` is possible. Not counted as a site. +- dw/server/jobs.py:310 `if not isinstance(seed, str) or not seed.startswith(VARIABLE_PREFIX):` + :312 `name = seed.removeprefix(VARIABLE_PREFIX)` (pair) -> ref_name +- dw/server/admission.py:277 `elif leaf.startswith(PROMPT_PREFIX):` -> `is_ref(PROMPT, leaf)` +- dw/server/catalog.py:50 `if value.startswith(PROMPT_PREFIX):` + :51 `references.add(value.removeprefix(PROMPT_PREFIX).strip())` (pair) -> `ref_name(PROMPT, value)` + `.strip()` (note: the local variable is named `references`, shadowing the module name there) +- dw/vram_estimate.py:61 `...value.startswith(ref_prefixes.VARIABLE):` + :62 `variables.get(value[len(ref_prefixes.VARIABLE) :])` (pair) -> ref_name +- dw/vram_estimate.py:256 `if value.startswith(ref_prefixes.VARIABLE):` + :257 `yield value[len(ref_prefixes.VARIABLE) :]` (pair) -> ref_name +- dw/vram_estimate.py:298 `...entries.startswith(ref_prefixes.VARIABLE):` + :299 `name = entries[len(ref_prefixes.VARIABLE) :]` (pair) -> ref_name +- dw/adapter_compatibility.py:183 `...not reference.startswith(references.VARIABLE):` + :185 `variable = reference.removeprefix(references.VARIABLE)` (pair) -> ref_name +- dw/adapter_compatibility.py:143-145 `weight_name.startswith(_UNRESOLVED_PREFIXES)` -> `is_ref(U, weight_name)` +- dw/variable_constraints.py:423 `...reference.startswith(references.CONSTRAINT):` + :425 `constraints[reference[len(references.CONSTRAINT) :]]` (pair) -> ref_name(CONSTRAINT, ...) +- dw/variable_constraints.py:453 `and value.startswith(references.CONSTRAINT)` + :454 `and value[len(references.CONSTRAINT) :] not in constraints` (pair) -> ref_name +- dw/shots.py:255 `if isinstance(reference, str) and reference.startswith(SHOT_REFERENCE_PREFIX):` + :257 `member = reference[len(ref_prefixes.PREVIOUS_RESULT) :]` (pair) -> no helper fits for the composite prefix (see A3); the slice -> `ref_name(PREVIOUS_RESULT, reference)` +- dw/shots.py:259-261 `elif isinstance(reference, str) and reference.startswith(ref_prefixes.PREVIOUS_RESULT):` + :263 `step = reference[len(ref_prefixes.PREVIOUS_RESULT) :]` (pair) -> ref_name +- dw/workflow_run.py:178-180 `candidate.startswith(references.PREVIOUS_RESULT)` + :181 `field["entry"] = candidate[len(references.PREVIOUS_RESULT) :]` (pair) -> ref_name +- dw/locations.py:533 `return value.startswith(references.DEFERRED)` -> `is_ref(references.DEFERRED, value)` (is_ref also guards non-str; check the function's guard) +- dw/prompts.py:156 `if text.startswith(RESERVED_TEXT_PREFIXES):` and dw/server/routes/library.py:569 `if str(request.prompt.get("text", "")).startswith(RESERVED_TEXT_PREFIXES):` -> `is_ref(RESERVED_TEXT_PREFIXES, ...)`; the tuple itself: no helper fits (see A3) +- Alias-tuple startswith, all -> `is_ref(S/U, value)`: + dw/subfolders.py:89, dw/content_types.py:146, dw/kernel_availability.py:145 (all S, `isinstance` guard sits beside it), dw/step_value_checks.py:74 (S), dw/video_extensions.py:44 (U), dw/probe_paths.py:41 (U), dw/reference_limits.py:102 (U), dw/reference_names.py:100 (U) +- dw/reference_names.py:97 `if not value.startswith(prefix):` + :99 `rest = value[len(prefix) :].strip()` (pair; loops over `_KINDS` of (OUTPUT_PREFIX, ASSET_PREFIX, PROMPT_PREFIX)) -> `ref_name(prefix, value)`; the `_KINDS` tuple entries hold alias constants (lines 48-50) -> `references.OUTPUT/ASSET/PROMPT` + +removeprefix / slice / replace sites not paired above: +- dw/assets.py:113 `name = validate_asset_reference(reference.removeprefix(ASSET_PREFIX).strip())` -> ref_name(ASSET, reference).strip() +- dw/runs.py:236 `...reference.removeprefix(OUTPUT_PREFIX).strip()` -> ref_name(OUTPUT, ...) +- dw/prompts.py:97 `...reference.removeprefix(PROMPT_PREFIX).strip()` -> ref_name(PROMPT, ...) +- dw/arguments.py:315 `name = validate_constant_name(reference.removeprefix(references.CONSTANT).strip())` -> ref_name(CONSTANT, ...) +- dw/realize.py:173 `name = reference.removeprefix(PROMPT_PREFIX).strip()` ; :189 `name = reference.removeprefix(OUTPUT_PREFIX).strip()` +- dw/reference_names.py:44 `return reference.removeprefix(OUTPUT_PREFIX).strip()` +- dw/server/admission.py:268 `name = leaf.removeprefix(ASSET_PREFIX).strip()` +- dw/server/exports.py:284 `reference.removeprefix(ASSET_PREFIX).strip()` ; :309 `name = reference.removeprefix(OUTPUT_PREFIX).strip()` +- dw/server/outputs.py:173 `return name.removeprefix(OUTPUT_PREFIX).strip()` ; :245 `reference.removeprefix(ASSET_PREFIX).strip(),` +- dw/workflow.py:450 and :921 `builtin_name = path.replace(references.BUILTIN, "")` -> `ref_name(BUILTIN, path)`. This one is a latent bug: `replace` strips the substring anywhere, not just the prefix; `ref_name` strips only the prefix. Both sit in the duplicated builtin-resolution blocks (see B3). +No `split(":", 1)` or `partition(":")` strips a reference prefix. (`vram_estimate.py:161` `.split(":")[0]` is a device string; `validation.py:189/736` partitions are warning-separator / dotted paths; `media_frames.py:191` and `server/routes/media.py:298` use the `frame:` prefix, which is not a reference prefix and has no constant.) +No `previous_results.py:165` hit: `name[len(result_name)+1:]` strips a step name, not a prefix. `for_each.py:283` `reference[len(group):]` likewise. + +### A4. Building a reference by hand +- dw/elision.py:210 `reference = references.VARIABLE + name` -> make_ref(VARIABLE, name) +- dw/for_each.py:203 `return references.PREVIOUS_RESULT + _rewrite_reference(` -> make_ref(PREVIOUS_RESULT, ...) +- dw/for_each.py:263 `references.PREVIOUS_RESULT + member_name(group, key)` -> make_ref +- dw/for_each.py:292 (inside f-string, message) `f"'{references.GATHER}{group}' for every member's result, ...` -> `make_ref(GATHER, group)` +- dw/realize.py:201 `return f"{OUTPUT_PREFIX}{relative}"` -> make_ref(OUTPUT, relative) +- dw/server/routes/assets.py:363 `logger.info(f"Kept output {body.name} as asset:{asset_name}")` and :433 `logger.info(f"Deleted asset:{relative} ({path})")` -> literal text "asset:" in log lines (a spelled prefix, not a reference value) -> `make_ref(ASSET, ...)`. The rest of routes/assets.py already uses `make_ref(ASSET, ...)` (161, 219, 319, 365, 438); docstrings at 85/176/281/380... only mention "asset:" in prose. +- dw/server/enhancers.py:29 `"workflow": "builtin:h3_context_ir.json",` -> a whole reference literal in the PRESETS dict. -> `make_ref(BUILTIN, "h3_context_ir.json")` (constants can't be built at class level without an import; module already can import references) +- dw/arguments.py:327-328 message `f"'{references.CONSTANT}' reads a value, ..."` -> uses the constant for text only; no change needed (not a reference construction). +Prose-only strings that mention a prefix in error text (no handling; leave): assets.py:124, runs.py:253, locations.py:155/554 (no actual prefix), serve.py:89, step_value_checks.py:299, tasks/assess.py:222, tasks/task.py:552, type_helpers.py:209, validation.py:445, variables.py:389, video_extensions.py:58, workflow.py:457/929, dw_mcp/assets.py:157/185, dw_mcp/server.py:58, result.py:482. + +### A5. Regexes +- Python: none embed a prefix (grep of `re.compile/match/search/sub` in dw, dw_mcp: nothing prefix-related; `security.py:652` CONSTANT_NAME_PATTERN is a name rule). +- JSON schema (not Python, outside the metric): dw/workflow_schema.json:103, :201, :211, :531 `"pattern": "^variable:"` and :400 `"pattern": "^constraint:[a-zA-Z_][a-zA-Z0-9_-]*$"`. 5 spellings. No helper (a schema is data); a test that pins the schema patterns to `references.VARIABLE`/`CONSTRAINT` is the way to guard them. + +### Totals (lines, outside dw/references.py) +- Form 1 startswith: 31 lines (+1 generic parameter site, not counted) +- Form 2 removeprefix 18 + slice 9 (incl. reference_names:99) + replace 2 = 29 lines +- Form 3 aliases: 16 lines (+5 inline ad-hoc tuples, optional) +- Form 4 building: 3 concat/f-string in engine (elision, for_each x2) + realize:201 + for_each:292 + routes/assets x2 + enhancers x1 = 8 +- Form 5 regex: 0 in Python, 5 in JSON schema +- Sum: 84 lines in dw/ (about 70 logical sites once the ~15 startswith+strip/slice pairs are merged), spread over 31 Python modules: adapter_compatibility, arguments, assets, content_types, elision, for_each, kernel_availability, locations, plan, probe_paths, prompts, realize, reference_limits, reference_names, runs, shots, step_value_checks, subfolders, variable_constraints, video_extensions, vram_estimate, workflow, workflow_run, server/admission, server/catalog, server/enhancers, server/exports, server/jobs, server/outputs, server/routes/assets, server/routes/library. dw_mcp has none. Tests were not surveyed. +- Biggest clusters: realize.py 9, plan.py 5, vram_estimate.py 6, for_each.py 3, shots.py 5, variable_constraints.py 4, server/exports 2, server/outputs 2. +- Importers of alias constants (the aliases cannot be deleted until each is changed): ASSET_PREFIX: reference_names, server/exports, server/outputs, server/admission; OUTPUT_PREFIX: reference_names, realize, server/exports, server/outputs; PROMPT_PREFIX: arguments (already uses is_ref at :289), reference_names, realize, server/catalog, server/admission; BUILTIN_PREFIX/VARIABLE_PREFIX: plan, server/jobs; RESERVED_TEXT_PREFIXES: server/routes/library. + +## scripts/arch_metrics.py: how `prefix_literals` is counted +- `REFERENCE_PREFIXES` (lines 28-42) is a frozenset of the 10 prefix strings; `PREFIX_OWNERS = {"dw/references.py"}` (line 43). +- `measure()` (lines 185-210): for every `.py` under packages dw and dw_mcp (excludes community_pipelines etc.), it `ast.walk`s the parse tree and increments `prefix_literals` for each `ast.Constant` whose `value in REFERENCE_PREFIXES` (exact string equality) in a file not in PREFIX_OWNERS (lines 202-207). Only `ast.Constant` nodes are examined, nothing else. +- What it catches: a literal that is exactly `"asset:"`, `"variable:"`, ... in any position: `startswith("asset:")`, `"asset:" + name`, a bare alias assignment, and an f-string whose constant fragment is exactly the prefix (`f"asset:{x}"` yields a Constant fragment "asset:"). +- What it misses: + - Form 1, 2, 3: every site uses a constant name, never a literal, so all 31 + 29 + 16 lines above are invisible to it by design. The metric cannot see "spelling" handling, only literals. + - Form 4: concat/f-string via constants (elision, for_each, realize) are invisible; f-strings where the fragment is not exactly the prefix are missed: routes/assets.py:363 (fragment " as asset:"), :433 (fragment "Deleted asset:"); a whole reference literal `"builtin:h3_context_ir.json"` (enhancers.py:29) is missed, as is any `"asset:foo.png"` literal (anything with a suffix) or a message embedding a prefix. `"frame:"` is not in the set. + - Form 5: regexes with the prefix inside a longer pattern are missed; and the JSON schema (not .py) is out of scope entirely. + - tests/ and scripts/ are not scanned (packages are only dw and dw_mcp). +- Suggested metric upgrade: add a second counter `prefix_handling` that flags (a) `ast.Call` with attr startswith/removeprefix/replace/partition whose argument resolves to a REFERENCE constant/alias/tuple name, (b) `ast.Subscript` slices `[len():]`, (c) `ast.BinOp(+)` / `JoinedStr` with a refs const operand, (d) module-level `Assign` whose value is `references.`; plus `prefix_literals` extended to a substring/startswith check (`any(p in s for p in REFERENCE_PREFIXES)` on non-docstring Constants, with an allowlist for prose). My AST scan script (not saved) found the 65 call/slice/concat/fstring hits plus 16 aliases that make up the totals above. + +## Part B. Carried items + +### B1. `for_each._copy_leaf` (dw/for_each.py:217-227), callers :214, :238, :251 +Code: +``` +def _copy_leaf(value): + try: + return copy.deepcopy(value) + except Exception: + return value +``` +Correction to the premise: it does NOT share leaves. Inside a member (`member is not None`, `_rewrite` tail at :214, `_item` at :238 and :251) every non-str, non-container leaf, and every `item:` field value, is `copy.deepcopy`'d, with a fall-back to sharing when deepcopy raises. Outside a member the leaf is returned as is (comment :205-213). The ROADMAP (phase 2d follow-up, docs/stabilization/ROADMAP.md:371) states it correctly: "`for_each._copy_leaf` and the step-cache snapshot still copy media leaves". The leaves are the ones `realize_args` has already turned into loaded images, decoded frame lists, torch tensors, `from_file()` dataclasses (see the comment at :205-213 and deep_equal in step_cache.py:313+), so each for_each member deep-copies the template's media: N members hold N copies of the media, plus another copy per member's snapshot (B2). Risk of the current behaviour: memory multiplication and time, and a silent fall-back to sharing for anything not deep-copyable (so behaviour differs by leaf type). The fix would change copying of leaves to sharing, i.e. use `copy_containers` semantics (containers rebuilt, leaves shared; `_rewrite` already rebuilds dict/list containers itself so only the scalar/media leaf and any tuple remain). Risk of that fix: any task that mutates a leaf in place (the workflow.py:195-203 comment names `conform_artifact` stamping fps onto what `select` handed back) would now edit the object every member shares; tuples as leaves currently get a deep copy of their contents. Import note: `copy_containers` lives in dw/step_cache.py, which imports torch/numpy/PIL; `for_each` currently imports only `references`, so a direct import adds a heavy edge and possibly a cycle (step_cache is imported by realize, workflow_run, pipeline). Moving `copy_containers` to `dw/references.py`-style leaf module (or a tiny `dw/copying.py`) avoids it. + +### B2. Step-cache snapshot and `copy_containers` +- Defined: `copy_containers` at dw/step_cache.py:285-310 (dicts, lists, tuples exact types rebuilt, everything else, including subclasses, shared). Test pins it: tests/test_step_cache.py:859. +- Current users: dw/realize.py:86, :92 (definition + variables copy, import at :37); dw/workflow_run.py:290 (`recorded_variables = copy_containers(variables)`, import :52); dw/pipeline_processors/pipeline.py:119 (import :42). +- The snapshot: `cache_lookup` in dw/workflow_run.py:395-436: + ``` + step_data_snapshot = None + if is_cacheable: + try: + step_data_snapshot = copy.deepcopy(step_data) # line 398 + if parent_saves_this: step_data_snapshot["__saved_by_parent__"] = True + ... step_data_snapshot["__borrowed_pipelines__"] = borrowed + except Exception as ex: + logger.debug(... "not copyable ... skipping the step cache for it"); is_cacheable = False + ``` + The snapshot is then the lookup key (`step_cache.get(workflow_id, step_data_snapshot, step_seed, ...)` at :421-425) and, via `cache_entry` (:712) and `step_cache.put` (:727-730), the stored key. Matching uses `deep_equal` (step_cache.py:313+, with `a is b` short-circuit and value comparison of Image/tensor/array). +- Current behaviour: every cacheable step deep-copies `step_data`, which at that point already holds the realized media arguments, so each step run copies all its images/frames/tensors once for the lookup and keeps that copy alive in the cache entry. A leaf that cannot be copied disables caching for that step. +- Fix: `copy_containers(step_data)` in place of `copy.deepcopy` gives an independent key structure (the later in-place edits the comment at :380-387 worries about are entry replacements/key assignments, which the container copy isolates) with shared media leaves; the `except` branch ("not copyable") becomes unreachable for leaves, so remove it or narrow it. Risks: (1) a leaf mutated in place after the snapshot silently changes the key (and `deep_equal`'s `a is b` fast path would then report a false hit); (2) the cache entry now pins the same media objects the live run holds (retention semantics change; LRU byte accounting in step_cache.py ~:530-600 would need a look); (3) `copy_containers` shares subclass containers (OrderedDict etc.), so a definition with those is not isolated. Also the sub-workflow path `workflow.py:619` (`copy.deepcopy(value) if name in declared`) and workflow_run.py:450/497 `copy.deepcopy(workflow.workflow_definition)` are the same family (not named in the carry list). + +### B3. Sub-workflow path resolution: every place +Core resolver: `resolve_sub_workflow(path, base_dir, confine_to)` at dw/library.py:616-717 (raises SubWorkflowNotFound(path, tried), library.py:593/717). +Places that resolve a path: +1. `Workflow.resolve_sub_workflow_path(path)` dw/workflow.py:439-473. Code: builtin branch (:449-467) with `path.replace(references.BUILTIN, "")` and name checks, `confine_to = builtin_root()`, existence check raising SubWorkflowNotFound; else `if confine_to is None and not os.path.isabs(path): confine_to = catalog_root_dir(self.file_spec)` then `resolve_sub_workflow(path, os.path.dirname(self.file_spec), confine_to)` (:469) and `validate_workflow_path(resolved, confine_to)`. Returns (path, root). +2. `Workflow._sub_workflow_action(step_definition, default_seed)` dw/workflow.py:908-998 (reached from `create_step_action` at :776, dispatch `if "workflow" in step_definition: return self._sub_workflow_action(...)` at :811). It re-implements the SAME resolution inline (:919-968): builtin branch at :920-941 duplicating the name check and message (differs: `confine_to = os.path.join(os.path.dirname(os.path.abspath(__file__)), "workflows")` instead of `builtin_root()`, no isfile check), `resolve_sub_workflow(path, os.path.dirname(self.file_spec), confine_to)` at :960, `validate_workflow_path` + `workflow_from_file(validated_path, self.output_dir, confine_to)`. It does not call resolve_sub_workflow_path, although that method's docstring says it is "the same resolution create_step_action does". The two copies already diverge in builtin confinement and in the replace-based name extraction (A2). +3. `Workflow.open_sub_workflow(path)` dw/workflow.py:475-481: calls `self.resolve_sub_workflow_path(path)` (:481) then `workflow_from_file(...)`. Used by validation. +4. `validation.sub_workflow_errors` dw/validation.py:755-807: per sub-workflow step it resolves twice: `workflow.resolve_sub_workflow_path(path)` at :776 (to get `resolved` for the cycle check and a precise error), then `workflow.open_sub_workflow(path)` at :793, which resolves again at workflow.py:481. Message ownership is the reason: the first call yields resolution errors verbatim; the second wraps any failure as "Sub-workflow '': ...". Fix is to have `open_sub_workflow` take/return the resolved path (or split into `resolve` + `open_resolved(resolved, root)`) so one resolution feeds both. +5. `validation.sub_workflow_argument_warnings` dw/validation.py:809-838: `workflow.open_sub_workflow(reference["path"])` at :821 for every step with arguments; a third resolution per step, swallowing the error (":824 an unresolvable path is an error, reported by sub_workflow_errors"). Registered check "sub_workflow_warnings" (validation.py:545). +6. `read_sub_workflow` dw/realize.py:232-249 (called by `_digest`, `_record_sub_workflows` scan at :218-226): `resolve_sub_workflow(path, base_dir or ".", workflow_dir)` at :244 + `validate_workflow_path(candidate, root.root if root else None)`. Its docstring says it resolves "the way `Workflow.create_step_action` resolves it", yet it skips the `catalog_root_dir` fallback (:468 in workflow.py) and builtins (:218 skips them). +7. dw/server/routes/jobs.py:557-560 (observed-cost lookup for a composed child): `resolve_sub_workflow(path, base_dir or ".", candidate.workflow_dir)`, errors swallowed; again no catalog_root_dir fallback or builtin handling. +Also: `create_step_action` on a warm/cached path can reach only #2. `dw/library.py:128` documents the confinement root. So seven sites, four distinct implementations of the preamble (workflow.py x2, realize x1, routes/jobs x1) around the one real resolver. + +### B4. `Workflow._run_dir` +- Declared as class attributes: dw/workflow.py:167 `_run_dir = None`, :172 `_run_dir_inherited = False`, :177 `_run_version = None` (comment :162-176). +- Set: dw/workflow_run.py:557 `workflow._run_dir = None` (flat layout, inside `_claim_run_dir`) and :566 `workflow._run_dir, workflow._run_version = claim_run_dir(...)`; both only when `not workflow._run_dir_inherited` (open_run :584-585). Sub-workflow: dw/workflow.py:993-994 `workflow._run_dir = self._run_dir; workflow._run_dir_inherited = self._run_dir is not None` (child is a fresh Workflow each time, inside `_sub_workflow_action`). +- Read: dw/workflow.py:289-290 (`step_output_dir`: `if self._run_dir: return self._run_dir`), :714 (via `workflow_run.owns_run_dir`, which reads workflow_run.py:145-148); dw/workflow_run.py:148, :217, :222 (`_run_version`), :252, :254, :573-574, :591, :602, :620/622, :970 (`owns_run_dir`) and the `finally` in `run()` (workflow.py:714). +- Not reset at the top of `run` (workflow.py:623-662): `run()` resets `pipeline_ownership` (:655), `manifest` (:656), `_elided_steps` (:659), and creates a new `RunRecord`, but never `_run_dir`, `_run_version` or `_run_dir_inherited`. A persistent worker reuses the Workflow across jobs (the comment at :650-654 says so). Consequence: if `prepare_run` raises (workflow.py:662) before `open_run` reaches `_claim_run_dir`, the `finally` at :713-715 sees the previous run's `_run_dir` and `owns_run_dir` true, so `write_run_manifest` rewrites the previous run's manifest.json with the failed run's record; `step_output_dir` also keeps returning the old directory before the claim. In the flat layout `_claim_run_dir` sets None itself, so only the nested layout is exposed. Fix: at the top of `run()` (before `prepare_run`) reset `self._run_dir = None; self._run_version = None` when `not self._run_dir_inherited` (a child's values are set by the parent and must stay). +- Test touchpoints: tests/test_runs.py:628 sets `workflow._run_dir` directly; tests/test_events.py:536/551/566 and tests/test_shots.py:1127/1133 read it after a run. + +### B5. kernels-hub "Cannot find a build variant" +- Not produced by dw: it comes from the third-party `kernels` package (venv: kernels/resolver.py:103, :132, :221; the per-variant lines come from `variants_trace_str(trace)` in kernels/variants.py:615-628, which sorts via `_sort_variants` but, per the roadmap note, the lines come out in set-driven order across processes). +- dw only embeds the text: dw/kernel_availability.py:125-126 + ``` + except Exception as e: + raise _KernelFault(f"'{value}' {KERNEL_FAULT_MARKER}: {e}") from e + ``` + so the validation error the author sees carries the variant lines in whatever order the hub library printed. +- The only normaliser lives in scripts/surface_snapshot.py:212-229, `stable_message(value)`: if "Cannot find a build variant" is in the string it sorts the lines that start with "torch" in place; recursion over list/dict (used at :256 on warning-check output). tests/test_kernel_availability.py:31/103 uses a stub FileNotFoundError("no build variant ..."). +- Fix: sort the `torch...` lines in `_fault_for_name` (or a small `stable_variant_message(e)` in kernel_availability.py) so the message is deterministic for users and agents, then drop `stable_message` from the snapshot script (or keep it as belt and braces). The line detection must keep to the "torch" prefix convention or the `variant_str: reason` shape. Checking the installed kernels version is needed because the format could change. + +### B6. `argument_template` +- Schema: dw/workflow_schema.json:106-109 + `"argument_template": { "description": "Engine-injected: the arguments a parent workflow passed to this one when it ran it as a sub-workflow. Written by create_step_action from the step's 'arguments' block, not authored - a workflow file carrying one is read, but a sub-workflow step is how they are meant to be supplied.", "type": "object" }` +- Current behaviour: `create_step_action` writes nothing into the definition. It dispatches to `_sub_workflow_action` (workflow.py:811), which sets `workflow._handed_arguments = workflow_reference.get("arguments", {})` at :977 (attribute declared :209 with the rationale at :195-209: "kept here rather than written into workflow_definition, where every validate() and run() of the child deep-copied them again"). `Workflow.argument_template` (workflow.py:229-233) is now a property returning `self._handed_arguments` when not None, else `self.workflow_definition.get("argument_template", {})` (so an authored `argument_template` in a file is still read, as the description's last clause says). Readers: dw/step.py:95 `get_iterations(step_action.argument_template, ...)` (step_action here is the pipeline/task action's own property: pipeline_processors/pipeline.py:179, tasks/task.py:849, not Workflow's) and dw/pipeline_ownership.py:276, pipeline.py:409/443 mutate the action's template, not the workflow's. docs/RELEASING.md:88 notes "a composed child's steps no longer carry argument_template". +- Fix: reword the description to say the handed arguments are held on the child workflow object by `_sub_workflow_action` at run time and never written into the definition; an authored value in a file is still read as the fallback. Mind that the schema file is part of the `workflow-schema` MCP surface (scripts/surface_snapshot.py:196 `workflow_schema()` snapshots it, so the description change shows in the surface diff). + +### B7. `gain_audio` region end vs `slice_audio` (after #557) +`slice_audio` (dw/tasks/audio_utils.py:142-248) delegates to `slice_region` (dw/task_domains.py:406-441), which rounds a frame-addressed end once: +``` +start = frames_to_samples(start_frame or 0, fps, sample_rate) +if num_frames is not None: + end = frames_to_samples((start_frame or 0) + num_frames, fps, sample_rate) + return start, end - start +``` +`gain_audio` (audio_utils.py:251-377), the frame branch at :332-343, still has the two-halves computation: +``` +elif start_frame is not None or num_frames is not None: + if fps is None: raise ValueError("gain_audio needs 'fps' to address a region in frames") + start = frames_to_samples(start_frame or 0, fps, sample_rate) + length = ( + max(total - start, 0) if num_frames is None + else frames_to_samples(num_frames, fps, sample_rate) # <- rounded separately from start + ) +``` +with `frames_to_samples(frames, fps, sample_rate) = int(round(frames / fps * sample_rate))` (task_domains.py:401). Because `round(a)+round(b) != round(a+b)` in general, `start + length` can land a sample off the exact `round((start_frame+num_frames)/fps*sr)`; slice_audio's end is exact. The seconds branches (:317-325) match slice_region's. Other differences: gain_audio checks `fps is None` while slice_region uses `not fps`; gain_audio has an "everything" branch (`start=0; length=total`, :345-347) where slice_region returns None. Fix: have gain_audio call `slice_region(...)` (it already imports from task_domains; `region is None` plus "no region args at all" distinguishes the whole-track case), keeping its "needs fps" message. The clip at `region_end = max(region_start, min(start + max(length, 0), total))` (:350-351) stays. tests: check tests for gain_audio frame regions (not surveyed). + +### B8. `concat_videos` vs `dissolve_videos` with a track that has no sample rate +Shared function: dw/tasks/joins.py:70-133 `reconcile_sample_rates(command, videos, names, waveforms, sample_rate=None, skip_unrated=True)`. +``` +rates = [video.sample_rate for video, waveform in zip(videos, waveforms) + if waveform is not None and (video.sample_rate or not skip_unrated)] +sample_rate = sample_rate or (max(set(rates)) if rates else None) +... +return [ waveform if waveform is None + or (skip_unrated and not video.sample_rate) # <- unrated track passed through unchanged + or video.sample_rate == sample_rate + else resample_waveform(waveform, video.sample_rate, sample_rate) + for video, waveform in zip(videos, waveforms)], sample_rate +``` +- concat_videos: dw/tasks/concat_videos.py `_prepare_inputs` (:~215-228) calls `reconcile_sample_rates("concat_videos", videos, names, _input_waveforms(videos), sample_rate)` with the default `skip_unrated=True`. A track whose AudioVideo carries `sample_rate` None/0 is excluded from `rates` (so it cannot influence the target) and is returned unresampled, i.e. joined as though already at the target rate, which plays at the wrong speed/pitch (the class of fault `as_track` and #140 refuse elsewhere; see dw/dsp.py:223-230). `audio_native_rate = getattr(video, "sample_rate", None)` (:~167) is also None for it. +- dissolve_videos: dw/tasks/dissolve_videos.py:297-304 `_dissolve_audio` passes `skip_unrated=False`. The unrated track enters `rates` as `None`/0 ... then `resample_waveform(waveform, None, target)` runs, and `dw/dsp.py:226-230` raises `ValueError("resample_waveform needs a sample_rate above zero, got None")`, so it refuses. Pinned by tests/test_dissolve_videos.py:94-106 ("refuses the unrated track"). +- The docstring at joins.py:82-86 calls skip_unrated "concat_videos' long-standing rule". Fix options: make concat refuse like dissolve (flip to `skip_unrated=False`; behaviour change: surface note, and check the templates/tests that rely on pipeline-reported None rates, tests/test_result.py:109 comment mentions a pipeline reporting no sample rate), or resample using a fall-back rate with a warning; then delete the `skip_unrated` parameter. + +### B9. The 8 one-line `Workflow` delegators (dw/workflow.py) +Registry: dw/validation.py ERROR_CHECKS/WARNING_CHECKS; warning checks are run in the registry by `admit()` directly, so the methods are not on the production path except where noted. +| method (def line) | body | delegates to | +|---|---|---| +| validation_context :498-502 | `return validation.workflow_context(self, arguments, composing, ceiling_index)` | dw/validation.py:607 `workflow_context` | +| sub_workflow_warnings :484-496 | `return validation.run_warning_check(self, "sub_workflow_warnings", arguments)` | validation.py:707 `run_warning_check` (check body `sub_workflow_argument_warnings` :809) | +| adapter_warnings :520-529 | `return validation.run_warning_check(self, "adapter_warnings", arguments)` | run_warning_check (check -> adapter_compatibility.adapter_warnings, validation.py:536) | +| inherited_vram_warnings :531-544 | `if not index: return []` then `validation.run_warning_check(self, "inherited_vram_warnings", arguments, ceiling_index=index)` (not strictly one line: has a guard) | run_warning_check (-> vram_inheritance.inherited_vram_warnings, validation.py:572-577) | +| slice_past_end_warnings :546-554 | `return validation.run_warning_check(self, "slice_past_end_warnings", arguments)` | run_warning_check (-> slice_preflight.slice_past_end_warnings, :554) | +| shot_span_warnings :556-565 | `return validation.run_warning_check(self, "shot_span_warnings", arguments)` | run_warning_check (-> shot_span_preflight.shot_span_warnings, :562) | +| null_variable_argument_warnings :567-583 | `return validation.run_warning_check(self, "null_variable_argument_warnings", arguments)` | run_warning_check (check at validation.py:541) | +| cache_hits :604-607 | `return workflow_run.cache_hits(self, arguments)` | dw/workflow_run.py:439 `cache_hits` | + +Callers, production: +- validation_context: dw/validation.py:662 (`workflow_errors`, `context = workflow.validation_context(arguments, composing)`), :712 (`run_warning_check`, `workflow.validation_context(arguments, **context_fields)`), dw/server/admission.py:158 (`candidate.validation_context(checked, ceiling_index=ceiling_index)`). +- null_variable_argument_warnings: dw/server/routes/library.py:298 (`warnings += candidate.null_variable_argument_warnings()`). +- cache_hits: dw/worker.py:513 (`cached = workflow.cache_hits(command.get("arguments") or {})`). +- sub_workflow_warnings, adapter_warnings, inherited_vram_warnings, slice_past_end_warnings, shot_span_warnings: NO production callers (the registry runs the module functions through run_warning_check/admit). Only tests and scripts/surface_snapshot.py:204-208 `WARNING_CHECKS` tuple (adapter_warnings, slice_past_end_warnings, shot_span_warnings, null_variable_argument_warnings, sub_workflow_warnings) called by name at scripts/surface_snapshot.py:256 `getattr(workflow, name)()`; inherited_vram_warnings and cache_hits are not in that script. +Callers, tests: +- validation_context: tests/test_validation.py:332, :349; tests/test_admission.py:499 (parametrised `(Workflow, "validation_context")` monkeypatch target, so removing the method breaks that test's target and it must change to `validation.workflow_context`). +- sub_workflow_warnings: tests/test_validation.py:301 (WARNING_ORDER list entry, string only), :374; tests/test_workflow.py:1452, :1478, :1521. +- adapter_warnings (method): tests/test_validation.py:370; tests/test_lora_disable.py:124, :131 (`h3_workflow(tmp_path).adapter_warnings(...)`); (module-function uses at test_lora_disable.py:143, test_h3_adapters.py:136-178 are the other function). +- inherited_vram_warnings (method): tests/test_validation.py:377-378 (`workflow.inherited_vram_warnings(None, {"identity": {}})`); tests/test_vram_inheritance.py:125 (`workflow.inherited_vram_warnings(arguments, index)`); tests/test_admission.py:237 (registry swap by name, string); tests/test_vram_inheritance.py:295 (message text). +- slice_past_end_warnings (method): tests/test_validation.py:375; tests/test_slice_preflight.py:186 (`workflow.slice_past_end_warnings()`). Module-function calls elsewhere are not the method. +- shot_span_warnings (method): tests/test_validation.py:376; tests/test_shot_span_preflight.py:172; tests/test_admission.py:220/233/307 are registry-swap strings. +- null_variable_argument_warnings: tests/test_validation.py:371-372. +- cache_hits: tests/test_realize.py:451; tests/test_workflow_step_cache.py:656, :1009, :1019, :1036, :1047, :1056; tests/test_worker_execute.py:287 and :322 (fake workflow classes define their own `cache_hits`, so the worker's call at dw/worker.py:513 must keep an object-with-`cache_hits` seam or those fakes change). +Note for removal planning: replacing the methods with a module call means 3 production changes (validation.py:662/712, admission.py:158, routes/library.py:298, worker.py:513 = 5 call lines) and ~25 test lines; `Workflow.validation_errors` (:504) and `validate` (:585) are further delegators not in this list. diff --git a/docs/stabilization/phase-4.md b/docs/stabilization/phase-4.md new file mode 100644 index 00000000..efa0fd9e --- /dev/null +++ b/docs/stabilization/phase-4.md @@ -0,0 +1,310 @@ +# Phase 4 Implementation Plan: carried fixes, guardrails, context diet + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Leave dw in a state that holds without a freeze. The quirks the freeze parked are fixed, every architecture rule that can be a check is one (in dw's CI and in the harness), agent context is a map rather than a manual, and then the freeze lifts. + +**Architecture:** Four stages, each merged to `develop` on its own with its own hot zone, as in Phases 2 and 3. A stage's detailed tasks are written when the previous stage merges, on the code that stage left. Stage 4a is detailed below. The order is: +- code first (4a), so the guardrails measure the code they will guard; +- guardrails (4b) before the diet, so CI is already enforcing when the diet re-baselines `claude_md_lines`; +- the seam map before the diet (both 4c), because the diet's pointers go to it; +- the gate (4d) last: the harness's stage C goes live *before* `FREEZE` is deleted. + +**Tech Stack:** Python 3.10+, pytest, GitHub Actions; metrics from `scripts/arch_metrics.py` and `scripts/arch_report.py`. + +**Spec:** [ROADMAP.md](ROADMAP.md) (Phase 4 row: "Context diet (CLAUDE.md <= 250 lines total) and guardrails installed"; gate: "Guardrails live in dw CI and the harness; freeze lifted"), [ASSESSMENT.md](ASSESSMENT.md) ("Guardrail principles"), and ROADMAP.md, Gate 3, "Carried to Phase 4". Don's rulings of 2026-10-01 bind the whole phase: "in the tension between freezing behavior and stabilizing the codebase, err on the side of stabilization", and the 1,000-line rule has hysteresis ("I don't want a refactor for +2 lines of SLOC"). + +## Where Phase 4 starts (gate 3, `ee5c4944`; develop `e768378c`, 0.7.0-alpha.1) + +- **Ratchets** (`baseline.json`): modules 165, modules over 1,000 lines 0, functions over 150 lines 0, prefix literals 0, test patch targets 284, CLAUDE.md lines 957, duplicate blocks 5, complex functions 7, import cycles 0, modules in cycles 0. +- **Nothing in dw's CI runs the ratchet.** `ci.yml`'s `backend` job installs the dev extra (which carries grimp, networkx, pylint, pygount) and runs ruff and pytest only. `scripts/preflight.sh` doesn't run it either. The only enforcement is the harness's stage B hand-off gate. +- **Modules closest to 1,000 lines:** `workflow.py` 996, `workflow_run.py` 972, `arguments.py` 954, `tasks/task.py` 947, `plan.py` 926. +- **CLAUDE.md, 957 lines in four files:** root 674 (of which "Critical Gotchas" is 395), `ui/CLAUDE.md` 154, `dw_mcp/CLAUDE.md` 95, `dw/server/CLAUDE.md` 34. `.github/copilot-instructions.md` (114) is a sibling that the metric does not count. `tests/test_plugin_skills.py::test_a_skill_is_enumerated_where_the_plugin_describes_itself` requires every plugin skill's name in backticks in the root CLAUDE.md. +- **`Workflow`'s 8 one-line delegators** (gate 3 LCOM4: 9 components). Production callers: `cache_hits` (`worker.py`, `workflow.py`), `validation_context` (`validation.py`, `server/admission.py`), `null_variable_argument_warnings` (`server/routes/library.py`). The other five are called only from tests. + +## Stages + +| Stage | Scope | Done when | +| --- | --- | --- | +| 4a | **Carried code.** The behaviour items the freeze parked (each with a failing-first test and a 0.7.0 release note); `Workflow`'s 8 delegators replaced by their module functions; the remaining prefix spellings moved onto `references.py`'s helpers, with a metric that counts them. | Every "Carried to Phase 4" code item in ROADMAP.md, Gate 3, is fixed or ruled out in Decisions; `Workflow` LCOM4 is 1; the new prefix metric is 0 and ratcheted | +| 4b | **Guardrails in dw.** The size rule gets its warn band; `arch_metrics.py --check` runs in CI's `backend` job and in `preflight.sh`; the re-baseline rule is written down where the check prints it. | A PR that raises any ratchet fails CI; a module growing past 1,000 lines warns and does not fail until the ceiling | +| 4c | **Seam map, then the context diet.** `docs/ARCHITECTURE.md`: concept → owning module → the rule, one row each. Then every CLAUDE.md is cut to a map: each paragraph is deleted (the knowledge is already in a doc, a docstring or a test), moved into the owning module's docstring, or moved into the seam map. `.github/copilot-instructions.md` becomes a pointer. | the root CLAUDE.md <= 150 lines, every CLAUDE.md through the triage, `claude_md_lines` re-baselined; every plugin skill still named in the root CLAUDE.md; no paragraph moved to `docs/` verbatim without a ruling | +| 4d | **Gate 4.** The harness stage C prompt (`harness/stage-c-guardrails.md`), which Don runs; then `FREEZE` deleted and `hot-zone.txt` emptied; the metrics report, tag, re-baseline, lem deploy, ROADMAP and ASSESSMENT refreshed. | Stage C is committed in the harness, *then* the freeze is lifted on `develop` | + +## Carried into Phase 4 + +From ROADMAP.md, Gate 3, "Carried to Phase 4", plus 3e's carried list. Each item goes to the stage whose files it touches: +- **4a:** + - `for_each._copy_leaf` sharing leaves, and the step-cache snapshot via `copy_containers`: both **ruled out** in Decisions (4a), measured at gate 4. + - One sub-workflow path resolver for `create_step_action` and `validation.sub_workflow_errors` (which resolves each path twice to keep its messages). + - `_run_dir` reset at the top of `run`. + - The kernels-hub "Cannot find a build variant" message's set order (nondeterministic). + - `workflow_schema.json`'s `argument_template` description (still says `create_step_action` writes it into the definition). + - 3d: `gain_audio`'s frame-based region end rounding (the pre-#557 way, one sample short). + - 3d: `concat_videos` joining a track with no sample rate unresampled (`dissolve_videos` refuses it). + - `Workflow`'s 8 delegators. + - The remaining prefix spellings (`startswith` / `removeprefix` / slicing / alias constants, the `routes/assets.py` f-strings) and a metric that counts them. +- **4b:** the module-size warn band. + +## Global Constraints (all stages) + +- The freeze holds until 4d deletes `FREEZE`: no net-new features or functionality. Phase 4 may change behaviour where Don's 2026-10-01 ruling covers it (a parked quirk or duplicate path removed), and every such change gets a line under `### 0.7.0` in `docs/RELEASING.md` in the same commit. +- `scripts/arch_metrics.py --check docs/stabilization/baseline.json` passes at the end of every task. Phase 4 names no `modules` rise; a new module is a regression unless a stage's Decisions name it. +- Every new test fails before its fix. No count-pinning tests. Add no string `patch("dw...")` targets. A task that moves a patched name retargets the patch in the same commit. +- No compatibility shims: a moved or deleted name is imported from its home by every caller, tests included (Phase 3, Decisions). +- Behaviour-preserving tasks prove preservation with the existing suite: it passes unchanged, apart from the import paths, call spellings and patch targets the task changes. +- Tests: `venv/bin/python -m pytest -q -x -p no:cacheprovider` from the worktree root, plus `ruff check` and `ruff format --check` on `dw dw_mcp tests`. The worktree's `venv` is shared with the main checkout, so never run `pip install -e .` from the worktree. +- Never use `git stash`, in any form, including `git stash list`. +- Filesystem access keeps going through a `dw/security.py` validator, and a validator that moves is re-modelled in `.github/codeql/` in the same commit. Each stage's merge checks the CodeQL run on `develop` and treats a new alert as a finding. +- No lem deploy before gate 4. + +## Decisions (rulings, 2026-10-01; each open to Don's correction) + +- **The diet's budget is the root CLAUDE.md, which every session loads (Don, 2026-10-01: "economize tokens but not compromise quality").** + - The root file is 51 KB, about 12-13k tokens, read by every agent session and subagent working in the repo. The three sub-files total about 18 KB and load only when work touches their directory. + - Root: **<= 150 lines**, a hard target. Sub-files: the same per-paragraph triage (below), with no line quota. A rule that keeps a UI or MCP agent from drifting stays where that agent reads it. + - The ROADMAP row's "<= 250 total" becomes "root <= 150; total whatever the triage leaves, then ratcheted". Expected total: about 250-300. + - Cost if wrong: the sub-files keep a few hundred tokens they could have shed, paid only in sessions that work there. +- **The diet deletes before it moves.** Every CLAUDE.md paragraph goes one of three ways, decided per paragraph in a table committed with 4c: *deleted* (a doc, a docstring or a test already says it; the table names which), *docstring* (it is a rule about one module, so it moves to that module's docstring) or *seam map* (it is a rule spanning modules, so it becomes one row in `docs/ARCHITECTURE.md`, a sentence, not the paragraph). Nothing moves to `docs/*.md` verbatim. + - Why: ASSESSMENT's "docs churn exceeds code churn". Moving 700 lines of prose from CLAUDE.md into docs takes it out of agent context and keeps all the churn. + - Cost if wrong: something an agent needed falls out of context. The seam map is the net: it names every owner, so an agent finds the module, and the module's docstring holds the rule. +- **After the diet, `claude_md_lines` only goes down.** No later fix adds a CLAUDE.md line without a Decision; a fix that wants one writes it in the owning module's docstring or the seam map. This is the ratchet as ASSESSMENT intended it ("not more prose in agent context"). +- **The size rule's warn band (Don, 2026-10-01: hysteresis).** + - The ratchet `modules_over_1000_lines` is replaced by `modules_over_size_ceiling` at a ceiling of **1,100 lines**, ratcheted at 0. + - `--check` (and the default output) also prints a non-failing `warning:` line for every module between 1,001 and 1,100 lines, naming it and its length. A warning is not a metric, so it never enters `baseline.json` and the harness's key-by-key comparison never sees it. + - `functions_over_150_lines` stays a cliff. No ruling asked for a band there, and a 150-line function is already four screens. + - Cost if wrong: a module can sit at 1,099 indefinitely. The warning makes that visible on every run; tightening is a one-line change. + - The key rename is safe for stage B's harness ratchet: `regressions()` skips a key missing on either side, so the comparison across the rename sees neither key for one session and both sides of it afterwards. +- **The re-baseline rule.** `baseline.json` may be lowered by any commit that improves a ratchet, and is rewritten at every gate. It is raised only by a commit that names the rise and why in its message, and from 4d on the harness refuses that unless the issue carries `arch-approved` (stage C). The check prints this rule when it fails. +- **Release: open.** Phase 4 changes behaviour (4a) and develop is 0.7.0-alpha.1. Whether gate 4 ships 0.7.0 is Don's call at the gate; 4a collects the notes under `### 0.7.0` either way. + +## Review Focus (all stages) + +1. **A parked fix that changes a message an agent reads.** Every 4a behaviour change is listed under `### 0.7.0` in `docs/RELEASING.md`, and the MCP surface snapshot (`scripts/surface_snapshot.py`) diff at each merge is exactly that list. +2. **A delegator's caller left on the method.** After 4a, `git grep -nE "(workflow|candidate|self|wf|child|Workflow)\.(validation_context|sub_workflow_warnings|adapter_warnings|inherited_vram_warnings|slice_past_end_warnings|shot_span_warnings|null_variable_argument_warnings|cache_hits)\b" dw dw_mcp tests scripts` returns nothing. The receiver is in the pattern because the destination module functions share the names (`workflow_run.cache_hits`, `adapter_compatibility.adapter_warnings`). +3. **The CI ratchet passes on a broken environment.** If `arch_metrics.py` cannot import grimp, the CI step fails rather than skipping (stage B's "fails closed", in dw). +4. **A diet that drops a pinned name.** `test_a_skill_is_enumerated_where_the_plugin_describes_itself` and every other test reading a CLAUDE.md still pass after 4c, without the test being loosened. +5. **The freeze lifted before the harness can hold.** 4d deletes `FREEZE` only after Don confirms stage C is committed in the harness. + +--- + +## Stage 4a: carried code + +Work on branch `stabilization/phase-4a` in the worktree, from `develop` at `e768378c` or later. + +**What exists (survey, 2026-10-01, at `e768378c`; full notes in [phase-4-surveys/carried-items.md](phase-4-surveys/carried-items.md)).** Line numbers are at that commit. + +- **The 8 delegators** (`dw/workflow.py`): + - `validation_context` :498 → `validation.workflow_context`; + - `sub_workflow_warnings` :484, `adapter_warnings` :520, `inherited_vram_warnings` :531 (with an `if not index: return []` guard), `slice_past_end_warnings` :546, `shot_span_warnings` :556 and `null_variable_argument_warnings` :567, each → `validation.run_warning_check(self, "", arguments[, ceiling_index=])`; + - `cache_hits` :604 → `workflow_run.cache_hits`. + - Production callers: `validation.py:662` and `:712`, `server/admission.py:158` (`validation_context`); `server/routes/library.py:298` (`null_variable_argument_warnings`); `worker.py:513` (`cache_hits`). + - The other five are called only by tests and by `scripts/surface_snapshot.py:256` (`getattr(workflow, name)()` over its `WARNING_CHECKS` tuple). + - Test seams on the methods: `tests/test_admission.py:499` monkeypatches `(Workflow, "validation_context")`; the fake workflows at `tests/test_worker_execute.py:287` and `:322` define their own `cache_hits`. +- **Sub-workflow paths are resolved at seven sites with four preambles** around the one resolver `library.resolve_sub_workflow` (`library.py:616`): + - `Workflow.resolve_sub_workflow_path` (`workflow.py:439`): builtin branch with `builtin_root()` and an existence check; otherwise falls back to `catalog_root_dir(self.file_spec)`. + - `Workflow._sub_workflow_action` (`workflow.py:908`, from `create_step_action`): **re-implements it inline** (:919-968). Its builtin root is `os.path.join(dirname(__file__), "workflows")`, it has no existence check, and both copies strip the prefix with `path.replace(references.BUILTIN, "")` (:450, :921), which removes the text anywhere in the path, not only as a prefix. + - `Workflow.open_sub_workflow` (:475) calls `resolve_sub_workflow_path`. + - `validation.sub_workflow_errors` resolves each path twice (:776 directly, :793 through `open_sub_workflow`) to keep its two message forms. `sub_workflow_argument_warnings` (:821) resolves a third time. + - `realize.read_sub_workflow` (`realize.py:232`) and `server/routes/jobs.py:557` call `resolve_sub_workflow` with their own preamble, without the `catalog_root_dir` fallback. +- **`_run_dir`, `_run_version` and `_run_dir_inherited`** are class attributes (`workflow.py:167-177`), set in `workflow_run._claim_run_dir` (:557, :566) and by a parent for its child (`workflow.py:993`). `run()` (:623) resets `pipeline_ownership`, `manifest` and `_elided_steps` but not these. On a worker reusing a `Workflow`, if `prepare_run` raises before the claim, the `finally` (:713) sees the previous run's directory as its own and rewrites that run's `manifest.json` with the failed run's record. +- **The kernels message** comes from the third-party `kernels` package; dw embeds it as `f"'{value}' {KERNEL_FAULT_MARKER}: {e}"` (`kernel_availability.py:126`). Its per-variant lines (each starting `torch`) come out in set order. Only `scripts/surface_snapshot.py`'s `stable_message` sorts them, so an author sees a different order on every process. +- **`workflow_schema.json:107`**, `argument_template`'s description, says "Written by create_step_action from the step's 'arguments' block". Since 3e the handed arguments live on the child as `_handed_arguments` (`workflow.py:977`), and the property reads an authored `argument_template` only as the fallback. The schema is part of the surface snapshot. +- **`gain_audio`'s frame region** (`tasks/audio_utils.py:332-343`) rounds `start` and `length` separately (`frames_to_samples(num_frames, ...)`), so `start + length` can miss `round((start_frame + num_frames) / fps * sr)` by a sample. `slice_audio` goes through `task_domains.slice_region` (:406), which rounds the end once (#557). +- **`reconcile_sample_rates`** (`tasks/joins.py:70`) takes `skip_unrated=True` by default. `concat_videos` uses the default, so a track with no sample rate is left out of the rate choice and joined unresampled, which plays it at the wrong speed and pitch. `dissolve_videos` passes `skip_unrated=False` and fails in `dsp.resample_waveform` ("needs a sample_rate above zero"); `tests/test_dissolve_videos.py:94` pins that. +- **Prefix spellings: about 70 sites (84 lines) in 31 modules, none in `dw_mcp`.** + - Alias constants (16): `ASSET_PREFIX` (`assets.py:30`), `OUTPUT_PREFIX` (`runs.py:57`), `PROMPT_PREFIX` (`prompts.py:28`), `BUILTIN_PREFIX` and `VARIABLE_PREFIX` (`realize.py:42-43`), and `_UNRESOLVED_PREFIXES` in eight modules: five bind it to `SUBSTITUTED` and three to `UNRESOLVED`, under one name. Also `RESERVED_TEXT_PREFIXES` (`prompts.py:33`, a six-tuple no named tuple matches) and `SHOT_REFERENCE_PREFIX` (`shots.py:42`, `previous_result:shot@`). + - `startswith()` 31 lines; `removeprefix` / `[len():]` / `replace` 29 lines, about 12 of them `.removeprefix(X).strip()`. + - Built by hand (8): `references.VARIABLE + name` (`elision.py:210`), `for_each.py:203`, `:263` and `:292`, `realize.py:201`, the log f-strings `routes/assets.py:363` and `:433` (literal `asset:`), and `"builtin:h3_context_ir.json"` (`server/enhancers.py:29`). + - The JSON schema spells `^variable:` four times and `^constraint:` once (:103, :201, :211, :531, :400). + - `prefix_literals` counts only an `ast.Constant` exactly equal to a bare prefix, so it sees none of this. + +### Decisions (4a) + +- **The 8 delegators are deleted; callers call the module function.** Production: `validation.workflow_context(workflow, ...)`, `validation.run_warning_check(workflow, "null_variable_argument_warnings")`, `workflow_run.cache_hits(workflow, arguments)`. Tests call `validation.run_warning_check(workflow, "", ...)`. `surface_snapshot.py` does the same by name. `test_admission.py:499` patches `validation.workflow_context` with `patch.object`. The worker's fakes in `test_worker_execute.py` move to a `patch.object(workflow_run, "cache_hits", ...)`. + - `inherited_vram_warnings`' empty-index guard moves into the check, if `run_warning_check` does not already return nothing for an empty index. The task checks this first. + - `Workflow.validation_errors` and `validate` stay methods: they are the class's public entry points, not delegators to a check. +- **One sub-workflow resolver.** `resolve_sub_workflow_reference(path, file_spec, confine_to)` in `dw/library.py` owns the whole preamble: the builtin branch on `builtin_root()` with `ref_name(BUILTIN, ...)` and the existence check, then the `catalog_root_dir` fallback, `resolve_sub_workflow` and `validate_workflow_path`. It returns `(path, root)`. + - `Workflow.resolve_sub_workflow_path` becomes a one-line call to it, and `_sub_workflow_action`'s inline copy is deleted in favour of that call. + - `open_sub_workflow(path, resolved=None)` opens an already-resolved path without resolving again. `sub_workflow_errors` resolves once and hands the result on, with both message forms unchanged. + - `realize.read_sub_workflow` and `routes/jobs.py:557` call it too. That gives them the `catalog_root_dir` fallback they lacked, so a catalog sub-workflow a run could open is now also digested and costed. That goes in the release notes. + - Whether a builtin missing from `builtin_root()` now fails at `create_step_action` with `SubWorkflowNotFound` instead of later is checked against the tests; if its message changes, the change goes in the release notes. +- **`run()` resets `_run_dir` and `_run_version` at its top, unless `_run_dir_inherited`.** A child's values come from its parent, which sets them on a fresh `Workflow` just before `run`. +- **dw sorts the kernels variant lines itself.** `stable_message`'s rule moves into `kernel_availability.py` and applies where the fault is raised; `surface_snapshot.py` drops its copy. A message with no `Cannot find a build variant` text passes through untouched, so a change in the `kernels` package's format degrades to today's behaviour, not to an error. +- **`argument_template`'s description is reworded** to what the code does: the handed arguments are held on the child workflow at run time and never written into the definition, and an authored value is read as the fallback. This is a surface snapshot change and a release-note line. +- **`gain_audio`'s region goes through `slice_region`,** keeping its own "needs 'fps'" message and its whole-track branch. A frame-addressed region's end then matches `slice_audio`'s to the sample. +- **`concat_videos` refuses a track with no sample rate, as `dissolve_videos` does,** and `skip_unrated` is deleted from `reconcile_sample_rates`. + - Why: playing a waveform at a rate it was not sampled at changes its speed and pitch. `as_track` and #140 already refuse that everywhere else. + - The refusal names the input (`"concat_videos: '' has audio with no sample rate"`), not the `resample_waveform` text dissolve surfaces today, and dissolve gets the same message. This is a release-note line. + - Where an unrated track comes from: only from a pipeline none of whose components reports a rate. `attach_audio_sample_rate` (`pipeline_processors/pipeline.py:796`) then logs "Pipeline generated audio but no component reports its sample rate - set 'sample_rate' ... or 'audio_sample_rate' ... in the step result". The catalog's audio pipelines (LTX-2's vocoder, H3) report one. The ruling assumes the catalog's three `concat_videos` templates (`assemble-and-score`, `minimax/music-video`, `minimax/dialogue-short`) never feed it an unrated track. Task 4 checks that *first*. If one can, the fix moves upstream (the step's declared rate stamped onto the `AudioVideo` at extraction), and the task stops and reports instead. + - The refusal names the same remedy as the generation-time warning (set the step result's `audio_sample_rate`). + - Cost if wrong: a hand-written workflow over a pipeline that reports no rate stops joining. It was already joining wrong (wrong speed and pitch), with a warning at generation. +- **Ruled out: sharing leaves in `for_each._copy_leaf` and in the step-cache snapshot.** Both carry items assumed a leaf that is safe to share, and the survey found otherwise. + - `_copy_leaf` deep-copies a member's media leaves on purpose. Tasks mutate their inputs in place (`conform_artifact` stamps fps onto what `select` hands back, the comment at `workflow.py:195`), so a shared leaf would carry one member's edit into the next. + - The cache snapshot is the cache's *key*. A shared leaf mutated after the snapshot changes the stored key with it, and `deep_equal`'s `a is b` shortcut then reports a hit for a changed input. + - What staying costs: memory and time per member and per cached step. That is measured, not assumed: the gate 4 lem timing runs a `for_each` over media. If it shows a real cost, sharing becomes its own design with a mutation audit, after the freeze. +- **References are built and read only through `references.py`.** + - New in `references.py`: `RESERVED_TEXT` (the six-tuple `prompts.py` owns today) and `LAZY_MEDIA = (PREVIOUS_RESULT, VARIABLE)` (the pair spelled inline three times). `SHOT_REFERENCE_PREFIX` stays in `shots.py` as `make_ref(PREVIOUS_RESULT, "shot" + MEMBER_SEPARATOR)`: it is a shots concept built from the owned parts. + - Every alias constant is deleted, and every caller uses `references.` or a helper. `_UNRESOLVED_PREFIXES` goes away with them, so no module names a tuple after something it isn't. + - `x.removeprefix(K).strip()` becomes `ref_name(K, x).strip()`. `ref_name` gets no strip variant: whitespace hygiene is the caller's rule. Where a site relied on `removeprefix` returning an unprefixed value unchanged, the task checks its guard and keeps the semantics. `ref_name` returns `None` there. + - The JSON schema's patterns stay (a schema is data). A new test pins each `pattern` that begins `^variable:` or `^constraint:` to `references.VARIABLE` / `references.CONSTRAINT`. +- **The metric: `prefix_handling`, a second ratchet beside `prefix_literals`.** Outside `PREFIX_OWNERS`, in `dw` and `dw_mcp`, it counts: + - (a) a call to `startswith`, `removeprefix`, `removesuffix`, `replace`, `split` or `partition` whose argument is a `references` constant or tuple (as `references.X`, an imported name `X`, or a name bound at module level to either); + - (b) a slice whose start is `len()`; + - (c) `+` or an f-string with one of them as an operand; + - (d) a module-level assignment whose value is one of them; + - (e) a string constant that *starts with* a prefix (`"builtin:h3_context_ir.json"`), or an f-string fragment that *ends with* one (`" as asset:"`, the fragment before a name is spliced in). Prose that names a prefix with no name after it (`"'asset:' reads ..."`) is not counted. + - So a message that embeds a *reference* builds it with `make_ref`, error and log text included: `f"Kept output {name} as {make_ref(ASSET, asset_name)}"`. That is a ruling, not a side effect. Any f-string message that rule (e) catches is migrated in Task 6, not explained away. + - It lands counting today's sites, failing first against a fixture tree in `tests/test_arch_metrics.py`. The baseline takes that count, and the migration task drives it to 0 and re-baselines. + +### Review Focus (4a) + +1. **A failed run on a reused `Workflow` leaves the previous run's manifest alone.** Test: run once into a nested-layout run directory, make `prepare_run` raise on a second run of the same instance, and assert the first `manifest.json` is byte-identical. A child's inherited directory still holds: the existing composed-run tests pass. +2. **A sub-workflow path the run can open, validation can open, and the other way round.** Tests through the one resolver: + - a relative path that resolves only through `catalog_root_dir`; + - `builtin:` with a name containing `builtin:` again (the `replace` bug); + - a missing builtin; + - a path escaping its root. That one is refused with the `validate_workflow_path` message, by every one of the seven call sites. +3. **Message parity where the plan promises it.** `sub_workflow_errors`' two message forms, `gain_audio`'s fps message and the seven warning checks' texts are unchanged; the surface snapshot diff is exactly the `argument_template` description. +4. **The prefix migration changes no answer.** For each `removeprefix` site turned into `ref_name`, an unprefixed input reaches the same result as before; the reviewer checks each guard. A test where the site has none. +5. **The metric counts what it claims and nothing it doesn't.** The fixture tree holds one instance of each form (a)-(e) and one prose mention. The count is exact, and the prose mention is not counted. + +### Before Task 1: the hot zone goes live + +Once Don approves this plan, and before any 4a code changes: +- commit the 4a list below into `docs/stabilization/hot-zone.txt` on `develop`, with the plan docs and ROADMAP.md's Phase 4 row (Plan: `[phase-4.md](phase-4.md) (staged: 4a-4d)`, as the Phase 3 row reads); +- push; +- start the `stabilization/phase-4a` branch from that commit. + +### Task 1: The 8 delegators + +Smallest first: it takes about 50 lines out of `workflow.py` (996), which the later tasks then have room in. + +- [ ] **Step 1:** Check whether `run_warning_check(workflow, "inherited_vram_warnings", arguments, ceiling_index=None)` and `ceiling_index={}` already return `[]`. If not, move the guard into the check's registry entry first, with a test that fails first. +- [ ] **Step 2:** Repoint the five production call lines (Decisions), `scripts/surface_snapshot.py`, and every test caller (survey list: `test_validation.py`, `test_workflow.py`, `test_lora_disable.py`, `test_vram_inheritance.py`, `test_slice_preflight.py`, `test_shot_span_preflight.py`, `test_realize.py`, `test_workflow_step_cache.py`, `test_admission.py`, `test_worker_execute.py`). +- [ ] **Step 3:** Delete the 8 methods. Run `git grep` per Review Focus (all stages) item 2, then the suite, ruff, and `scripts/surface_snapshot.py`; the snapshot must be byte-identical. +- [ ] **Step 4:** Run `scripts/arch_report.py` for LCOM4 on `Workflow`; expect 1. `validate` reads `self.name` and calls `validation_errors`, so both join the core component. `validation_errors` passes `self` on and reads no attribute, but `validate` calling it links them. If it reads more than 1, record each remaining component in the report. Fix it in this task only when it is one more delegator. +- [ ] **Step 5:** Commit. No release note: no surface changes. + +### Task 2: One sub-workflow resolver + +- [ ] **Step 1:** Write the Review Focus 2 tests against `resolve_sub_workflow_reference`, plus one per call site showing it reaches the resolver: `create_step_action`, `open_sub_workflow`, `sub_workflow_errors`, `read_sub_workflow` and the observed-cost lookup. Run them; the resolver tests fail on the missing name, and the `replace` and `catalog_root_dir` cases fail on today's code. +- [ ] **Step 2:** Add `resolve_sub_workflow_reference` to `dw/library.py`. Move every call site onto it. Delete `_sub_workflow_action`'s inline copy, and add `open_sub_workflow`'s `resolved` parameter. +- [ ] **Step 3:** The suite, ruff and the snapshot. Any changed message goes under `### 0.7.0`. +- [ ] **Step 4:** Commit, with the release-note lines. + +### Task 3: Three small fixes: the run directory, the kernels message, the schema description + +- [ ] **Step 1:** Review Focus 1's test, and a kernels test: a fault carrying three `torch...` variant lines in reverse order comes out sorted, and the rest of the message keeps its place. Both fail. +- [ ] **Step 2:** Reset in `run()`; move `stable_message` into `kernel_availability.py` (applied where the fault is raised) and delete it from `surface_snapshot.py`; reword the `argument_template` description. +- [ ] **Step 3:** The suite, ruff and the snapshot. The only snapshot diff is the description. +- [ ] **Step 4:** Commit, with three release-note lines: a failed run no longer rewrites the previous run's manifest; the build-variant lines are sorted; the schema text. + +### Task 4: The two audio rules + +- [ ] **Step 1:** The ruling's premise. For each of the three catalog templates using `concat_videos`, trace each `videos` input back to the step that made it, and confirm that step's pipeline reports a rate (`_component_sample_rate`). Record that in the report. If any can arrive unrated, stop with DONE_WITH_CONCERNS (Decisions (4a)). +- [ ] **Step 2:** Tests, failing first: + - `gain_audio` over a frame region where `round(start) + round(length) != round(end)`. Pick fps and rate so it bites, for example 24 fps at 44,100 Hz with `start_frame=1, num_frames=1`; compute it in the test. + - `concat_videos` over two inputs, one with audio and no sample rate, refused with the input's name. + - `dissolve_videos` with the same message. +- [ ] **Step 3:** `gain_audio` on `slice_region`; `reconcile_sample_rates` loses `skip_unrated` and raises the named refusal; update `test_dissolve_videos.py:94` to the new message. +- [ ] **Step 4:** The suite, ruff and the snapshot (a task description that mentions the rule may change; list it). Commit, with two release-note lines. + +### Task 5: The `prefix_handling` metric + +- [ ] **Step 1:** In `tests/test_arch_metrics.py`, add a fixture tree with one instance of each form (a)-(e), one prose mention, and one use inside `dw/references.py`. Assert `prefix_handling` is exactly 5. Fails: no such key. +- [ ] **Step 2:** Implement it in `scripts/arch_metrics.py` beside `prefix_literals`. Update the docstring's counting rules and ROADMAP.md's Metrics paragraph in two lines. +- [ ] **Step 3:** Run it on the worktree. The count should be close to the survey's (about 84 lines; the metric counts nodes, so pairs count twice). Explain any gap above 10% in the report before baselining. +- [ ] **Step 4:** Re-baseline (`--write`) with `prefix_handling` at today's count; the commit message names the new key and its number. + +### Task 6: The prefix migration + +- [ ] **Step 1:** Add `RESERVED_TEXT` and `LAZY_MEDIA` to `references.py`, and the schema-pattern test (fails only if a pattern drifts, so it is a characterization test; say so in its docstring). +- [ ] **Step 2:** Work module by module, per the survey's list, deleting each alias when its last importer has moved. After each module, run its tests. +- [ ] **Step 3:** For every `removeprefix` site, check its guard (Review Focus 4). Write a test where an unprefixed input reaches it and the semantics could differ. +- [ ] **Step 4:** `prefix_handling` and `prefix_literals` are 0. The suite, ruff and the snapshot are byte-identical, apart from the two log lines' text if their wording changed. Re-baseline both to 0. +- [ ] **Step 5:** Commit. No release note. + +### Task 7: Stage 4a merge + +- [ ] **Step 1: Re-baseline.** Run `venv/bin/python scripts/arch_metrics.py --write docs/stabilization/baseline.json`. The diff against the committed baseline is exactly: `prefix_handling` 0 (new), and anything that went down. `modules` is 165. +- [ ] **Step 2: Docs.** Find every sentence naming a deleted delegator, alias or `skip_unrated` (`git grep -nE "ASSET_PREFIX|OUTPUT_PREFIX|PROMPT_PREFIX|_UNRESOLVED_PREFIXES|skip_unrated|Workflow\.(cache_hits|validation_context|[a-z_]+_warnings)" CLAUDE.md '*/CLAUDE.md' docs/ ':!docs/stabilization/'`) and fix it. CLAUDE.md may only shrink. +- [ ] **Step 3: Hot zone.** Put `hot-zone.txt` back to the two standing entries. +- [ ] **Step 4: Merge.** Merge `stabilization/phase-4a` to `develop` with `--no-ff` and push. Check the CodeQL and CI runs on `develop`. No lem deploy. Report the changed metrics rows and `Workflow`'s LCOM4 to Don. +- [ ] **Step 5: Detail stage 4b** in this file, on the code 4a left, with its hot zone. Cross-check the design with Fable before Task 1 of 4b. + +### Hot zone (4a) + +``` +dw/workflow.py +dw/workflow_run.py +dw/validation.py +dw/library.py +dw/realize.py +dw/worker.py +dw/kernel_availability.py +dw/workflow_schema.json +dw/references.py +dw/tasks/audio_utils.py +dw/tasks/joins.py +dw/tasks/concat_videos.py +dw/tasks/dissolve_videos.py +dw/server/admission.py +dw/server/routes/library.py +dw/server/routes/jobs.py +dw/server/routes/assets.py +dw/server/catalog.py +dw/server/enhancers.py +dw/server/exports.py +dw/server/jobs.py +dw/server/outputs.py +dw/adapter_compatibility.py +dw/argument_media.py +dw/arguments.py +dw/assets.py +dw/content_types.py +dw/elision.py +dw/for_each.py +dw/locations.py +dw/plan.py +dw/probe_paths.py +dw/prompts.py +dw/reference_limits.py +dw/reference_names.py +dw/runs.py +dw/shots.py +dw/step_value_checks.py +dw/subfolders.py +dw/variable_constraints.py +dw/video_extensions.py +dw/vram_estimate.py +dw/server/catalog_shape.py +scripts/arch_metrics.py +scripts/surface_snapshot.py +docs/stabilization/ +``` + +It is a long list because the prefix migration touches a line or two in 31 modules. Most of those edits are one-line import or call changes, so a harness edit to one would be a small merge conflict, not a lost fix. If the harness needs one of these files for a field bug during 4a, Don can drop it from the list. Task 6 then rebases over that fix. + +## Stage 4b: guardrails in dw + +Detailed when 4a merges. + +## Stage 4c: seam map, then the context diet + +Detailed when 4b merges. + +## Stage 4d: gate 4 + +Detailed when 4c merges. + +The stage C prompt (`harness/stage-c-guardrails.md`, same format as A and B) covers the six guardrails stage B deferred to it, split by where each runs: +- **dw-side, already done by 4b/4c, which the prompt points at:** the CI ratchet (4b) and the seam map (`docs/ARCHITECTURE.md`, 4c). +- **Harness-side, which the prompt asks for:** + - an architecture reviewer that reads the seam map and refuses a hand-off that adds a second owner for a concept; + - new-module approval: stage A's new-file refusal under `dw/`, `dw_mcp/` and so on outlives `FREEZE`, still waived by `arch-approved`; + - build-vs-buy: a new hand-rolled implementation of something a dependency already covers is refused, with the reviewer as the check; + - the consolidation cadence: a periodic pass, run by the curator, that reads the gate report's change-coupling pairs and files consolidation issues for Don. +- **Also:** agents may not edit `baseline.json` upward without `arch-approved` (the re-baseline rule). + +`FREEZE`'s own text says the freeze lifts "at the Phase 4 gate"; this ordering is what that means. Its first criterion: Don confirms the harness stage C prompt is committed and its tests pass in `/Users/don/testing/harnest`. Then, on `develop`: `FREEZE` deleted, `hot-zone.txt` emptied, `arch_report.py` with a Gate 4 column into ROADMAP.md, re-baseline, tag `stabilization-gate-4`, lem deploy and smoke, ROADMAP row 4 status, ASSESSMENT refreshed, the Claude Doc's metrics table, and memory. From d285971b38b87eb0412b0ea2c2c4ae4b9fc8de75 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 12:17:45 -0500 Subject: [PATCH 03/41] refactor(workflow): delete the 8 one-line delegators; callers call the module function validation_context, the six warning methods and cache_hits become direct calls to validation.workflow_context / run_warning_check and workflow_run.cache_hits. validation_errors and validate stay methods. No surface change (snapshot byte-identical). Co-Authored-By: Claude Opus 5.5 --- dw/server/admission.py | 4 +- dw/server/routes/library.py | 5 +- dw/validation.py | 4 +- dw/worker.py | 5 +- dw/workflow.py | 90 ------------------------------- scripts/surface_snapshot.py | 8 ++- tests/test_admission.py | 2 +- tests/test_lora_disable.py | 9 +++- tests/test_realize.py | 3 +- tests/test_shot_span_preflight.py | 3 +- tests/test_slice_preflight.py | 5 +- tests/test_validation.py | 31 ++++++++--- tests/test_vram_inheritance.py | 7 ++- tests/test_worker_execute.py | 43 +++++++++------ tests/test_workflow.py | 9 ++-- tests/test_workflow_step_cache.py | 15 +++--- 16 files changed, 105 insertions(+), 138 deletions(-) diff --git a/dw/server/admission.py b/dw/server/admission.py index 5906b8c8..cc966078 100644 --- a/dw/server/admission.py +++ b/dw/server/admission.py @@ -155,7 +155,9 @@ def admit( # and the caller's list is the one a for_each expands over. The # context expands lazily, inside validation_errors' gates, so an # expansion failure is still answered there as a finding - context = candidate.validation_context(checked, ceiling_index=ceiling_index) + context = validation.workflow_context( + candidate, checked, ceiling_index=ceiling_index + ) admission.errors = candidate.validation_errors(context=context) except Exception as e: raise _validator_failure(e) from e diff --git a/dw/server/routes/library.py b/dw/server/routes/library.py index 5cbed9a4..5ec4212a 100644 --- a/dw/server/routes/library.py +++ b/dw/server/routes/library.py @@ -19,6 +19,7 @@ from ...argument_warnings import workflow_argument_warnings from ...prompts import RESERVED_TEXT_PREFIXES from ...schema import format_validation_errors, load_schema, validate_data +from ... import validation from ...security import InvalidInputError, SecurityError, validate_prompt_reference from ...workflow import Workflow from ...library import ( @@ -295,7 +296,9 @@ def save_workflow( # to shape-first discovery metadata = derive_catalog_metadata(request.workflow) warnings = list(workflow_argument_warnings(request.workflow)) - warnings += candidate.null_variable_argument_warnings() + warnings += validation.run_warning_check( + candidate, "null_variable_argument_warnings", None + ) if not metadata["summary"]: warnings.append( "No summary: add a 'description' (its first sentence becomes " diff --git a/dw/validation.py b/dw/validation.py index dac49eb8..13510ff2 100644 --- a/dw/validation.py +++ b/dw/validation.py @@ -659,7 +659,7 @@ def workflow_errors(workflow, arguments=None, composing=None, context=None): internal error rather than a lost verdict (B10). """ if context is None: - context = workflow.validation_context(arguments, composing) + context = workflow_context(workflow, arguments, composing) elif (arguments is not None and arguments != context.arguments) or ( composing is not None and tuple(composing) != context.composing ): @@ -709,7 +709,7 @@ def run_warning_check(workflow, name, arguments, **context_fields): call's own - how the Workflow's warning methods answer when called directly rather than through admit(). An expansion that fails raises inside the check, which makes it one internal warning.""" - context = workflow.validation_context(arguments, **context_fields) + context = workflow_context(workflow, arguments, **context_fields) check = warning_check(name) return to_warnings(run_checks(context, [check], WARNING)) diff --git a/dw/worker.py b/dw/worker.py index 41e7af13..ece2fcbb 100644 --- a/dw/worker.py +++ b/dw/worker.py @@ -15,6 +15,7 @@ # Add parent directory to path for imports sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) +from dw import workflow_run from dw.workflow import workflow_from_snapshot from dw.step_cache import step_cache from dw.assets import activate_asset_dir, deactivate_asset_dir @@ -510,7 +511,9 @@ def _handle_probe_cache(self, command: Dict[str, Any]): else None ) try: - cached = workflow.cache_hits(command.get("arguments") or {}) + cached = workflow_run.cache_hits( + workflow, command.get("arguments") or {} + ) finally: if asset_token is not None: deactivate_asset_dir(asset_token) diff --git a/dw/workflow.py b/dw/workflow.py index 2044365b..ef06b7b1 100644 --- a/dw/workflow.py +++ b/dw/workflow.py @@ -481,26 +481,6 @@ def open_sub_workflow(self, path): resolved, root = self.resolve_sub_workflow_path(path) return workflow_from_file(resolved, self.output_dir, root), resolved - def sub_workflow_warnings(self, arguments=None): - """An argument a sub-workflow step passes down that the workflow it - composes declares no variable for - dropped in silence at run time, - and composition is exactly where a name drifts (#89). - - `arguments` are the caller's, folded in the same way every other - warning source uses them. Each entry is a string, `"path: message"`, - matching every other warnings source - and the path names the step - index the *author* wrote, not the index the step lands at after - `for_each` expansion. Runs the registry's `sub_workflow_warnings` - check; one that raises is an internal warning (B10). - """ - return validation.run_warning_check(self, "sub_workflow_warnings", arguments) - - def validation_context(self, arguments=None, composing=(), *, ceiling_index=None): - """One validation request's ValidationContext (dw.validation's - `workflow_context`): built per request, never stored on the - Workflow, and it expands nothing until a check reads it.""" - return validation.workflow_context(self, arguments, composing, ceiling_index) - def validation_errors(self, arguments=None, composing=None, *, context=None): """Every schema violation in the definition, as [{path, message}]; empty when it validates. `arguments` are the caller's, so a @@ -517,71 +497,6 @@ def validation_errors(self, arguments=None, composing=None, *, context=None): """ return validation.workflow_errors(self, arguments, composing, context) - def adapter_warnings(self, arguments=None): - """Every adapter whose file name says nothing about which checkpoint - partition it was trained for - valid, and worth saying, since - nothing at run time will (#155). - - Runs the registry's check: a definition the expander refuses is one - internal warning, since its own errors are validation_errors' to - report. - """ - return validation.run_warning_check(self, "adapter_warnings", arguments) - - def inherited_vram_warnings(self, arguments=None, index=None): - """Every catalog VRAM ceiling this workflow's expanded steps project - past, matched by pipeline identity (`dw/vram_inheritance.py`) - for a - workflow that declares no `vram_estimate` of its own. A warning, not - an error: the catalog's numbers were measured on the catalog's - offload and quantization config (#479). - - Runs the registry's check, like `adapter_warnings`. - """ - if not index: - return [] - return validation.run_warning_check( - self, "inherited_vram_warnings", arguments, ceiling_index=index - ) - - def slice_past_end_warnings(self, arguments=None): - """Every `slice_audio` step whose source's real duration is already - knowable and whose requested slice reaches past it - valid, padded - with silence rather than refused, but worth saying before the run - rather than only after it (#402). - - Runs the registry's check, like `adapter_warnings`. - """ - return validation.run_warning_check(self, "slice_past_end_warnings", arguments) - - def shot_span_warnings(self, arguments=None): - """Every assessment-probe step (`analyze_shots`, `analyze_seams`, - `analyze_sync_drift`) whose `shots` argument already reaches past a - statically-knowable video's real frame count - valid, silently - clipped to the file rather than refused, but worth saying before the - run rather than only after it (#425). - - Runs the registry's check, like `adapter_warnings`. - """ - return validation.run_warning_check(self, "shot_span_warnings", arguments) - - def null_variable_argument_warnings(self, arguments=None): - """Every required task argument fed by `variable:name` where name's - value is null - downgraded out of `validation_errors` when - `arguments` is None (#364), surfaced here so a caller checking the - document without arguments of its own (save_workflow, - validate_workflow with no `arguments`) still sees it, just not as a - reason the document is invalid. - - Empty once `arguments` is given: at that point the same condition is - a hard error in `validation_errors`, since a real run or a validate - call naming its own arguments needed the variable to hold something. - - Runs the registry's check, like `adapter_warnings`. - """ - return validation.run_warning_check( - self, "null_variable_argument_warnings", arguments - ) - def validate(self, arguments=None): """Validates workflow definition against JSON schema. @@ -601,11 +516,6 @@ def validate(self, arguments=None): raise Exception(message) logger.debug(f"Workflow {self.name} validated successfully") - def cache_hits(self, arguments): - """The steps the step cache would serve for a run with `arguments`, - in step order, executing nothing (dw.workflow_run's `cache_hits`).""" - return workflow_run.cache_hits(self, arguments) - def _owned_arguments(self, arguments): """The composed child's own copy of what its parent handed it (_composed) - only of the names its fold keeps, the ones its diff --git a/scripts/surface_snapshot.py b/scripts/surface_snapshot.py index 13a6a785..f365af69 100644 --- a/scripts/surface_snapshot.py +++ b/scripts/surface_snapshot.py @@ -22,6 +22,7 @@ from fastapi.testclient import TestClient # noqa: E402 +from dw import validation # noqa: E402 from dw.introspection import describe_task, list_tasks # noqa: E402 from dw.server.app import create_app # noqa: E402 from dw.server.jobs import JobManager # noqa: E402 @@ -253,7 +254,12 @@ def catalog_validation(root): continue for name in ("validation_errors", *WARNING_CHECKS): try: - entry[name] = stable_message(getattr(workflow, name)()) + found = ( + workflow.validation_errors() + if name == "validation_errors" + else validation.run_warning_check(workflow, name, None) + ) + entry[name] = stable_message(found) except Exception as error: entry[name] = f"error: {type(error).__name__}: {error}" verdicts[path.relative_to(repo).as_posix()] = entry diff --git a/tests/test_admission.py b/tests/test_admission.py index 2081229d..c693b9db 100644 --- a/tests/test_admission.py +++ b/tests/test_admission.py @@ -496,7 +496,7 @@ def _raise_with_a_path(*_args, **_kwargs): @pytest.mark.parametrize( "target, name", - [(Workflow, "validation_context"), (admission_module, "argument_errors")], + [(validation, "workflow_context"), (admission_module, "argument_errors")], ids=["context", "argument_errors"], ) def test_a_failing_gate_names_only_its_exception_type( diff --git a/tests/test_lora_disable.py b/tests/test_lora_disable.py index b527927f..34cfeca7 100644 --- a/tests/test_lora_disable.py +++ b/tests/test_lora_disable.py @@ -15,6 +15,7 @@ import pytest +from dw import validation from dw.adapter_compatibility import adapter_warnings, warn_adapters from dw.pipeline_processors.adapters import active_loras, load_loras from dw.workflow import Workflow @@ -121,14 +122,18 @@ def test_all_null_validates(self, tmp_path): assert h3_workflow(tmp_path).validation_errors(arguments=arguments) == [] def test_the_caller_is_told_where_they_switched_it_off(self, tmp_path): - warnings = h3_workflow(tmp_path).adapter_warnings({"lora_model_name": None}) + warnings = validation.run_warning_check( + h3_workflow(tmp_path), "adapter_warnings", {"lora_model_name": None} + ) disabled = [w for w in warnings if "not loaded" in w] assert len(disabled) == 1 assert disabled[0].startswith("arguments.lora_model_name: ") assert "num_inference_steps" in disabled[0] def test_nothing_is_said_when_the_lora_is_on(self, tmp_path): - warnings = h3_workflow(tmp_path).adapter_warnings() + warnings = validation.run_warning_check( + h3_workflow(tmp_path), "adapter_warnings", None + ) assert not [w for w in warnings if "not loaded" in w] def test_a_literal_null_is_reported_at_its_step(self): diff --git a/tests/test_realize.py b/tests/test_realize.py index 08016107..6f4d574e 100644 --- a/tests/test_realize.py +++ b/tests/test_realize.py @@ -7,6 +7,7 @@ import pytest +from dw import workflow_run from dw.realize import realize_workflow, strings_with_prefix from dw.runs import new_run_id from dw.schema import load_schema, validate_data @@ -448,7 +449,7 @@ def test_cache_hits_prepares_without_copying_media(self, tmp_path): source["seed"] = 3 wf = Workflow(source, str(tmp_path), None) - assert wf.cache_hits({"image": media, "frames": [media]}) == [] + assert workflow_run.cache_hits(wf, {"image": media, "frames": [media]}) == [] class CountingMedia: diff --git a/tests/test_shot_span_preflight.py b/tests/test_shot_span_preflight.py index 16aacf8f..85b04d48 100644 --- a/tests/test_shot_span_preflight.py +++ b/tests/test_shot_span_preflight.py @@ -8,6 +8,7 @@ import os +from dw import validation from dw.media import probe_metadata from dw.runs import activate_output_root, deactivate_output_root from dw.shot_span_preflight import shot_span_warnings @@ -169,7 +170,7 @@ def test_reachable_from_the_workflow_method(self, monkeypatch, tmp_path): definition, os.path.join(base_dir, "workflow.json") ) - warnings = workflow.shot_span_warnings() + warnings = validation.run_warning_check(workflow, "shot_span_warnings", None) assert len(warnings) == 1 assert "'b'" in warnings[0] diff --git a/tests/test_slice_preflight.py b/tests/test_slice_preflight.py index fddf3b70..e7bfaeed 100644 --- a/tests/test_slice_preflight.py +++ b/tests/test_slice_preflight.py @@ -12,6 +12,7 @@ import numpy +from dw import validation from dw.media import probe_metadata from dw.runs import activate_output_root, deactivate_output_root from dw.slice_preflight import slice_past_end_warnings @@ -183,7 +184,9 @@ def test_reachable_from_the_workflow_method(self, monkeypatch): definition, os.path.join(base_dir, "workflow.json") ) - warnings = workflow.slice_past_end_warnings() + warnings = validation.run_warning_check( + workflow, "slice_past_end_warnings", None + ) assert len(warnings) == 1 assert "score.wav" in warnings[0] diff --git a/tests/test_validation.py b/tests/test_validation.py index ff92be48..c8ad70eb 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -329,7 +329,7 @@ def test_building_a_context_does_not_expand(tmp_path): _unexpandable_definition(), str(tmp_path), str(tmp_path), None ) - context = workflow.validation_context() + context = validation.workflow_context(workflow) errors = workflow.validation_errors(context=context) assert errors == workflow.validation_errors() @@ -346,7 +346,7 @@ def test_a_context_with_other_arguments_is_a_caller_bug(tmp_path): _for_each_definition(), str(tmp_path), str(tmp_path), None ) arguments = {"shots": [{"name": "x", "text": "X"}]} - context = workflow.validation_context(arguments) + context = validation.workflow_context(workflow, arguments) # the context alone, or the context with the same arguments, is fine assert workflow.validation_errors(context=context) == [] @@ -367,15 +367,30 @@ def test_a_warning_method_on_an_unexpandable_definition_says_it_failed(tmp_path) _unexpandable_definition(), str(tmp_path), str(tmp_path), None ) calls = { - "adapter_warnings": lambda: workflow.adapter_warnings(), + "adapter_warnings": lambda: validation.run_warning_check( + workflow, "adapter_warnings", None + ), "null_variable_argument_warnings": ( - lambda: workflow.null_variable_argument_warnings() + lambda: validation.run_warning_check( + workflow, "null_variable_argument_warnings", None + ) + ), + "sub_workflow_warnings": lambda: validation.run_warning_check( + workflow, "sub_workflow_warnings", None + ), + "slice_past_end_warnings": lambda: validation.run_warning_check( + workflow, "slice_past_end_warnings", None + ), + "shot_span_warnings": lambda: validation.run_warning_check( + workflow, "shot_span_warnings", None ), - "sub_workflow_warnings": lambda: workflow.sub_workflow_warnings(), - "slice_past_end_warnings": lambda: workflow.slice_past_end_warnings(), - "shot_span_warnings": lambda: workflow.shot_span_warnings(), "inherited_vram_warnings": ( - lambda: workflow.inherited_vram_warnings(None, {"identity": {}}) + lambda: validation.run_warning_check( + workflow, + "inherited_vram_warnings", + None, + ceiling_index={"identity": {}}, + ) ), } for name, call in calls.items(): diff --git a/tests/test_vram_inheritance.py b/tests/test_vram_inheritance.py index 381be40b..6c289b7f 100644 --- a/tests/test_vram_inheritance.py +++ b/tests/test_vram_inheritance.py @@ -122,8 +122,11 @@ def _inline(*steps, **extra): def _warnings(definition, index=None, arguments=None): with cuda_24gb(): workflow = workflow_from_definition(definition, tempfile.mkdtemp()) - return workflow.inherited_vram_warnings( - arguments, build_index(_catalog()) if index is None else index + return dw.validation.run_warning_check( + workflow, + "inherited_vram_warnings", + arguments, + ceiling_index=build_index(_catalog()) if index is None else index, ) diff --git a/tests/test_worker_execute.py b/tests/test_worker_execute.py index 390ceffc..606a61c5 100644 --- a/tests/test_worker_execute.py +++ b/tests/test_worker_execute.py @@ -9,6 +9,7 @@ import pytest +from dw import workflow_run from dw.events import WorkflowCancelled from dw.pipeline_ownership import PipelineOwnership @@ -278,29 +279,35 @@ def record(): assert alive_at_cleanup == [False] -class ProbableWorkflow(StubWorkflow): - def __init__(self, hits): - super().__init__() - self.hits = hits - self.probed_with = None +def probe_answering(hits, seen=None): + """A stand-in for `workflow_run.cache_hits`, recording what it was asked.""" - def cache_hits(self, arguments): - self.probed_with = arguments - return list(self.hits) + def cache_hits(workflow, arguments): + if seen is not None: + seen["workflow"] = workflow + seen["arguments"] = arguments + return list(hits) + + return cache_hits def test_probe_cache_answers_with_the_workflows_hits(): worker = _make_worker() - workflow = ProbableWorkflow(["gen"]) + workflow = StubWorkflow() + seen = {} command = snapshot_command( type="probe_cache", request_id="p-1", arguments={"prompt": "p"} ) - with patch("dw.worker.workflow_from_snapshot", return_value=workflow): + with ( + patch("dw.worker.workflow_from_snapshot", return_value=workflow), + patch.object(workflow_run, "cache_hits", probe_answering(["gen"], seen)), + ): worker._handle_probe_cache(command) assert _drain(worker.result_queue) == [ {"type": "probe_cache", "request_id": "p-1", "cached": ["gen"]} ] - assert workflow.probed_with == {"prompt": "p"} + assert seen["workflow"] is workflow + assert seen["arguments"] == {"prompt": "p"} def test_probe_cache_reports_a_failure_as_unknown_not_as_a_crash(): @@ -318,14 +325,16 @@ def test_probe_cache_activates_the_jobs_asset_dir(tmp_path): worker = _make_worker() seen = {} - class AssetAwareWorkflow(ProbableWorkflow): - def cache_hits(self, arguments): - from dw.assets import get_asset_dir + def asset_aware_cache_hits(workflow, arguments): + from dw.assets import get_asset_dir - seen["asset_dir"] = get_asset_dir() - return [] + seen["asset_dir"] = get_asset_dir() + return [] - with patch("dw.worker.workflow_from_snapshot", return_value=AssetAwareWorkflow([])): + with ( + patch("dw.worker.workflow_from_snapshot", return_value=StubWorkflow()), + patch.object(workflow_run, "cache_hits", asset_aware_cache_hits), + ): worker._handle_probe_cache( snapshot_command(type="probe_cache", asset_dir=str(tmp_path)) ) diff --git a/tests/test_workflow.py b/tests/test_workflow.py index 123da102..6aaff8d7 100644 --- a/tests/test_workflow.py +++ b/tests/test_workflow.py @@ -3,6 +3,7 @@ import torch import tempfile from unittest.mock import MagicMock +from dw import validation from dw.workflow import Workflow, workflow_from_file from dw.library import workflow_output_subfolder from dw.step_cache import referenced_result_names @@ -1449,7 +1450,9 @@ def test_a_resolvable_child_validates_clean(self, tmp_path): workflow = self._parent(workflows, "child", {"prompt": "a dog"}) assert workflow.validation_errors() == [] - assert workflow.sub_workflow_warnings() == [] + assert ( + validation.run_warning_check(workflow, "sub_workflow_warnings", None) == [] + ) def test_an_argument_the_child_does_not_declare_warns(self, tmp_path): import json @@ -1475,7 +1478,7 @@ def test_an_argument_the_child_does_not_declare_warns(self, tmp_path): ) workflow = self._parent(workflows, "child", {"promt": "a dog"}) - warnings = workflow.sub_workflow_warnings() + warnings = validation.run_warning_check(workflow, "sub_workflow_warnings", None) assert all(isinstance(w, str) for w in warnings) assert warnings and warnings[0].startswith( @@ -1518,7 +1521,7 @@ def test_sub_workflow_warnings_are_strings_at_the_authors_step(self, tmp_path): }, ) - warnings = workflow.sub_workflow_warnings() + warnings = validation.run_warning_check(workflow, "sub_workflow_warnings", None) assert all(isinstance(w, str) for w in warnings) assert warnings and warnings[0].startswith( diff --git a/tests/test_workflow_step_cache.py b/tests/test_workflow_step_cache.py index e6975e53..b38cf499 100644 --- a/tests/test_workflow_step_cache.py +++ b/tests/test_workflow_step_cache.py @@ -653,7 +653,10 @@ def fake_step_run(self, previous_results, previous_pipelines, step_action): patch.object(Pipeline, "load", _editing_loader([])), ): workflow.run({"prompt_b": "first"}) - assert workflow.cache_hits({"prompt_b": "first"}) == ["A", "B"] + assert workflow_run_module.cache_hits(workflow, {"prompt_b": "first"}) == [ + "A", + "B", + ] def test_a_released_deferred_hit_still_shares_its_components(tmp_path): @@ -1006,7 +1009,7 @@ def test_a_cold_cache_reports_no_hits(self, tmp_path): step_cache.clear() workflow, _ = build_test_workflow_and_call_count_spy(str(tmp_path)) try: - assert workflow.cache_hits({}) == [] + assert workflow_run_module.cache_hits(workflow, {}) == [] finally: for p in workflow._test_patcher: p.stop() @@ -1016,7 +1019,7 @@ def test_after_a_run_the_probe_names_what_the_next_run_reuses(self, tmp_path): workflow, call_count = build_test_workflow_and_call_count_spy(str(tmp_path)) try: workflow.run({}) - probe = workflow.cache_hits({}) + probe = workflow_run_module.cache_hits(workflow, {}) workflow.run({}) reused = [ entry["step"] for entry in workflow.manifest if entry.get("reused") @@ -1033,7 +1036,7 @@ def test_a_changed_argument_is_a_miss(self, tmp_path): workflow, _ = build_test_workflow_and_call_count_spy(str(tmp_path)) try: workflow.run({"prompt": "a cat"}) - assert workflow.cache_hits({"prompt": "a dog"}) == [] + assert workflow_run_module.cache_hits(workflow, {"prompt": "a dog"}) == [] finally: for p in workflow._test_patcher: p.stop() @@ -1044,7 +1047,7 @@ def test_an_unseeded_workflow_has_no_hits(self, tmp_path): del workflow.workflow_definition["seed"] try: workflow.run({}) - assert workflow.cache_hits({}) == [] + assert workflow_run_module.cache_hits(workflow, {}) == [] finally: for p in workflow._test_patcher: p.stop() @@ -1053,7 +1056,7 @@ def test_the_probe_writes_nothing(self, tmp_path): step_cache.clear() workflow, _ = build_test_workflow_and_call_count_spy(str(tmp_path)) try: - workflow.cache_hits({}) + workflow_run_module.cache_hits(workflow, {}) assert list(tmp_path.iterdir()) == [] finally: for p in workflow._test_patcher: From 0c1a61fcd5e449329f083f38417e9c53dca9015a Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 12:23:34 -0500 Subject: [PATCH 04/41] refactor(library): one sub-workflow path resolver for every site resolve_sub_workflow_reference owns the builtin branch, the catalog-root fallback, the search path and the confinement check. create_step_action, resolve_sub_workflow_path, open_sub_workflow (now takes resolved, so sub_workflow_errors resolves once), realize.read_sub_workflow and the observed-cost lookup all call it. Fixes the builtin name 'replace' bug. Co-Authored-By: Claude Opus 5.5 --- docs/RELEASING.md | 7 ++ dw/library.py | 53 ++++++++ dw/realize.py | 25 ++-- dw/server/routes/jobs.py | 13 +- dw/validation.py | 5 +- dw/workflow.py | 99 ++------------- tests/test_sub_workflow_resolver.py | 179 ++++++++++++++++++++++++++++ 7 files changed, 274 insertions(+), 107 deletions(-) create mode 100644 tests/test_sub_workflow_resolver.py diff --git a/docs/RELEASING.md b/docs/RELEASING.md index 8cd1d41f..9cd1fdaa 100644 --- a/docs/RELEASING.md +++ b/docs/RELEASING.md @@ -7,6 +7,13 @@ notes from commits at tag time (see below). This section is a scratch pad for items a branch's author wants the next release note to name; clear it when a release ships. +### 0.7.0 + +- A sub-workflow path is resolved by one function (`library.resolve_sub_workflow_reference`) at every site, so a path a run can open is one validation, the realized workflow's digest and the observed-cost lookup can open too. + - `builtin:builtin:x.json` no longer loads `x.json`: only the leading prefix is stripped, so the name `builtin:x.json` is looked up and reported as `SubWorkflowNotFound`. + - A run with no `workflow_dir` now resolves a catalog name (`models/x`) in a sub-workflow step through the catalog root, as validation already did, and a missing builtin fails with `SubWorkflowNotFound` naming the packaged root. + - The realized workflow's sub-workflow digest and a composed child's observed cost now fall back to the same catalog root, so a catalog sub-workflow a run could open is also digested and costed. + ### 0.6.0 diff --git a/dw/library.py b/dw/library.py index f55da7c6..eb338916 100644 --- a/dw/library.py +++ b/dw/library.py @@ -28,7 +28,9 @@ import logging import os +from . import references from .security import ( + InvalidInputError, PathTraversalError, SecurityError, contained, @@ -715,3 +717,54 @@ def resolve_sub_workflow(path, base_dir, confine_to): return candidate, root raise SubWorkflowNotFound(path, tried) + + +def resolve_sub_workflow_reference(path, file_spec, confine_to): + """Where one sub-workflow step's `path` resolves to, validated and + confined, as (path, root) - `root` is the directory the child is confined + to (None when the run is unconfined). The one preamble every site that + asks shares: a run (`create_step_action`), validation, the realized + workflow's digest and the observed-cost lookup, so a path one can open is + a path the others can. + + `file_spec` is the workflow that names the step; `confine_to` its + `workflow_dir`. + + - `builtin:.json` is looked up only in `builtin_root()`, the + packaged workflows, and confined to it whatever `confine_to` is + - any other relative path in an unconfined run is confined to the + catalog root (`catalog_root_dir`) so it can reach a sibling folder + but not leave the catalog + - everything else goes through `resolve_sub_workflow`'s search path + + Raises SubWorkflowNotFound, SecurityError or InvalidInputError, each + carrying the message the run itself would fail with. + """ + builtin_name = references.ref_name(references.BUILTIN, path) + if builtin_name is not None: + if ( + not builtin_name.endswith(".json") + or "/" in builtin_name + or "\\" in builtin_name + ): + raise InvalidInputError( + f"Invalid builtin workflow name: {builtin_name}. It must " + "be a bare '.json' filename with no path segments - " + f"'builtin:' only looks in the packaged workflows root: " + f"{builtin_root()}" + ) + root = builtin_root() + resolved = os.path.join(root, builtin_name) + # The name is bare, so it lands in `root`; the validator's answer is + # what gets stat'ed + candidate = validate_path(resolved, root, allow_create=True) + if not os.path.isfile(candidate): + raise SubWorkflowNotFound(path, [resolved]) + return validate_workflow_path(resolved, root), root + if confine_to is None and not os.path.isabs(path): + confine_to = catalog_root_dir(file_spec) + resolved, library_root = resolve_sub_workflow( + path, os.path.dirname(file_spec), confine_to + ) + confine_to = library_root.root if library_root else None + return validate_workflow_path(resolved, confine_to), confine_to diff --git a/dw/realize.py b/dw/realize.py index 6caebf6c..6670a0b7 100644 --- a/dw/realize.py +++ b/dw/realize.py @@ -33,9 +33,9 @@ resolve_output_reference, version_selector, ) -from .security import SecurityError, validate_workflow_path +from .security import SecurityError from .step_cache import copy_containers -from .library import resolve_sub_workflow, SubWorkflowNotFound +from .library import resolve_sub_workflow_reference, SubWorkflowNotFound logger = logging.getLogger("dw") @@ -233,16 +233,21 @@ def read_sub_workflow(path, base_dir, workflow_dir): """The bytes of the sub-workflow file a step's `path` names, or None when it cannot be read. - Resolved the way `Workflow.create_step_action` resolves it - beside the - referencing file, then across the workflow search path, then through - `validate_workflow_path` confined to the root it came from - so a path - this run could not have loaded is not one realization (or the planner) - reads either, and a catalog name the run composed is read rather than - recorded as unreadable (#90). + Resolved by `resolve_sub_workflow_reference`, the resolver + `Workflow.create_step_action` uses - beside the referencing file, then + across the workflow search path, confined to the root it came from - so + a path this run could not have loaded is not one realization (or the + planner) reads either, and a catalog name the run composed is read + rather than recorded as unreadable (#90). `base_dir` is the referencing + file's directory; a run with no `workflow_dir` is confined to the + catalog root above it. """ try: - candidate, root = resolve_sub_workflow(path, base_dir or ".", workflow_dir) - validated = validate_workflow_path(candidate, root.root if root else None) + # Only the directory of the referencing file is known here; the + # resolver wants the file, and takes its directory back off + validated, _ = resolve_sub_workflow_reference( + path, os.path.join(base_dir or ".", "workflow.json"), workflow_dir + ) with open(validated, "rb") as file: return file.read() except (SecurityError, OSError, ValueError, SubWorkflowNotFound) as e: diff --git a/dw/server/routes/jobs.py b/dw/server/routes/jobs.py index 24e3d813..de75ff86 100644 --- a/dw/server/routes/jobs.py +++ b/dw/server/routes/jobs.py @@ -22,7 +22,7 @@ from ...plan import build_plan, gate_warnings from ...schema import format_validation_errors from ...security import SecurityError -from ...library import SubWorkflowNotFound, resolve_sub_workflow +from ...library import SubWorkflowNotFound, resolve_sub_workflow_reference from ...workspace import Workspace from ..admission import ( ACKNOWLEDGED_COST_FIELD, @@ -548,18 +548,15 @@ def observed_for_child(path, child_definition, arguments=None): *that* value rather than always the child's stored defaults, which silently answered the default bucket's history for every override.""" - base_dir = ( - os.path.dirname(os.path.abspath(candidate.file_spec)) - if candidate.file_spec - else None + file_spec = os.path.abspath( + candidate.file_spec or os.path.join(".", "workflow.json") ) try: - child_path, child_library_root = resolve_sub_workflow( - path, base_dir or ".", candidate.workflow_dir + child_path, child_root = resolve_sub_workflow_reference( + path, file_spec, candidate.workflow_dir ) except (SecurityError, OSError, ValueError, SubWorkflowNotFound): return None - child_root = child_library_root.root if child_library_root else None child_name = catalog_name_from_root(child_path, child_root) if not child_name: return None diff --git a/dw/validation.py b/dw/validation.py index 13510ff2..253eecea 100644 --- a/dw/validation.py +++ b/dw/validation.py @@ -773,7 +773,8 @@ def sub_workflow_errors(workflow, expanded, source_indices=None, composing=None) where = f"steps[{source}].workflow.path" path = reference["path"] try: - resolved, _ = workflow.resolve_sub_workflow_path(path) + resolution = workflow.resolve_sub_workflow_path(path) + resolved = resolution[0] except (SubWorkflowNotFound, SecurityError, InvalidInputError) as e: errors.append({"path": where, "message": str(e)}) continue @@ -790,7 +791,7 @@ def sub_workflow_errors(workflow, expanded, source_indices=None, composing=None) ) continue try: - child, _ = workflow.open_sub_workflow(path) + child, _ = workflow.open_sub_workflow(path, resolution) except Exception as e: errors.append({"path": where, "message": f"Sub-workflow '{path}': {e}"}) continue diff --git a/dw/workflow.py b/dw/workflow.py index ef06b7b1..8659c758 100644 --- a/dw/workflow.py +++ b/dw/workflow.py @@ -3,7 +3,6 @@ import json import copy import logging -from . import references from .arguments import realize_constants, fetch_constant, is_constant_reference from .events import ( RunContext, @@ -51,11 +50,8 @@ UntrustedWorkflowError, ) from .library import ( - builtin_root, - catalog_root_dir, - resolve_sub_workflow, + resolve_sub_workflow_reference, workflow_output_subfolder, - SubWorkflowNotFound, ) logger = logging.getLogger("dw") @@ -445,40 +441,17 @@ def resolve_sub_workflow_path(self, path): Raises SubWorkflowNotFound, SecurityError or InvalidInputError, each carrying the message the run would have failed with. """ - confine_to = self.workflow_dir - if references.is_ref(references.BUILTIN, path): - builtin_name = path.replace(references.BUILTIN, "") - if ( - not builtin_name.endswith(".json") - or "/" in builtin_name - or "\\" in builtin_name - ): - raise InvalidInputError( - f"Invalid builtin workflow name: {builtin_name}. It must " - "be a bare '.json' filename with no path segments - " - f"'builtin:' only looks in the packaged workflows root: " - f"{builtin_root()}" - ) - confine_to = builtin_root() - resolved = os.path.join(confine_to, builtin_name) - if not os.path.isfile(resolved): - raise SubWorkflowNotFound(path, [resolved]) - return validate_workflow_path(resolved, confine_to), confine_to - if confine_to is None and not os.path.isabs(path): - confine_to = catalog_root_dir(self.file_spec) - resolved, resolved_root = resolve_sub_workflow( - path, os.path.dirname(self.file_spec), confine_to - ) - confine_to = resolved_root.root if resolved_root else None - return validate_workflow_path(resolved, confine_to), confine_to + return resolve_sub_workflow_reference(path, self.file_spec, self.workflow_dir) - def open_sub_workflow(self, path): + def open_sub_workflow(self, path, resolved=None): """The workflow one sub-workflow step's `path` names, opened as (child, resolved) - resolved is its path, which is what a composition chain records. How dw.validation builds a composed child - without importing this module. Raises what resolution or the load - raises.""" - resolved, root = self.resolve_sub_workflow_path(path) + without importing this module. `resolved` is what + `resolve_sub_workflow_path` already answered for `path`, as + (path, root); given, it is opened without resolving again. Raises + what resolution or the load raises.""" + resolved, root = resolved or self.resolve_sub_workflow_path(path) return workflow_from_file(resolved, self.output_dir, root), resolved def validation_errors(self, arguments=None, composing=None, *, context=None): @@ -824,58 +797,10 @@ def _sub_workflow_action(self, step_definition, default_seed): try: # Sub-workflow steps are confined to the same directory this - # workflow is (workflow_dir for a server-submitted run) - confine_to = self.workflow_dir - # Handle built-in workflows - if references.is_ref(references.BUILTIN, path): - builtin_name = path.replace(references.BUILTIN, "") - # Validate builtin workflow name - if ( - not builtin_name.endswith(".json") - or "/" in builtin_name - or "\\" in builtin_name - ): - raise InvalidInputError( - f"Invalid builtin workflow name: {builtin_name}. " - "It must be a bare '.json' filename with no " - "path segments - 'builtin:' only looks in the " - f"packaged workflows root: {builtin_root()}" - ) - # Builtins ship inside the package, outside any - # workflow_dir - confine them to their own directory - # instead (the name check above already forbids escaping it) - confine_to = os.path.join( - os.path.dirname(os.path.abspath(__file__)), "workflows" - ) - path = os.path.join(confine_to, builtin_name) - # Everything else - a relative path, or a catalog name as - # list_workflows reports it - goes through the search path. - # A template under templates/ names a model config as - # '../models/x.json', so a path relative to the referencing - # file still resolves first and the '..' is collapsed here, - # which is what lets the validator judge where the path - # actually lands rather than refusing the spelling; - # containment is still checked on the resolved path below. - # An unconfined run (no workflow_dir - a bare CLI - # invocation) used to rely on the '..' regex alone to stop a - # relative reference from leaving the file's own directory; - # normalizing the path removes that guard, so confine it to - # the catalog root instead - the referencing file's nearest - # ancestor literally named 'workflows', which still lets it - # climb to a sibling folder like models/ but not out of the - # catalog - else: - if confine_to is None and not os.path.isabs(path): - confine_to = catalog_root_dir(self.file_spec) - path, resolved_root = resolve_sub_workflow( - path, os.path.dirname(self.file_spec), confine_to - ) - confine_to = resolved_root.root if resolved_root else None - - # Validate the resolved path - confined when this workflow - # itself is (an inline/server-submitted run), so a - # sub-workflow step cannot escape that boundary - validated_path = validate_workflow_path(path, confine_to) + # workflow is (workflow_dir for a server-submitted run); the + # resolver confines a builtin to the packaged root instead and + # validates the path it hands back + validated_path, confine_to = self.resolve_sub_workflow_path(path) workflow = workflow_from_file(validated_path, self.output_dir, confine_to) except SecurityError as e: diff --git a/tests/test_sub_workflow_resolver.py b/tests/test_sub_workflow_resolver.py new file mode 100644 index 00000000..8317f6c4 --- /dev/null +++ b/tests/test_sub_workflow_resolver.py @@ -0,0 +1,179 @@ +"""One resolver for a sub-workflow step's path, and every site that asks it. + +`resolve_sub_workflow_reference` owns the preamble (the builtin branch, the +catalog-root fallback, the search path, the confinement check), so a path a run +can open is one validation can open, one realization can digest and one cost +lookup can find - and the other way round. +""" + +import json +import os + +import pytest + +from dw.library import ( + SubWorkflowNotFound, + builtin_root, + resolve_sub_workflow_reference, +) +from dw.security import PathTraversalError +from dw.validation import sub_workflow_errors +from dw.workflow import Workflow + +with open(os.path.join(builtin_root(), "test.json")) as file: + CHILD = {**json.load(file), "id": "child"} + + +@pytest.fixture +def catalog(tmp_path): + """workflows/templates/ beside workflows/models/, a decoy outside it and + a parent file under templates/.""" + root = tmp_path / "workflows" + (root / "templates").mkdir(parents=True) + (root / "models").mkdir() + (root / "models" / "Child.json").write_text(json.dumps(CHILD)) + (tmp_path / "Outside.json").write_text(json.dumps(CHILD)) + return root, str(root / "templates" / "parent.json") + + +def parent_in(catalog, confined=False): + root, file_spec = catalog + return Workflow( + {"id": "parent", "steps": []}, + str(root.parent / "outputs"), + file_spec, + str(root) if confined else None, + ) + + +class TestTheResolver: + def test_a_catalog_name_resolves_through_the_catalog_root(self, catalog): + """'models/Child' is no path beside templates/parent.json: it is found + only because an unconfined run falls back to the catalog root.""" + root, file_spec = catalog + path, confine_to = resolve_sub_workflow_reference( + "models/Child", file_spec, None + ) + assert path == str(root / "models" / "Child.json") + assert confine_to == str(root) + + def test_a_builtin_resolves_in_the_packaged_root(self, catalog): + _root, file_spec = catalog + path, confine_to = resolve_sub_workflow_reference( + "builtin:test.json", file_spec, None + ) + assert path.startswith(builtin_root()) + assert confine_to == builtin_root() + + def test_only_the_prefix_is_stripped_from_a_builtin_name(self, catalog): + """A name carrying 'builtin:' again was read with it removed + everywhere, which made 'builtin:builtin:test.json' load test.json; + the second one is part of the name, which names no file.""" + _root, file_spec = catalog + with pytest.raises(SubWorkflowNotFound) as exc_info: + resolve_sub_workflow_reference("builtin:builtin:test.json", file_spec, None) + assert "builtin:test.json" in str(exc_info.value) + + def test_a_missing_builtin_is_not_found(self, catalog): + _root, file_spec = catalog + with pytest.raises(SubWorkflowNotFound) as exc_info: + resolve_sub_workflow_reference("builtin:absent.json", file_spec, None) + assert "absent.json" in str(exc_info.value) + + @pytest.mark.parametrize("confined", [False, True]) + def test_a_path_escaping_its_root_is_refused(self, catalog, confined): + root, file_spec = catalog + with pytest.raises(PathTraversalError): + resolve_sub_workflow_reference( + "../../Outside.json", file_spec, str(root) if confined else None + ) + + +class TestEverySiteReachesIt: + def test_create_step_action(self, catalog): + parent = parent_in(catalog) + step = {"name": "child", "workflow": {"path": "models/Child"}} + + child = parent.create_step_action(step, {}, {}, 42, "cpu") + + assert child.name == "child" + + def test_create_step_action_refuses_a_climb_like_validation_does(self, catalog): + parent = parent_in(catalog, confined=True) + step = {"name": "child", "workflow": {"path": "../../Outside.json"}} + + with pytest.raises(PathTraversalError): + parent.create_step_action(step, {}, {}, 42, "cpu") + + def test_open_sub_workflow(self, catalog): + root, _ = catalog + child, resolved = parent_in(catalog).open_sub_workflow("models/Child") + assert resolved == str(root / "models" / "Child.json") + assert child.name == "child" + + def test_open_sub_workflow_takes_an_already_resolved_path(self, catalog): + parent = parent_in(catalog) + resolved, root = parent.resolve_sub_workflow_path("models/Child") + + child, again = parent.open_sub_workflow("not-resolved-again", (resolved, root)) + + assert again == resolved + assert child.name == "child" + + def test_sub_workflow_errors_resolves_a_catalog_name(self, catalog): + parent = parent_in(catalog) + expanded = {"steps": [{"name": "c", "workflow": {"path": "models/Child"}}]} + assert sub_workflow_errors(parent, expanded) == [] + + def test_sub_workflow_errors_names_a_missing_builtin(self, catalog): + parent = parent_in(catalog) + expanded = { + "steps": [{"name": "c", "workflow": {"path": "builtin:absent.json"}}] + } + (error,) = sub_workflow_errors(parent, expanded) + assert "absent.json" in error["message"] + + def test_read_sub_workflow_falls_back_to_the_catalog_root(self, catalog): + from dw.realize import read_sub_workflow + + _root, file_spec = catalog + raw = read_sub_workflow("models/Child", str(file_spec).rsplit("/", 1)[0], None) + assert json.loads(raw) == CHILD + + def test_the_observed_cost_lookup_falls_back_to_the_catalog_root( + self, catalog, monkeypatch + ): + from types import SimpleNamespace + + from dw.server.routes import jobs + + root, file_spec = catalog + seen = {} + monkeypatch.setattr( + jobs, "build_plan", lambda *a, **kw: seen.update(kw) or {"plan": True} + ) + monkeypatch.setattr( + jobs, + "observed_for_name", + lambda state, name, *a, **kw: ("observed", name), + ) + candidate = SimpleNamespace( + file_spec=file_spec, workflow_dir=None, workflow_definition={"steps": []} + ) + request = SimpleNamespace(workflow_path=None, arguments={}) + workspace = SimpleNamespace( + name="default", outputs=None, assets=None, prompts=None, workflows=str(root) + ) + jobs._validation_plan( + SimpleNamespace(job_manager=None), + candidate, + request, + workspace, + None, + None, + False, + ) + + found = seen["observed_for_child"]("models/Child", CHILD) + + assert found == ("observed", "models/Child") From aad0bc4f8e84c08940656fe5a69958477b6eed03 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 12:28:16 -0500 Subject: [PATCH 05/41] refactor(library): the sub-workflow resolver takes base_dir; tests assert only what the shared resolver gives Co-Authored-By: Claude Opus 5.5 --- docs/RELEASING.md | 2 +- docs/stabilization/phase-4.md | 2 +- dw/library.py | 14 +- dw/realize.py | 4 +- dw/server/routes/jobs.py | 8 +- dw/validation.py | 5 +- dw/workflow.py | 4 +- tests/test_sub_workflow_resolver.py | 198 ++++++++++++++++++---------- 8 files changed, 150 insertions(+), 87 deletions(-) diff --git a/docs/RELEASING.md b/docs/RELEASING.md index 9cd1fdaa..dceaceec 100644 --- a/docs/RELEASING.md +++ b/docs/RELEASING.md @@ -11,7 +11,7 @@ when a release ships. - A sub-workflow path is resolved by one function (`library.resolve_sub_workflow_reference`) at every site, so a path a run can open is one validation, the realized workflow's digest and the observed-cost lookup can open too. - `builtin:builtin:x.json` no longer loads `x.json`: only the leading prefix is stripped, so the name `builtin:x.json` is looked up and reported as `SubWorkflowNotFound`. - - A run with no `workflow_dir` now resolves a catalog name (`models/x`) in a sub-workflow step through the catalog root, as validation already did, and a missing builtin fails with `SubWorkflowNotFound` naming the packaged root. + - At run time a missing `builtin:` workflow now raises `SubWorkflowNotFound` (naming the packaged root) instead of `validate_workflow_path`'s missing-file error. - The realized workflow's sub-workflow digest and a composed child's observed cost now fall back to the same catalog root, so a catalog sub-workflow a run could open is also digested and costed. ### 0.6.0 diff --git a/docs/stabilization/phase-4.md b/docs/stabilization/phase-4.md index efa0fd9e..118d081c 100644 --- a/docs/stabilization/phase-4.md +++ b/docs/stabilization/phase-4.md @@ -124,7 +124,7 @@ Work on branch `stabilization/phase-4a` in the worktree, from `develop` at `e768 - **The 8 delegators are deleted; callers call the module function.** Production: `validation.workflow_context(workflow, ...)`, `validation.run_warning_check(workflow, "null_variable_argument_warnings")`, `workflow_run.cache_hits(workflow, arguments)`. Tests call `validation.run_warning_check(workflow, "", ...)`. `surface_snapshot.py` does the same by name. `test_admission.py:499` patches `validation.workflow_context` with `patch.object`. The worker's fakes in `test_worker_execute.py` move to a `patch.object(workflow_run, "cache_hits", ...)`. - `inherited_vram_warnings`' empty-index guard moves into the check, if `run_warning_check` does not already return nothing for an empty index. The task checks this first. - `Workflow.validation_errors` and `validate` stay methods: they are the class's public entry points, not delegators to a check. -- **One sub-workflow resolver.** `resolve_sub_workflow_reference(path, file_spec, confine_to)` in `dw/library.py` owns the whole preamble: the builtin branch on `builtin_root()` with `ref_name(BUILTIN, ...)` and the existence check, then the `catalog_root_dir` fallback, `resolve_sub_workflow` and `validate_workflow_path`. It returns `(path, root)`. +- **One sub-workflow resolver.** `resolve_sub_workflow_reference(path, base_dir, confine_to)` in `dw/library.py` owns the whole preamble: the builtin branch on `builtin_root()` with `ref_name(BUILTIN, ...)` and the existence check, then the `catalog_root(base_dir)` fallback, `resolve_sub_workflow` and `validate_workflow_path`. It returns `(path, root)`. - `Workflow.resolve_sub_workflow_path` becomes a one-line call to it, and `_sub_workflow_action`'s inline copy is deleted in favour of that call. - `open_sub_workflow(path, resolved=None)` opens an already-resolved path without resolving again. `sub_workflow_errors` resolves once and hands the result on, with both message forms unchanged. - `realize.read_sub_workflow` and `routes/jobs.py:557` call it too. That gives them the `catalog_root_dir` fallback they lacked, so a catalog sub-workflow a run could open is now also digested and costed. That goes in the release notes. diff --git a/dw/library.py b/dw/library.py index eb338916..4b89f561 100644 --- a/dw/library.py +++ b/dw/library.py @@ -719,7 +719,7 @@ def resolve_sub_workflow(path, base_dir, confine_to): raise SubWorkflowNotFound(path, tried) -def resolve_sub_workflow_reference(path, file_spec, confine_to): +def resolve_sub_workflow_reference(path, base_dir, confine_to): """Where one sub-workflow step's `path` resolves to, validated and confined, as (path, root) - `root` is the directory the child is confined to (None when the run is unconfined). The one preamble every site that @@ -727,13 +727,13 @@ def resolve_sub_workflow_reference(path, file_spec, confine_to): workflow's digest and the observed-cost lookup, so a path one can open is a path the others can. - `file_spec` is the workflow that names the step; `confine_to` its - `workflow_dir`. + `base_dir` is the directory of the workflow that names the step; + `confine_to` its `workflow_dir`. - `builtin:.json` is looked up only in `builtin_root()`, the packaged workflows, and confined to it whatever `confine_to` is - any other relative path in an unconfined run is confined to the - catalog root (`catalog_root_dir`) so it can reach a sibling folder + catalog root (`catalog_root`) so it can reach a sibling folder but not leave the catalog - everything else goes through `resolve_sub_workflow`'s search path @@ -762,9 +762,7 @@ def resolve_sub_workflow_reference(path, file_spec, confine_to): raise SubWorkflowNotFound(path, [resolved]) return validate_workflow_path(resolved, root), root if confine_to is None and not os.path.isabs(path): - confine_to = catalog_root_dir(file_spec) - resolved, library_root = resolve_sub_workflow( - path, os.path.dirname(file_spec), confine_to - ) + confine_to = catalog_root(base_dir) + resolved, library_root = resolve_sub_workflow(path, base_dir, confine_to) confine_to = library_root.root if library_root else None return validate_workflow_path(resolved, confine_to), confine_to diff --git a/dw/realize.py b/dw/realize.py index 6670a0b7..6f878268 100644 --- a/dw/realize.py +++ b/dw/realize.py @@ -243,10 +243,8 @@ def read_sub_workflow(path, base_dir, workflow_dir): catalog root above it. """ try: - # Only the directory of the referencing file is known here; the - # resolver wants the file, and takes its directory back off validated, _ = resolve_sub_workflow_reference( - path, os.path.join(base_dir or ".", "workflow.json"), workflow_dir + path, base_dir or ".", workflow_dir ) with open(validated, "rb") as file: return file.read() diff --git a/dw/server/routes/jobs.py b/dw/server/routes/jobs.py index de75ff86..cfed5fa5 100644 --- a/dw/server/routes/jobs.py +++ b/dw/server/routes/jobs.py @@ -548,12 +548,14 @@ def observed_for_child(path, child_definition, arguments=None): *that* value rather than always the child's stored defaults, which silently answered the default bucket's history for every override.""" - file_spec = os.path.abspath( - candidate.file_spec or os.path.join(".", "workflow.json") + base_dir = ( + os.path.dirname(os.path.abspath(candidate.file_spec)) + if candidate.file_spec + else "." ) try: child_path, child_root = resolve_sub_workflow_reference( - path, file_spec, candidate.workflow_dir + path, base_dir, candidate.workflow_dir ) except (SecurityError, OSError, ValueError, SubWorkflowNotFound): return None diff --git a/dw/validation.py b/dw/validation.py index 253eecea..ca1182ba 100644 --- a/dw/validation.py +++ b/dw/validation.py @@ -773,8 +773,7 @@ def sub_workflow_errors(workflow, expanded, source_indices=None, composing=None) where = f"steps[{source}].workflow.path" path = reference["path"] try: - resolution = workflow.resolve_sub_workflow_path(path) - resolved = resolution[0] + resolved, root = workflow.resolve_sub_workflow_path(path) except (SubWorkflowNotFound, SecurityError, InvalidInputError) as e: errors.append({"path": where, "message": str(e)}) continue @@ -791,7 +790,7 @@ def sub_workflow_errors(workflow, expanded, source_indices=None, composing=None) ) continue try: - child, _ = workflow.open_sub_workflow(path, resolution) + child, _ = workflow.open_sub_workflow(path, (resolved, root)) except Exception as e: errors.append({"path": where, "message": f"Sub-workflow '{path}': {e}"}) continue diff --git a/dw/workflow.py b/dw/workflow.py index 8659c758..c2d5a0a7 100644 --- a/dw/workflow.py +++ b/dw/workflow.py @@ -441,7 +441,9 @@ def resolve_sub_workflow_path(self, path): Raises SubWorkflowNotFound, SecurityError or InvalidInputError, each carrying the message the run would have failed with. """ - return resolve_sub_workflow_reference(path, self.file_spec, self.workflow_dir) + return resolve_sub_workflow_reference( + path, os.path.dirname(self.file_spec), self.workflow_dir + ) def open_sub_workflow(self, path, resolved=None): """The workflow one sub-workflow step's `path` names, opened as diff --git a/tests/test_sub_workflow_resolver.py b/tests/test_sub_workflow_resolver.py index 8317f6c4..fe13da0b 100644 --- a/tests/test_sub_workflow_resolver.py +++ b/tests/test_sub_workflow_resolver.py @@ -46,21 +46,26 @@ def parent_in(catalog, confined=False): ) +def catalog_dirs(catalog): + root, file_spec = catalog + return root, os.path.dirname(file_spec) + + class TestTheResolver: def test_a_catalog_name_resolves_through_the_catalog_root(self, catalog): """'models/Child' is no path beside templates/parent.json: it is found only because an unconfined run falls back to the catalog root.""" - root, file_spec = catalog + root, base_dir = catalog_dirs(catalog) path, confine_to = resolve_sub_workflow_reference( - "models/Child", file_spec, None + "models/Child", base_dir, None ) assert path == str(root / "models" / "Child.json") assert confine_to == str(root) def test_a_builtin_resolves_in_the_packaged_root(self, catalog): - _root, file_spec = catalog + _root, base_dir = catalog_dirs(catalog) path, confine_to = resolve_sub_workflow_reference( - "builtin:test.json", file_spec, None + "builtin:test.json", base_dir, None ) assert path.startswith(builtin_root()) assert confine_to == builtin_root() @@ -69,47 +74,46 @@ def test_only_the_prefix_is_stripped_from_a_builtin_name(self, catalog): """A name carrying 'builtin:' again was read with it removed everywhere, which made 'builtin:builtin:test.json' load test.json; the second one is part of the name, which names no file.""" - _root, file_spec = catalog + _root, base_dir = catalog_dirs(catalog) with pytest.raises(SubWorkflowNotFound) as exc_info: - resolve_sub_workflow_reference("builtin:builtin:test.json", file_spec, None) + resolve_sub_workflow_reference("builtin:builtin:test.json", base_dir, None) assert "builtin:test.json" in str(exc_info.value) def test_a_missing_builtin_is_not_found(self, catalog): - _root, file_spec = catalog + _root, base_dir = catalog_dirs(catalog) with pytest.raises(SubWorkflowNotFound) as exc_info: - resolve_sub_workflow_reference("builtin:absent.json", file_spec, None) + resolve_sub_workflow_reference("builtin:absent.json", base_dir, None) assert "absent.json" in str(exc_info.value) @pytest.mark.parametrize("confined", [False, True]) def test_a_path_escaping_its_root_is_refused(self, catalog, confined): - root, file_spec = catalog + root, base_dir = catalog_dirs(catalog) with pytest.raises(PathTraversalError): resolve_sub_workflow_reference( - "../../Outside.json", file_spec, str(root) if confined else None + "../../Outside.json", base_dir, str(root) if confined else None ) class TestEverySiteReachesIt: - def test_create_step_action(self, catalog): + def test_create_step_action_reads_a_doubled_builtin_prefix_as_a_name(self, catalog): parent = parent_in(catalog) - step = {"name": "child", "workflow": {"path": "models/Child"}} + step = {"name": "c", "workflow": {"path": "builtin:builtin:test.json"}} - child = parent.create_step_action(step, {}, {}, 42, "cpu") - - assert child.name == "child" + with pytest.raises(SubWorkflowNotFound): + parent.create_step_action(step, {}, {}, 42, "cpu") - def test_create_step_action_refuses_a_climb_like_validation_does(self, catalog): - parent = parent_in(catalog, confined=True) - step = {"name": "child", "workflow": {"path": "../../Outside.json"}} + def test_create_step_action_names_a_missing_builtin(self, catalog): + parent = parent_in(catalog) + step = {"name": "c", "workflow": {"path": "builtin:absent.json"}} - with pytest.raises(PathTraversalError): + with pytest.raises(SubWorkflowNotFound) as exc_info: parent.create_step_action(step, {}, {}, 42, "cpu") - def test_open_sub_workflow(self, catalog): - root, _ = catalog - child, resolved = parent_in(catalog).open_sub_workflow("models/Child") - assert resolved == str(root / "models" / "Child.json") - assert child.name == "child" + assert builtin_root() in str(exc_info.value) + + def test_open_sub_workflow_reads_a_doubled_builtin_prefix_as_a_name(self, catalog): + with pytest.raises(SubWorkflowNotFound): + parent_in(catalog).open_sub_workflow("builtin:builtin:test.json") def test_open_sub_workflow_takes_an_already_resolved_path(self, catalog): parent = parent_in(catalog) @@ -120,60 +124,120 @@ def test_open_sub_workflow_takes_an_already_resolved_path(self, catalog): assert again == resolved assert child.name == "child" - def test_sub_workflow_errors_resolves_a_catalog_name(self, catalog): - parent = parent_in(catalog) - expanded = {"steps": [{"name": "c", "workflow": {"path": "models/Child"}}]} - assert sub_workflow_errors(parent, expanded) == [] - - def test_sub_workflow_errors_names_a_missing_builtin(self, catalog): - parent = parent_in(catalog) + def test_sub_workflow_errors_reads_a_doubled_builtin_prefix_as_a_name( + self, catalog + ): expanded = { - "steps": [{"name": "c", "workflow": {"path": "builtin:absent.json"}}] + "steps": [{"name": "c", "workflow": {"path": "builtin:builtin:test.json"}}] } - (error,) = sub_workflow_errors(parent, expanded) - assert "absent.json" in error["message"] + (error,) = sub_workflow_errors(parent_in(catalog), expanded) + assert "could not be resolved" in error["message"] + + def test_sub_workflow_errors_does_not_resolve_a_catalog_child_twice( + self, catalog, monkeypatch + ): + from dw import validation + + calls = [] + real = Workflow.resolve_sub_workflow_path + + def counting(self, path): + calls.append(path) + return real(self, path) + + monkeypatch.setattr(Workflow, "resolve_sub_workflow_path", counting) + expanded = {"steps": [{"name": "c", "workflow": {"path": "models/Child"}}]} + + assert validation.sub_workflow_errors(parent_in(catalog), expanded) == [] + assert calls == ["models/Child"] + + +class TestAnEscapeIsRefusedAtEverySite: + ESCAPE = "../../Outside.json" + + def test_create_step_action(self, catalog): + step = {"name": "c", "workflow": {"path": self.ESCAPE}} + with pytest.raises(PathTraversalError): + parent_in(catalog, confined=True).create_step_action( + step, {}, {}, 42, "cpu" + ) + + def test_open_sub_workflow(self, catalog): + with pytest.raises(PathTraversalError): + parent_in(catalog, confined=True).open_sub_workflow(self.ESCAPE) + + def test_sub_workflow_errors_carries_the_refusal(self, catalog): + root, file_spec = catalog + expanded = {"steps": [{"name": "c", "workflow": {"path": self.ESCAPE}}]} + with pytest.raises(PathTraversalError) as expected: + resolve_sub_workflow_reference( + self.ESCAPE, os.path.dirname(file_spec), str(root) + ) + + (error,) = sub_workflow_errors(parent_in(catalog, confined=True), expanded) + + assert error["message"] == str(expected.value) + + def test_read_sub_workflow_reads_none(self, catalog): + from dw.realize import read_sub_workflow + + root, file_spec = catalog + assert read_sub_workflow(self.ESCAPE, os.path.dirname(file_spec), None) is None + assert ( + read_sub_workflow(self.ESCAPE, os.path.dirname(file_spec), str(root)) + is None + ) + + def test_the_observed_cost_lookup_finds_none(self, catalog, monkeypatch): + observed_for_child = child_lookup(catalog, monkeypatch) + assert observed_for_child(self.ESCAPE, CHILD) is None + + +def child_lookup(catalog, monkeypatch): + """The `observed_for_child` callback `_validation_plan` hands the planner, + with the observed figure stubbed to the catalog name it was asked for.""" + from types import SimpleNamespace + from dw.server.routes import jobs + + root, file_spec = catalog + seen = {} + monkeypatch.setattr( + jobs, "build_plan", lambda *a, **kw: seen.update(kw) or {"plan": True} + ) + monkeypatch.setattr( + jobs, "observed_for_name", lambda state, name, *a, **kw: ("observed", name) + ) + candidate = SimpleNamespace( + file_spec=file_spec, workflow_dir=None, workflow_definition={"steps": []} + ) + request = SimpleNamespace(workflow_path=None, arguments={}) + workspace = SimpleNamespace( + name="default", outputs=None, assets=None, prompts=None, workflows=str(root) + ) + jobs._validation_plan( + SimpleNamespace(job_manager=None), + candidate, + request, + workspace, + None, + None, + False, + ) + return seen["observed_for_child"] + + +class TestRealizationAndCostReachIt: def test_read_sub_workflow_falls_back_to_the_catalog_root(self, catalog): from dw.realize import read_sub_workflow _root, file_spec = catalog - raw = read_sub_workflow("models/Child", str(file_spec).rsplit("/", 1)[0], None) + raw = read_sub_workflow("models/Child", os.path.dirname(file_spec), None) assert json.loads(raw) == CHILD def test_the_observed_cost_lookup_falls_back_to_the_catalog_root( self, catalog, monkeypatch ): - from types import SimpleNamespace - - from dw.server.routes import jobs - - root, file_spec = catalog - seen = {} - monkeypatch.setattr( - jobs, "build_plan", lambda *a, **kw: seen.update(kw) or {"plan": True} - ) - monkeypatch.setattr( - jobs, - "observed_for_name", - lambda state, name, *a, **kw: ("observed", name), - ) - candidate = SimpleNamespace( - file_spec=file_spec, workflow_dir=None, workflow_definition={"steps": []} - ) - request = SimpleNamespace(workflow_path=None, arguments={}) - workspace = SimpleNamespace( - name="default", outputs=None, assets=None, prompts=None, workflows=str(root) - ) - jobs._validation_plan( - SimpleNamespace(job_manager=None), - candidate, - request, - workspace, - None, - None, - False, - ) - - found = seen["observed_for_child"]("models/Child", CHILD) + found = child_lookup(catalog, monkeypatch)("models/Child", CHILD) assert found == ("observed", "models/Child") From 8c7f5eb53b27473fd5805d54379f7d797cc6635e Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 12:32:22 -0500 Subject: [PATCH 06/41] fix: reset the run directory per run, sort kernels variant lines in dw, reword argument_template description Co-Authored-By: Claude Opus 5.5 --- docs/RELEASING.md | 3 +++ dw/kernel_availability.py | 19 +++++++++++++++++- dw/workflow.py | 7 +++++++ dw/workflow_schema.json | 2 +- scripts/surface_snapshot.py | 23 +-------------------- tests/test_kernel_availability.py | 33 +++++++++++++++++++++++++++++++ tests/test_runs.py | 27 +++++++++++++++++++++++++ 7 files changed, 90 insertions(+), 24 deletions(-) diff --git a/docs/RELEASING.md b/docs/RELEASING.md index dceaceec..0e895944 100644 --- a/docs/RELEASING.md +++ b/docs/RELEASING.md @@ -13,6 +13,9 @@ when a release ships. - `builtin:builtin:x.json` no longer loads `x.json`: only the leading prefix is stripped, so the name `builtin:x.json` is looked up and reported as `SubWorkflowNotFound`. - At run time a missing `builtin:` workflow now raises `SubWorkflowNotFound` (naming the packaged root) instead of `validate_workflow_path`'s missing-file error. - The realized workflow's sub-workflow digest and a composed child's observed cost now fall back to the same catalog root, so a catalog sub-workflow a run could open is also digested and costed. +- A run that fails before it opens its run directory no longer rewrites the previous run's `manifest.json` when the same workflow instance is reused: `Workflow.run` resets the directory and version it carried. +- The per-variant lines of a kernels "Cannot find a build variant" error are sorted by dw (`kernel_availability.stable_message`), so the message no longer varies by process. +- The `argument_template` schema description now says what the code does: handed arguments are held on the child at run time, never written into the definition, and an authored value is the fallback. ### 0.6.0 diff --git a/dw/kernel_availability.py b/dw/kernel_availability.py index c311cf8c..69d4ff94 100644 --- a/dw/kernel_availability.py +++ b/dw/kernel_availability.py @@ -91,6 +91,21 @@ def _requires_remote_kernel(attn_processor_class): return "get_kernel(" in source +def stable_message(message): + """Sorts the per-variant lines of a kernels-hub "Cannot find a build + variant" error, which the hub library lists in set order (different on + every process). The lines are sorted in place, so their position against + the rest of the message still counts; any other message passes through + untouched.""" + if "Cannot find a build variant" not in message: + return message + lines = message.split("\n") + slots = [i for i, line in enumerate(lines) if line.startswith("torch")] + for i, line in zip(slots, sorted(lines[i] for i in slots)): + lines[i] = line + return "\n".join(lines) + + class _KernelFault(Exception): """Carries a fault message out of `_fault_for_name` without letting `lru_cache` memoize it - see that function's docstring.""" @@ -123,7 +138,9 @@ def _fault_for_name(value): try: attn_processor_class() except Exception as e: - raise _KernelFault(f"'{value}' {KERNEL_FAULT_MARKER}: {e}") from e + raise _KernelFault( + f"'{value}' {KERNEL_FAULT_MARKER}: {stable_message(str(e))}" + ) from e return None diff --git a/dw/workflow.py b/dw/workflow.py index c2d5a0a7..9a31441d 100644 --- a/dw/workflow.py +++ b/dw/workflow.py @@ -542,6 +542,13 @@ def run( # What elision dropped this run, filled by prepare_definition and # read by the warning pass and the manifest (#122) self._elided_steps = [] + # A reused Workflow (the persistent worker's) still holds the last + # run's directory: a run that fails before open_run would otherwise + # rewrite that run's manifest. A composed child's values were set by + # its parent just before this call, so they stay + if not self._run_dir_inherited: + self._run_dir = None + self._run_version = None record = workflow_run.RunRecord(arguments) try: prepared = workflow_run.prepare_run(self, record) diff --git a/dw/workflow_schema.json b/dw/workflow_schema.json index d3e5e57d..6f5a66c3 100644 --- a/dw/workflow_schema.json +++ b/dw/workflow_schema.json @@ -104,7 +104,7 @@ "format": "int64" }, "argument_template": { - "description": "Engine-injected: the arguments a parent workflow passed to this one when it ran it as a sub-workflow. Written by create_step_action from the step's 'arguments' block, not authored - a workflow file carrying one is read, but a sub-workflow step is how they are meant to be supplied.", + "description": "Engine-injected: the arguments a parent workflow hands this one when it runs it as a sub-workflow. At run time _sub_workflow_action holds them on the child workflow, and they are never written into the definition. A value authored here is read as the fallback when nothing was handed.", "type": "object" }, "steps": { diff --git a/scripts/surface_snapshot.py b/scripts/surface_snapshot.py index f365af69..092b5843 100644 --- a/scripts/surface_snapshot.py +++ b/scripts/surface_snapshot.py @@ -210,27 +210,6 @@ def workflow_schema(): ) -def stable_message(value): - """Sorts the per-variant lines of a kernels-hub "Cannot find a build - variant" error, which the hub library lists in set order (different on - every process). The lines are sorted in place, so their position against - the rest of the message still counts; any other string passes through - untouched.""" - if isinstance(value, str): - if "Cannot find a build variant" not in value: - return value - lines = value.split("\n") - slots = [i for i, line in enumerate(lines) if line.startswith("torch")] - for i, line in zip(slots, sorted(lines[i] for i in slots)): - lines[i] = line - return "\n".join(lines) - if isinstance(value, list): - return [stable_message(item) for item in value] - if isinstance(value, dict): - return {key: stable_message(item) for key, item in value.items()} - return value - - def catalog_validation(root): """`validation_errors()` and the warning checks that need no server state (no `ceiling_index`, no observed costs) for every JSON under `workflows/` @@ -259,7 +238,7 @@ def catalog_validation(root): if name == "validation_errors" else validation.run_warning_check(workflow, name, None) ) - entry[name] = stable_message(found) + entry[name] = found except Exception as error: entry[name] = f"error: {type(error).__name__}: {error}" verdicts[path.relative_to(repo).as_posix()] = entry diff --git a/tests/test_kernel_availability.py b/tests/test_kernel_availability.py index 599c5c3b..929da34d 100644 --- a/tests/test_kernel_availability.py +++ b/tests/test_kernel_availability.py @@ -31,6 +31,16 @@ def _get_kernel_that_fails(repo_id): raise FileNotFoundError("no build variant for this torch/CUDA combination") +def _get_kernel_that_lists_variants(repo_id): + raise FileNotFoundError( + "Cannot find a build variant for this system\n" + "torch29-cxx11-cu126-x86_64-linux\n" + "torch211-cxx11-cu126-x86_64-linux\n" + "torch210-cxx11-cu128-x86_64-linux\n" + "Available variants are listed above" + ) + + def _get_kernel_that_works(repo_id): return object() @@ -102,6 +112,29 @@ def test_a_kernel_backed_processor_that_fails_to_construct_is_faulted( assert "pkg.Natten" in fault assert "no build variant" in fault + def test_the_variant_lines_of_a_build_variant_error_come_out_sorted( + self, monkeypatch + ): + class _ListsVariants: + def __init__(self): + get_kernel = _get_kernel_that_lists_variants + get_kernel("shi-labs/natten") + + monkeypatch.setattr( + kernel_availability, + "load_type_from_name", + _resolving({"pkg.Natten": _ListsVariants}), + ) + fault = kernel_availability_fault("pkg.Natten") + lines = fault.split("\n") + assert lines[1:4] == [ + "torch210-cxx11-cu128-x86_64-linux", + "torch211-cxx11-cu126-x86_64-linux", + "torch29-cxx11-cu126-x86_64-linux", + ] + assert lines[0].endswith("Cannot find a build variant for this system") + assert lines[4] == "Available variants are listed above" + def test_a_kernel_backed_processor_that_constructs_cleanly_is_not_faulted( self, monkeypatch ): diff --git a/tests/test_runs.py b/tests/test_runs.py index a01f4d76..6c98e9de 100644 --- a/tests/test_runs.py +++ b/tests/test_runs.py @@ -460,6 +460,33 @@ def test_a_seed_can_come_from_a_variable(self, tmp_path, fake_pipeline): manifest = json.loads((run_dir / "manifest.json").read_text()) assert manifest["seed"] == 1234 + def test_a_run_that_fails_before_opening_leaves_the_previous_manifest_alone( + self, tmp_path, fake_pipeline + ): + import dw.workflow as workflow_module + from dw.workflow import Workflow + + workflow = Workflow( + _workflow_definition(), str(tmp_path), "/w/workflows/Gyre.json" + ) + workflow.run({}) + manifest_path = next((tmp_path / "Gyre").iterdir()) / "manifest.json" + before = manifest_path.read_bytes() + + # A persistent worker reuses one Workflow across jobs: a second run + # that fails in prepare_run, before it opens a directory of its own, + # must not find the first run's directory still set and rewrite its + # manifest from the finally block + with patch.object( + workflow_module.workflow_run, + "prepare_run", + side_effect=RuntimeError("refused"), + ): + with pytest.raises(RuntimeError): + workflow.run({}) + + assert manifest_path.read_bytes() == before + def test_a_failed_run_still_records_what_it_wrote(self, tmp_path, fake_pipeline): from dw.workflow import Workflow From 7f68454234bf4aafa3c5387a6b93e70f84f0d3d9 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 12:33:29 -0500 Subject: [PATCH 07/41] docs(schema): argument_template is read only outside a sub-workflow run Co-Authored-By: Claude Opus 5.5 --- dw/workflow_schema.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/dw/workflow_schema.json b/dw/workflow_schema.json index 6f5a66c3..28ae39f8 100644 --- a/dw/workflow_schema.json +++ b/dw/workflow_schema.json @@ -104,7 +104,7 @@ "format": "int64" }, "argument_template": { - "description": "Engine-injected: the arguments a parent workflow hands this one when it runs it as a sub-workflow. At run time _sub_workflow_action holds them on the child workflow, and they are never written into the definition. A value authored here is read as the fallback when nothing was handed.", + "description": "Engine-injected: the arguments a parent workflow hands this one when it runs it as a sub-workflow. At run time _sub_workflow_action holds them on the child workflow, and they are never written into the definition. A value authored here is read only when this workflow is not run as a sub-workflow.", "type": "object" }, "steps": { From 31463f5883cb7c0cc6f7abede47a2df233825215 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 12:36:30 -0500 Subject: [PATCH 08/41] fix(audio): gain_audio rounds a frame region once; joins refuse a track with no sample rate gain_audio's frame region goes through slice_region, so its end matches slice_audio's to the sample. reconcile_sample_rates loses skip_unrated and refuses an unrated track by name in concat_videos and dissolve_videos. Co-Authored-By: Claude Opus 5.5 --- docs/RELEASING.md | 2 ++ dw/tasks/audio_utils.py | 11 ++++++----- dw/tasks/dissolve_videos.py | 1 - dw/tasks/joins.py | 23 ++++++++++++----------- tests/test_audio_utils.py | 25 +++++++++++++++++++++++++ tests/test_concat_videos.py | 16 ++++++++++++++++ tests/test_dissolve_videos.py | 20 +++++++++++--------- 7 files changed, 72 insertions(+), 26 deletions(-) diff --git a/docs/RELEASING.md b/docs/RELEASING.md index 0e895944..c56c5764 100644 --- a/docs/RELEASING.md +++ b/docs/RELEASING.md @@ -16,6 +16,8 @@ when a release ships. - A run that fails before it opens its run directory no longer rewrites the previous run's `manifest.json` when the same workflow instance is reused: `Workflow.run` resets the directory and version it carried. - The per-variant lines of a kernels "Cannot find a build variant" error are sorted by dw (`kernel_availability.stable_message`), so the message no longer varies by process. - The `argument_template` schema description now says what the code does: handed arguments are held on the child at run time, never written into the definition, and an authored value is the fallback. +- `gain_audio` rounds a frame-addressed region's end once, as `slice_audio` does, so a region can no longer end one sample short of the matching slice. +- `concat_videos` refuses a track with no sample rate (`concat_videos: '' has audio with no sample rate`) instead of joining it unresampled at the wrong speed and pitch; `dissolve_videos` gives the same message in place of the resample error. Set `audio_sample_rate` in the result of the step that made it. ### 0.6.0 diff --git a/dw/tasks/audio_utils.py b/dw/tasks/audio_utils.py index ce6145fa..d3357f35 100644 --- a/dw/tasks/audio_utils.py +++ b/dw/tasks/audio_utils.py @@ -336,11 +336,12 @@ def gain_audio( elif start_frame is not None or num_frames is not None: if fps is None: raise ValueError("gain_audio needs 'fps' to address a region in frames") - start = frames_to_samples(start_frame or 0, fps, sample_rate) - length = ( - max(total - start, 0) - if num_frames is None - else frames_to_samples(num_frames, fps, sample_rate) + start, length = slice_region( + sample_rate, + start_frame=start_frame, + num_frames=num_frames, + fps=fps, + total=total, ) else: start = 0 diff --git a/dw/tasks/dissolve_videos.py b/dw/tasks/dissolve_videos.py index c7a82190..de2f0460 100644 --- a/dw/tasks/dissolve_videos.py +++ b/dw/tasks/dissolve_videos.py @@ -300,7 +300,6 @@ def _dissolve_audio( track_names, [as_channels_samples(v.audio) for v in tracks], sample_rate, - skip_unrated=False, ) crossfade_ms = dissolve_frames / fps * 1000 if dissolve_frames else 0 waveforms = level_waveforms( diff --git a/dw/tasks/joins.py b/dw/tasks/joins.py index 98c1545c..b0f77579 100644 --- a/dw/tasks/joins.py +++ b/dw/tasks/joins.py @@ -67,9 +67,7 @@ def load_named_inputs(videos): return names, [load_audio_video(v) if is_video_location(v) else v for v in videos] -def reconcile_sample_rates( - command, videos, names, waveforms, sample_rate=None, skip_unrated=True -): +def reconcile_sample_rates(command, videos, names, waveforms, sample_rate=None): """One rate for every track about to be joined: (waveforms, sample_rate). `videos` and `waveforms` run in step, and a None waveform is an input @@ -79,15 +77,20 @@ def reconcile_sample_rates( resample_audio step by hand (#108, #287). The highest rate among the inputs is the default target; the caller's `sample_rate` pins another. - A track whose rate is unknown (a pipeline that reported none) is passed - over when `skip_unrated` (concat_videos' long-standing rule). Otherwise - its missing rate takes part like any other, so it cannot be mistaken for - the target: dissolve_videos fails on it rather than joining it unscaled. + A track whose rate is unknown (a pipeline that reported none) is refused + by name, in either command: joined unscaled it would play at the wrong + speed and pitch, and there is no rate to convert it from. """ + for name, video, waveform in zip(names, videos, waveforms): + if waveform is not None and not video.sample_rate: + raise ValueError( + f"{command}: '{name}' has audio with no sample rate - set " + "'audio_sample_rate' in the result of the step that made it" + ) rates = [ video.sample_rate for video, waveform in zip(videos, waveforms) - if waveform is not None and (video.sample_rate or not skip_unrated) + if waveform is not None ] sample_rate = sample_rate or (max(set(rates)) if rates else None) if rates and len(set(rates)) == 1 and rates[0] != sample_rate: @@ -124,9 +127,7 @@ def reconcile_sample_rates( return [ ( waveform - if waveform is None - or (skip_unrated and not video.sample_rate) - or video.sample_rate == sample_rate + if waveform is None or video.sample_rate == sample_rate else resample_waveform(waveform, video.sample_rate, sample_rate) ) for video, waveform in zip(videos, waveforms) diff --git a/tests/test_audio_utils.py b/tests/test_audio_utils.py index 7ca0a129..e1ca11fe 100644 --- a/tests/test_audio_utils.py +++ b/tests/test_audio_utils.py @@ -581,6 +581,31 @@ def test_the_applied_gain_is_logged(self): assert logs[0]["duration_seconds"] == pytest.approx(0.5) assert logs[0]["sample_rate"] == 100 + def test_a_frame_region_ends_where_slice_audio_ends(self): + # The end of a frame-addressed region is rounded once (#557); two + # separately rounded halves land a sample off at 24 fps / 44.1 kHz + from dw.tasks.audio_utils import gain_audio + + rate, fps, start_frame, num_frames = 44100, 24, 1, 1 + start = frames_to_samples(start_frame, fps, rate) + end = frames_to_samples(start_frame + num_frames, fps, rate) + assert start + frames_to_samples(num_frames, fps, rate) != end + + track = numpy.ones((1, rate), dtype=numpy.float32) + gained = samples( + gain_audio( + track, + gain_db=-6.0, + start_frame=start_frame, + num_frames=num_frames, + fps=fps, + sample_rate=rate, + ) + ) + + changed = numpy.flatnonzero(~numpy.isclose(gained[:, 0], 1.0)) + assert (changed[0], changed[-1] + 1) == (start, end) + def test_no_region_gains_the_whole_track(self): # #395: validate_workflow let a region-less gain_audio step through # clean and the run then failed - the fix is to gain everything, diff --git a/tests/test_concat_videos.py b/tests/test_concat_videos.py index 01c1b56b..8cbba5a8 100644 --- a/tests/test_concat_videos.py +++ b/tests/test_concat_videos.py @@ -111,6 +111,22 @@ def test_the_resample_warning_names_which_video(self, caplog): assert "video 2: 200 Hz" in caplog.text assert "resample_audio" in caplog.text + def test_a_track_with_no_sample_rate_is_refused_by_name(self): + """Joined unscaled it would play at the wrong speed and pitch; the + refusal names the input and the remedy the generation warning gives.""" + unrated = audio_video(8, 1) + unrated.sample_rate = None + videos = [unrated, audio_video(8, 2, sample_rate=200)] + + with pytest.raises(ValueError) as raised: + concat_videos(videos) + + message = str(raised.value) + assert message.startswith( + "concat_videos: 'video 1' has audio with no sample rate" + ) + assert "audio_sample_rate" in message + def test_an_explicit_sample_rate_pins_the_target(self): videos = [audio_video(8, 1), audio_video(8, 2, sample_rate=200)] diff --git a/tests/test_dissolve_videos.py b/tests/test_dissolve_videos.py index 87e838bb..520f3e52 100644 --- a/tests/test_dissolve_videos.py +++ b/tests/test_dissolve_videos.py @@ -89,17 +89,21 @@ def test_a_single_shortfall_keeps_its_message(self): def test_a_track_with_no_sample_rate_is_not_joined_silently(self): """A pipeline that reports no rate leaves AudioVideo.sample_rate None. - Dissolve has never guessed one: unpinned, the target cannot be chosen - (TypeError from max); pinned, the mismatch is warned and the resample - refuses the unrated track.""" + Neither join guesses one: the track is refused by name, with the + remedy the generation-time warning gives.""" videos = [ unrated_video(), audio_video(8, 0, 1.0, sample_rate=32000), ] - with pytest.raises(TypeError): + with pytest.raises(ValueError) as raised: dissolve_videos(videos, 2, fps=4) - def test_a_pinned_rate_warns_and_then_refuses_an_unrated_track(self): + assert str(raised.value).startswith( + "dissolve_videos: 'video 1' has audio with no sample rate" + ) + assert "audio_sample_rate" in str(raised.value) + + def test_a_pinned_rate_still_refuses_an_unrated_track(self): from dw.events import RunContext, activate_context, deactivate_context videos = [ @@ -109,14 +113,12 @@ def test_a_pinned_rate_warns_and_then_refuses_an_unrated_track(self): events = [] token = activate_context(RunContext(on_event=events.append)) try: - with pytest.raises(ValueError, match="sample_rate above zero"): + with pytest.raises(ValueError, match="has audio with no sample rate"): dissolve_videos(videos, 2, fps=4, sample_rate=48000) finally: deactivate_context(token) - assert [e.get("kind") for e in events if e.get("event") == "warning"] == [ - "sample_rate_mismatch" - ] + assert [e for e in events if e.get("event") == "warning"] == [] def test_negative_counts_are_refused(self): with pytest.raises(ValueError, match="negative"): From 18053f54bbeaa08a0ab5a48936c09bdaa43c9bde Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 12:37:37 -0500 Subject: [PATCH 09/41] fix(audio): the unrated-track refusal names a remedy that works A declared audio_sample_rate applies at save, not to an in-memory input, so the message now says to save the step and join the file through output:. Co-Authored-By: Claude Opus 5.5 --- docs/RELEASING.md | 2 +- dw/tasks/joins.py | 6 ++++-- tests/test_concat_videos.py | 2 +- tests/test_dissolve_videos.py | 2 +- 4 files changed, 7 insertions(+), 5 deletions(-) diff --git a/docs/RELEASING.md b/docs/RELEASING.md index c56c5764..84793b1c 100644 --- a/docs/RELEASING.md +++ b/docs/RELEASING.md @@ -17,7 +17,7 @@ when a release ships. - The per-variant lines of a kernels "Cannot find a build variant" error are sorted by dw (`kernel_availability.stable_message`), so the message no longer varies by process. - The `argument_template` schema description now says what the code does: handed arguments are held on the child at run time, never written into the definition, and an authored value is the fallback. - `gain_audio` rounds a frame-addressed region's end once, as `slice_audio` does, so a region can no longer end one sample short of the matching slice. -- `concat_videos` refuses a track with no sample rate (`concat_videos: '' has audio with no sample rate`) instead of joining it unresampled at the wrong speed and pitch; `dissolve_videos` gives the same message in place of the resample error. Set `audio_sample_rate` in the result of the step that made it. +- `concat_videos` refuses a track with no sample rate (`concat_videos: '' has audio with no sample rate`) instead of joining it unresampled at the wrong speed and pitch; `dissolve_videos` gives the same message in place of the resample error. Save that step with `audio_sample_rate` in its result and join the saved file through an `output:` reference. An unpinned `dissolve_videos` with such a track now raises this `ValueError` rather than a `TypeError`. ### 0.6.0 diff --git a/dw/tasks/joins.py b/dw/tasks/joins.py index b0f77579..3fca16c1 100644 --- a/dw/tasks/joins.py +++ b/dw/tasks/joins.py @@ -84,8 +84,10 @@ def reconcile_sample_rates(command, videos, names, waveforms, sample_rate=None): for name, video, waveform in zip(names, videos, waveforms): if waveform is not None and not video.sample_rate: raise ValueError( - f"{command}: '{name}' has audio with no sample rate - set " - "'audio_sample_rate' in the result of the step that made it" + f"{command}: '{name}' has audio with no sample rate (the " + "pipeline that made it reported none) - save that step with " + "'audio_sample_rate' in its result and join the saved file " + "through an output: reference" ) rates = [ video.sample_rate diff --git a/tests/test_concat_videos.py b/tests/test_concat_videos.py index 8cbba5a8..5f878feb 100644 --- a/tests/test_concat_videos.py +++ b/tests/test_concat_videos.py @@ -125,7 +125,7 @@ def test_a_track_with_no_sample_rate_is_refused_by_name(self): assert message.startswith( "concat_videos: 'video 1' has audio with no sample rate" ) - assert "audio_sample_rate" in message + assert "join the saved file through an output: reference" in message def test_an_explicit_sample_rate_pins_the_target(self): videos = [audio_video(8, 1), audio_video(8, 2, sample_rate=200)] diff --git a/tests/test_dissolve_videos.py b/tests/test_dissolve_videos.py index 520f3e52..542733d5 100644 --- a/tests/test_dissolve_videos.py +++ b/tests/test_dissolve_videos.py @@ -101,7 +101,7 @@ def test_a_track_with_no_sample_rate_is_not_joined_silently(self): assert str(raised.value).startswith( "dissolve_videos: 'video 1' has audio with no sample rate" ) - assert "audio_sample_rate" in str(raised.value) + assert "through an output: reference" in str(raised.value) def test_a_pinned_rate_still_refuses_an_unrated_track(self): from dw.events import RunContext, activate_context, deactivate_context From aa9cc04f642e1db10d4052244dba397809269761 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 12:41:17 -0500 Subject: [PATCH 10/41] metrics: add prefix_handling ratchet (baseline 82) Counts hand-written reference-prefix handling outside dw/references.py: startswith-style calls, len() slices, + / f-string building, module-level alias assignments and prefixed strings. prefix_literals is unchanged. Co-Authored-By: Claude Opus 5.5 --- docs/stabilization/ROADMAP.md | 2 + docs/stabilization/baseline.json | 1 + scripts/arch_metrics.py | 212 +++++++++++++++++++++++++++++++ tests/test_arch_metrics.py | 63 +++++++++ 4 files changed, 278 insertions(+) diff --git a/docs/stabilization/ROADMAP.md b/docs/stabilization/ROADMAP.md index 6f36d13f..71df3aa8 100644 --- a/docs/stabilization/ROADMAP.md +++ b/docs/stabilization/ROADMAP.md @@ -27,6 +27,8 @@ Two kinds, kept small on purpose. - **Cyclomatic complexity:** functions above 15, by ruff's C901 (already a dev dependency). Baseline 21. - **Import cycles:** strongly connected components of more than one module in grimp's graph of `dw` + `dw_mcp`, lazy imports included, TYPE_CHECKING imports excluded. Baseline 6. - **Modules inside import cycles:** the sum of those components' sizes. Baseline 26, because one of the six is a 16-module knot (`arguments`, `result`, `pipeline`, `runs`, `step_cache`, `tasks`, ...) that could grow without the cycle count moving. It grew from 22 to 26 in Phase 0 (the `step_cache` → `pipeline` import). The target is 0 for both cycle ratchets, reached by Phases 2-3. Each ratchet only forbids getting worse. + - **Prefix handling (Phase 4a):** hand-written reference-prefix handling outside `references.py`, per AST node: `startswith`-style calls, `[len(PREFIX):]` slices, `+` and f-string building, module-level alias assignments, and strings that start or end with a prefix. Counting rules are in the `scripts/arch_metrics.py` docstring. Baseline 82; Phase 4a drives it to 0. + - `prefix_literals` stays beside it (exact bare prefix strings, baseline 0). **Gate reports.** These are produced at each phase gate by `scripts/arch_report.py`, which lands in Phase 1. They are written into the Gate reports section below. They are reports, never gates. diff --git a/docs/stabilization/baseline.json b/docs/stabilization/baseline.json index 79cc266b..8bdb1893 100644 --- a/docs/stabilization/baseline.json +++ b/docs/stabilization/baseline.json @@ -3,6 +3,7 @@ "modules_over_1000_lines": 0, "functions_over_150_lines": 0, "prefix_literals": 0, + "prefix_handling": 82, "test_dw_patch_targets": 284, "claude_md_lines": 957, "duplicate_blocks": 5, diff --git a/scripts/arch_metrics.py b/scripts/arch_metrics.py index 7cd83e08..f654e147 100644 --- a/scripts/arch_metrics.py +++ b/scripts/arch_metrics.py @@ -14,6 +14,20 @@ count, TYPE_CHECKING imports do not. Folders without an __init__.py (dw/tasks, dw/pipeline_processors) are named to grimp explicitly, since it walks only regular packages. +- prefix_literals counts a string constant that is exactly a reference + prefix, outside PREFIX_OWNERS. +- prefix_handling counts hand-written handling of a reference prefix, outside + PREFIX_OWNERS, per AST node. A "reference expression" is a constant of + dw/references.py (parsed from the measured root; a fixed list when the file + is absent) as `.NAME` (any import form), as a name imported + from references (with `as`), as a module-level name assigned from either, + or a name imported from a dw module that ends in _PREFIX or _PREFIXES. + Forms: (a) a startswith/removeprefix/removesuffix/replace/split/partition + call with one as an argument; (b) a slice starting at len(); (c) `+` + with one as an operand, or an f-string splicing one in; (d) a module-level + assignment of one (or a tuple holding one); (e) a non-docstring string that + starts with a prefix and is longer than it, or an f-string fragment that + ends with a prefix. Prose naming a bare prefix is not counted. """ import argparse @@ -43,6 +57,14 @@ PREFIX_OWNERS = frozenset({"dw/references.py"}) EXCLUDED = ("community_pipelines", "node_modules", "venv", ".git") PACKAGES = ("dw", "dw_mcp") +REFERENCES_MODULE = "dw/references.py" +FALLBACK_REFERENCE_NAMES = frozenset( + "ASSET OUTPUT PROMPT VARIABLE PREVIOUS_RESULT CONSTANT ITEM GATHER BUILTIN " + "CONSTRAINT SUBSTITUTED UNRESOLVED DEFERRED".split() +) +PREFIX_METHODS = frozenset( + ("startswith", "removeprefix", "removesuffix", "replace", "split", "partition") +) PATCH_TARGET = re.compile(r"""patch\(\s*["']dw[._]""") COMPLEXITY_LIMIT = 15 COMPLEXITY_MESSAGE = re.compile(r"^`(?P.+)` is too complex \((?P\d+) > 0\)$") @@ -181,6 +203,190 @@ def import_graph(root): return json.loads(result.stdout) +def _reference_names(root): + """(constant names, prefix strings) defined by dw/references.py under root: + a single prefix is an upper-case name bound to a string ending in a colon; + a tuple is one built from those names (and other tuples).""" + path = root / REFERENCES_MODULE + if not path.is_file(): + return FALLBACK_REFERENCE_NAMES, REFERENCE_PREFIXES + assigns = [ + (node.targets[0].id, node.value) + for node in ast.parse(path.read_text(encoding="utf-8")).body + if isinstance(node, ast.Assign) + and len(node.targets) == 1 + and isinstance(node.targets[0], ast.Name) + and node.targets[0].id.isupper() + ] + singles = { + name: value.value + for name, value in assigns + if isinstance(value, ast.Constant) + and isinstance(value.value, str) + and value.value.endswith(":") + } + names = set(singles) + + def built(node): + if isinstance(node, ast.Name): + return node.id in names + if isinstance(node, ast.Tuple): + return any(built(element) for element in node.elts) + if isinstance(node, ast.BinOp) and isinstance(node.op, ast.Add): + return built(node.left) or built(node.right) + return False + + grown = True + while grown: + grown = False + for name, value in assigns: + if name not in names and built(value): + names.add(name) + grown = True + return frozenset(names), frozenset(singles.values()) + + +def _dotted(node): + parts = [] + while isinstance(node, ast.Attribute): + parts.append(node.attr) + node = node.value + if isinstance(node, ast.Name): + parts.append(node.id) + return ".".join(reversed(parts)) + return None + + +class _Resolver: + """What counts as a reference expression in one module: the module's own + imports of references (any form), its bare imports from it, names imported + from a dw module that end in _PREFIX / _PREFIXES, and module-level names + assigned from any of those.""" + + def __init__(self, tree, constants): + self.constants = constants + self.modules = set() + self.names = set() + for node in ast.walk(tree): + if isinstance(node, ast.Import): + for alias in node.names: + if alias.name == "dw.references": + self.modules.add(alias.asname or alias.name) + elif isinstance(node, ast.ImportFrom): + self._from_import(node) + grown = True + while grown: + grown = False + for node in tree.body: + target, value = _assignment(node) + if ( + target is not None + and target not in self.names + and self.is_reference(value) + ): + self.names.add(target) + grown = True + + def _from_import(self, node): + in_dw = node.level > 0 or (node.module or "").split(".")[0] == "dw" + for alias in node.names: + bound = alias.asname or alias.name + if alias.name == "references" and node.module in (None, "dw"): + self.modules.add(bound) + elif (node.module or "").split(".")[-1] == "references" and in_dw: + if alias.name in self.constants: + self.names.add(bound) + elif in_dw and alias.name.endswith(("_PREFIX", "_PREFIXES")): + self.names.add(bound) + + def is_reference(self, node): + if isinstance(node, ast.Name): + return node.id in self.names + if isinstance(node, ast.Attribute): + return node.attr in self.constants and _dotted(node.value) in self.modules + if isinstance(node, ast.Tuple): + return any(self.is_reference(element) for element in node.elts) + return False + + +def _assignment(node): + if isinstance(node, ast.Assign) and len(node.targets) == 1: + target, value = node.targets[0], node.value + elif isinstance(node, ast.AnnAssign) and node.value is not None: + target, value = node.target, node.value + else: + return None, None + return (target.id if isinstance(target, ast.Name) else None), value + + +def prefix_handling_sites(tree, constants, prefixes): + """Every hand-written handling of a reference prefix in one parsed module, + as (line, form) with form one of "a".."e".""" + resolver = _Resolver(tree, constants) + refers = resolver.is_reference + docstrings = { + id(node.body[0].value) + for node in ast.walk(tree) + if isinstance( + node, (ast.Module, ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef) + ) + and node.body + and isinstance(node.body[0], ast.Expr) + and isinstance(node.body[0].value, ast.Constant) + } + fragments = { + id(part) + for node in ast.walk(tree) + if isinstance(node, ast.JoinedStr) + for part in node.values + } + found = [] + for node in tree.body: + target, value = _assignment(node) + if target is not None and refers(value): + found.append((node.lineno, "d")) + for node in ast.walk(tree): + if isinstance(node, ast.Call): + if ( + isinstance(node.func, ast.Attribute) + and node.func.attr in PREFIX_METHODS + and any(refers(arg) for arg in node.args) + ): + found.append((node.lineno, "a")) + elif isinstance(node, ast.Subscript): + lower = node.slice.lower if isinstance(node.slice, ast.Slice) else None + if ( + isinstance(lower, ast.Call) + and isinstance(lower.func, ast.Name) + and lower.func.id == "len" + and len(lower.args) == 1 + and refers(lower.args[0]) + ): + found.append((node.lineno, "b")) + elif isinstance(node, ast.BinOp): + if isinstance(node.op, ast.Add) and ( + refers(node.left) or refers(node.right) + ): + found.append((node.lineno, "c")) + elif isinstance(node, ast.JoinedStr): + if any( + isinstance(part, ast.FormattedValue) and refers(part.value) + for part in node.values + ): + found.append((node.lineno, "c")) + elif ( + isinstance(node, ast.Constant) + and isinstance(node.value, str) + and id(node) not in docstrings + ): + text = node.value + if any(text.startswith(p) and len(text) > len(p) for p in prefixes) or ( + id(node) in fragments and text.endswith(tuple(prefixes)) + ): + found.append((node.lineno, "e")) + return found + + def measure(root): root = pathlib.Path(root) engine = list(_sources(root, *PACKAGES)) @@ -189,12 +395,18 @@ def measure(root): "modules_over_1000_lines": 0, "functions_over_150_lines": 0, "prefix_literals": 0, + "prefix_handling": 0, } + constants, prefixes = _reference_names(root.resolve()) for path in engine: text = path.read_text(encoding="utf-8") if len(text.splitlines()) > 1000: metrics["modules_over_1000_lines"] += 1 tree = ast.parse(text) + if path.relative_to(root).as_posix() not in PREFIX_OWNERS: + metrics["prefix_handling"] += len( + prefix_handling_sites(tree, constants, prefixes) + ) for node in ast.walk(tree): if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)): if node.end_lineno - node.lineno > 150: diff --git a/tests/test_arch_metrics.py b/tests/test_arch_metrics.py index 906d8cf6..a1652ddd 100644 --- a/tests/test_arch_metrics.py +++ b/tests/test_arch_metrics.py @@ -181,3 +181,66 @@ def test_the_prefix_owner_may_spell_a_prefix(tmp_path): ) ) assert metrics["prefix_literals"] == 1 + + +_REFERENCES = 'ASSET = "asset:"\nVARIABLE = "variable:"\nSUBSTITUTED = (VARIABLE,)\n' + + +def test_each_form_of_prefix_handling_is_counted_once_and_prose_is_not(tmp_path): + metrics = _load().measure( + _tree( + tmp_path, + { + "dw/references.py": _REFERENCES + + "def f(x):\n return x.startswith(ASSET) and x[len(ASSET):]\n", + "dw/a.py": ( + "from . import references\n" + "def a(x):\n return x.startswith(references.ASSET)\n" # (a) + "def b(x):\n return x[len(references.ASSET) :]\n" # (b) + "def c(x):\n return references.ASSET + x\n" # (c) + "ALIAS = references.VARIABLE\n" # (d) + 'LITERAL = "builtin:h3.json"\n' # not a prefix here: ignored + 'def e():\n return "asset:cat.png"\n' # (e) + "def prose():\n return \"'asset:' reads a file\"\n" + ), + }, + ) + ) + assert metrics["prefix_handling"] == 5 + + +def test_a_name_bound_to_the_references_module_resolves_in_every_import_form( + tmp_path, +): + forms = { + "dw/a.py": "from . import references\nx.startswith(references.ASSET)\n", + "dw/b.py": "from . import references as refs\nx.startswith(refs.ASSET)\n", + "dw/c.py": "from dw import references\nx.startswith(references.ASSET)\n", + "dw/d.py": "import dw.references as r\nx.startswith(r.ASSET)\n", + "dw/e.py": "import dw.references\nx.startswith(dw.references.ASSET)\n", + "dw/server/f.py": "from .. import references as p\nx.startswith(p.ASSET)\n", + "dw/g.py": "from .references import ASSET\nx.startswith(ASSET)\n", + "dw/h.py": "from .references import ASSET as A\nx.startswith(A)\n", + "dw/i.py": "from . import references\nK = references.ASSET\nx.startswith(K)\n", + "dw/j.py": "from .assets import ASSET_PREFIX\nx.startswith(ASSET_PREFIX)\n", + "dw/k.py": "from . import references\nx.startswith((references.ASSET, 'z'))\n", + } + load = _load() + for name, text in forms.items(): + root = _tree(tmp_path / name.replace("/", "_"), {name: text}) + _tree(root, {"dw/references.py": _REFERENCES}) + count = 2 if name == "dw/i.py" else 1 # the alias assignment and its use + assert load.measure(root)["prefix_handling"] == count, name + + +def test_an_fstring_fragment_ending_in_a_prefix_is_counted(tmp_path): + metrics = _load().measure( + _tree( + tmp_path, + { + "dw/references.py": _REFERENCES, + "dw/a.py": 'def f(n):\n return f"Kept {n} as asset:{n}"\n', + }, + ) + ) + assert metrics["prefix_handling"] == 1 From a531a8cee16c6d96b566fe3633a18c037bbd3b6f Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 12:43:07 -0500 Subject: [PATCH 11/41] metrics: prefix_handling's fragment rule needs a whole prefix An f-string fragment ending '_output:' is the tail of another word, not an output: reference; only a prefix preceded by a non-identifier character counts. Count on the tree unchanged (82). Co-Authored-By: Claude Opus 5.5 --- scripts/arch_metrics.py | 13 ++++++++++++- tests/test_arch_metrics.py | 15 +++++++++++++++ 2 files changed, 27 insertions(+), 1 deletion(-) diff --git a/scripts/arch_metrics.py b/scripts/arch_metrics.py index f654e147..e3419878 100644 --- a/scripts/arch_metrics.py +++ b/scripts/arch_metrics.py @@ -319,6 +319,17 @@ def _assignment(node): return (target.id if isinstance(target, ast.Name) else None), value +def _ends_with_prefix(text, prefixes): + """True when `text` ends with a whole prefix: `" as asset:"` does, + `"_output:"` (the tail of some other word) does not.""" + for prefix in prefixes: + if text.endswith(prefix): + before = text[: -len(prefix)][-1:] + if not (before.isalnum() or before == "_"): + return True + return False + + def prefix_handling_sites(tree, constants, prefixes): """Every hand-written handling of a reference prefix in one parsed module, as (line, form) with form one of "a".."e".""" @@ -381,7 +392,7 @@ def prefix_handling_sites(tree, constants, prefixes): ): text = node.value if any(text.startswith(p) and len(text) > len(p) for p in prefixes) or ( - id(node) in fragments and text.endswith(tuple(prefixes)) + id(node) in fragments and _ends_with_prefix(text, prefixes) ): found.append((node.lineno, "e")) return found diff --git a/tests/test_arch_metrics.py b/tests/test_arch_metrics.py index a1652ddd..1f08763b 100644 --- a/tests/test_arch_metrics.py +++ b/tests/test_arch_metrics.py @@ -244,3 +244,18 @@ def test_an_fstring_fragment_ending_in_a_prefix_is_counted(tmp_path): ) ) assert metrics["prefix_handling"] == 1 + + +def test_a_fragment_ending_in_a_word_that_merely_ends_like_a_prefix_is_not_counted( + tmp_path, +): + metrics = _load().measure( + _tree( + tmp_path, + { + "dw/a.py": 'M = f"{x}_output:{y}"\nN = f"{x} audio_item:{y}"\n' + 'K = f"kept as asset:{y}"\n' + }, + ) + ) + assert metrics["prefix_handling"] == 1 From f0fb678da987ad968ae9d7fad6d81e44157b2aea Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 12:51:10 -0500 Subject: [PATCH 12/41] refactor(references): every prefix is read and built through references.py Adds RESERVED_TEXT and LAZY_MEDIA, deletes the alias constants and _UNRESOLVED_PREFIXES copies, and moves each startswith/removeprefix/slice/ concatenation site onto is_ref/ref_name/make_ref. prefix_handling 82 -> 0. Co-Authored-By: Claude Opus 5.5 --- docs/stabilization/baseline.json | 2 +- dw/adapter_compatibility.py | 13 ++++++------ dw/argument_media.py | 4 ++-- dw/arguments.py | 19 ++++++++++------- dw/assets.py | 10 +++++---- dw/content_types.py | 9 +++----- dw/elision.py | 2 +- dw/for_each.py | 10 +++++---- dw/kernel_availability.py | 3 +-- dw/locations.py | 2 +- dw/plan.py | 18 +++++++--------- dw/probe_paths.py | 10 ++++----- dw/prompts.py | 30 +++++++++----------------- dw/realize.py | 24 ++++++++++----------- dw/reference_limits.py | 8 +++---- dw/reference_names.py | 36 ++++++++++++++++---------------- dw/references.py | 7 +++++++ dw/runs.py | 12 +++++------ dw/server/admission.py | 9 ++++---- dw/server/catalog.py | 15 ++++++------- dw/server/enhancers.py | 4 +++- dw/server/exports.py | 11 +++++----- dw/server/jobs.py | 6 +++--- dw/server/outputs.py | 8 +++---- dw/server/routes/assets.py | 4 ++-- dw/server/routes/library.py | 7 +++---- dw/shots.py | 15 ++++++------- dw/step_value_checks.py | 9 +++----- dw/subfolders.py | 10 ++++----- dw/variable_constraints.py | 19 ++++++++--------- dw/video_extensions.py | 11 +++++----- dw/vram_estimate.py | 17 ++++++++------- dw/workflow_run.py | 7 +++---- tests/test_arguments.py | 8 +++++++ tests/test_assets.py | 10 +++++++++ tests/test_output_references.py | 8 +++++++ tests/test_prompt_references.py | 6 +++--- tests/test_prompts.py | 20 ++++++++++++++++-- tests/test_reference_sets.py | 36 -------------------------------- tests/test_references.py | 30 ++++++++++++++++++++++++++ tests/test_server_guides.py | 4 ++-- 41 files changed, 254 insertions(+), 239 deletions(-) delete mode 100644 tests/test_reference_sets.py diff --git a/docs/stabilization/baseline.json b/docs/stabilization/baseline.json index 8bdb1893..2e014edf 100644 --- a/docs/stabilization/baseline.json +++ b/docs/stabilization/baseline.json @@ -3,7 +3,7 @@ "modules_over_1000_lines": 0, "functions_over_150_lines": 0, "prefix_literals": 0, - "prefix_handling": 82, + "prefix_handling": 0, "test_dw_patch_targets": 284, "claude_md_lines": 957, "duplicate_blocks": 5, diff --git a/dw/adapter_compatibility.py b/dw/adapter_compatibility.py index eaebf6df..8734a44e 100644 --- a/dw/adapter_compatibility.py +++ b/dw/adapter_compatibility.py @@ -53,9 +53,6 @@ WEIGHT_NAME_KEY = "weight_name" WORKFLOW_KEY = "workflow" FROM_PRETRAINED_KEY = "from_pretrained_arguments" -# Values another pass resolves; one still spelled out here is not this -# pass's complaint -_UNRESOLVED_PREFIXES = references.UNRESOLVED def _trained_for(weight_name): @@ -140,8 +137,10 @@ def _lora_problems(steps, source_indices, written=None, supplied=()): if not isinstance(lora, dict): continue weight_name = lora.get(WEIGHT_NAME_KEY) - if not isinstance(weight_name, str) or weight_name.startswith( - _UNRESOLVED_PREFIXES + # Values another pass resolves; one still spelled out here is not + # this pass's complaint + if not isinstance(weight_name, str) or references.is_ref( + references.UNRESOLVED, weight_name ): continue problem = _problem(workflow, weight_name) @@ -180,9 +179,9 @@ def _path_for(written_steps, source, position, supplied, key=WEIGHT_NAME_KEY): return None entry = loras[position] reference = entry.get(key) if isinstance(entry, dict) else None - if not isinstance(reference, str) or not reference.startswith(references.VARIABLE): + variable = references.ref_name(references.VARIABLE, reference) + if variable is None: return None - variable = reference.removeprefix(references.VARIABLE) return f"arguments.{variable}" if variable in (supplied or ()) else None diff --git a/dw/argument_media.py b/dw/argument_media.py index 54c55d37..aa122ac0 100644 --- a/dw/argument_media.py +++ b/dw/argument_media.py @@ -135,7 +135,7 @@ def fetch_image(img_spec, base_dir=None): raise ValueError(f"Image specification must be a string, got {type(img_spec)}") # Skip cross-step and variable references — these are resolved later during execution - if references.is_ref((references.PREVIOUS_RESULT, references.VARIABLE), img_spec): + if references.is_ref(references.LAZY_MEDIA, img_spec): logger.debug(f"Skipping deferred reference: {img_spec}") return img_spec @@ -291,7 +291,7 @@ def fetch_video(video_spec, base_dir=None): ) # Skip cross-step and variable references — these are resolved later during execution - if references.is_ref((references.PREVIOUS_RESULT, references.VARIABLE), video_spec): + if references.is_ref(references.LAZY_MEDIA, video_spec): logger.debug(f"Skipping deferred reference: {video_spec}") return video_spec diff --git a/dw/arguments.py b/dw/arguments.py index 7bb645f4..c61a716d 100644 --- a/dw/arguments.py +++ b/dw/arguments.py @@ -14,7 +14,7 @@ load_constant_from_name, has_method, ) -from .prompts import PROMPT_PREFIX, fetch_prompt +from .prompts import fetch_prompt from .assets import fetch_asset, is_asset_reference from .runs import fetch_output, is_output_reference from .argument_media import ( @@ -286,7 +286,7 @@ def is_constant_reference(value): def is_prompt_reference(value): """Whether a value references a stored prompt in the prompt library.""" - return references.is_ref(PROMPT_PREFIX, value) + return references.is_ref(references.PROMPT, value) def fetch_constant(reference): @@ -312,7 +312,12 @@ def fetch_constant(reference): ValueError: If the name resolves to nothing, or to something callable InvalidInputError: If the name is not a dotted python name """ - name = validate_constant_name(reference.removeprefix(references.CONSTANT).strip()) + name = references.ref_name(references.CONSTANT, reference) + if name is None: + # A bare name reads as written: callers guard with is_constant_reference, + # a direct caller need not + name = reference + name = validate_constant_name(name.strip()) try: value = load_constant_from_name(name) @@ -325,8 +330,8 @@ def fetch_constant(reference): if callable(value): raise ValueError( f"'{name}' is a {type(value).__name__}, not a constant - " - f"'{references.CONSTANT}' reads a value, and a type is named with a " - f"'_type' argument instead" + f"'{references.make_ref(references.CONSTANT, '')}' reads a value, " + f"and a type is named with a '_type' argument instead" ) logger.info(f"Reading constant {name}") @@ -931,9 +936,7 @@ def _realize_lazy_frame_arguments(arguments, base_dir): video = arguments["video"] if is_path_reference(video) or isinstance(video, (list, dict)): video = resolve_path_references(video, base_dir) - deferred = references.is_ref( - (references.PREVIOUS_RESULT, references.VARIABLE), video - ) + deferred = references.is_ref(references.LAZY_MEDIA, video) url = isinstance(video, str) and ( video.startswith("http://") or video.startswith("https://") ) diff --git a/dw/assets.py b/dw/assets.py index 78293f16..c389c785 100644 --- a/dw/assets.py +++ b/dw/assets.py @@ -26,9 +26,6 @@ logger = logging.getLogger("dw") -# The prefix marking a value as a reference to a stored asset -ASSET_PREFIX = references.ASSET - # The asset library of the run in progress. A server holds several # workspaces and each has its own assets, so this cannot be a process-wide # environment variable there the way the prompt library can - there is one @@ -110,7 +107,12 @@ def resolve_asset_reference(reference, asset_dir=None, base_dir=None, library=No ValueError: If no file exists under that name in any directory on the search path """ - name = validate_asset_reference(reference.removeprefix(ASSET_PREFIX).strip()) + name = references.ref_name(references.ASSET, reference) + if name is None: + # A bare name resolves as written: callers guard with is_asset_reference, + # a direct caller need not + name = reference + name = validate_asset_reference(name.strip()) library = library or asset_library(asset_dir, base_dir) # Confined to the library it was found in: the name is joined onto a # directory, so the containment check is what makes a name a name diff --git a/dw/content_types.py b/dw/content_types.py index de9f7d27..a9d03320 100644 --- a/dw/content_types.py +++ b/dw/content_types.py @@ -67,11 +67,6 @@ # The only container encode_video writes - it always encodes h264 video MUXED_VIDEO_CONTENT_TYPE = "video/mp4" -# Reference prefixes substitution resolves before this pass runs. One still -# spelled out here is one nothing resolved, and that is the undeclared- -# variable pass's complaint rather than a shape error -_UNRESOLVED_PREFIXES = references.SUBSTITUTED - # Result types a browser would run as a document on the UI origin REFUSED_ACTIVE_CONTENT_TYPES = frozenset({"text/html", "text/xml"}) @@ -143,7 +138,9 @@ def content_type_errors(workflow_definition, source_indices=None): if not isinstance(result, dict) or CONTENT_TYPE_KEY not in result: continue value = result[CONTENT_TYPE_KEY] - if isinstance(value, str) and value.startswith(_UNRESOLVED_PREFIXES): + # A prefix substitution resolves before this pass: one still spelled + # out is the undeclared-variable pass's complaint, not a shape error + if references.is_ref(references.SUBSTITUTED, value): continue source = references.author_index(source_indices, index) name = step.get("name") diff --git a/dw/elision.py b/dw/elision.py index 6d3cbd75..473a6e8d 100644 --- a/dw/elision.py +++ b/dw/elision.py @@ -207,7 +207,7 @@ def overriding_variables(written, substituted_steps): def _reads_variable(tree, name): """Whether anything in `tree` references 'variable:'.""" - reference = references.VARIABLE + name + reference = references.make_ref(references.VARIABLE, name) if isinstance(tree, str): return tree == reference if isinstance(tree, dict): diff --git a/dw/for_each.py b/dw/for_each.py index 0e8d491f..d6a1a0eb 100644 --- a/dw/for_each.py +++ b/dw/for_each.py @@ -200,8 +200,9 @@ def _rewrite(value, path, groups, member): return _item(value, path, member) if references.is_ref(references.PREVIOUS_RESULT, value): reference = references.ref_name(references.PREVIOUS_RESULT, value) - return references.PREVIOUS_RESULT + _rewrite_reference( - reference, path, groups, member + return references.make_ref( + references.PREVIOUS_RESULT, + _rewrite_reference(reference, path, groups, member), ) return value # A leaf is only copied where the copy is needed: inside a member, where @@ -260,7 +261,7 @@ def _gather(value, path, groups): f"for_each steps available here: {sorted(groups)}", ) return [ - references.PREVIOUS_RESULT + member_name(group, key) + references.make_ref(references.PREVIOUS_RESULT, member_name(group, key)) for key in groups[group]["keys"] ] @@ -289,7 +290,8 @@ def _rewrite_reference(reference, path, groups, member): raise ForEachError( render_path(path), f"'{reference}' names the for_each step '{group}' {where}. Use " - f"'{references.GATHER}{group}' for every member's result, or a reference " + f"'{references.make_ref(references.GATHER, group)}' for every member's " + f"result, or a reference " f"from a for_each step over the same list for the same-keyed member", ) diff --git a/dw/kernel_availability.py b/dw/kernel_availability.py index 69d4ff94..e0844c27 100644 --- a/dw/kernel_availability.py +++ b/dw/kernel_availability.py @@ -33,7 +33,6 @@ from .type_helpers import load_type_from_name ATTN_PROCESSOR_KEY = "attn_processor_type" -_UNRESOLVED_PREFIXES = references.SUBSTITUTED KERNEL_FAULT_MARKER = "cannot be used on this machine" @@ -159,7 +158,7 @@ def kernel_availability_fault(value): A fault is not memoized, and so is re-probed on every call - see `_fault_for_name`. """ - if not isinstance(value, str) or value.startswith(_UNRESOLVED_PREFIXES): + if not isinstance(value, str) or references.is_ref(references.SUBSTITUTED, value): return None try: return _fault_for_name(value) diff --git a/dw/locations.py b/dw/locations.py index e128aa9c..161cd26a 100644 --- a/dw/locations.py +++ b/dw/locations.py @@ -530,7 +530,7 @@ def _is_media_key(key): def _deferred(value): """Whether a location is resolved later rather than being one now.""" - return value.startswith(references.DEFERRED) + return references.is_ref(references.DEFERRED, value) def _check(value, base_dir, what): diff --git a/dw/plan.py b/dw/plan.py index 1db5a0e3..fd74f857 100644 --- a/dw/plan.py +++ b/dw/plan.py @@ -25,12 +25,8 @@ from .elision import elide_definition from .hub_cache import repo_download_incomplete, scan_models -from .realize import ( - BUILTIN_PREFIX, - VARIABLE_PREFIX, - read_sub_workflow, - realize_workflow, -) +from . import references +from .realize import read_sub_workflow, realize_workflow from .security import validate_url from .validation import _is_seeded @@ -183,8 +179,8 @@ def list_entries(definition, realized): if not isinstance(step, dict): continue reference = step.get(FOR_EACH_KEY) - if isinstance(reference, str) and reference.startswith(VARIABLE_PREFIX): - name = reference.removeprefix(VARIABLE_PREFIX) + name = references.ref_name(references.VARIABLE, reference) + if name is not None: value = variables.get(name) if isinstance(value, list): entries[name] = len(value) @@ -227,8 +223,8 @@ def fingerprint(expanded, definition, annotations=None): for key in DOCUMENTATION_KEYS: doc.pop(key, None) written_seed = definition.get("seed") - if isinstance(written_seed, str) and written_seed.startswith(VARIABLE_PREFIX): - name = written_seed.removeprefix(VARIABLE_PREFIX) + name = references.ref_name(references.VARIABLE, written_seed) + if name is not None: variables = doc.get("variables") if isinstance(variables, dict) and name in variables: variables[name] = None @@ -649,7 +645,7 @@ def _sub_workflow_paths(expanded): for step in expanded.get("steps") or []: reference = step.get("workflow") if isinstance(step, dict) else None path = reference.get("path") if isinstance(reference, dict) else None - if isinstance(path, str) and not path.startswith(BUILTIN_PREFIX): + if isinstance(path, str) and not references.is_ref(references.BUILTIN, path): arguments = reference.get("arguments") yield path, arguments if isinstance(arguments, dict) else {} diff --git a/dw/probe_paths.py b/dw/probe_paths.py index c4c4c01f..a12444ab 100644 --- a/dw/probe_paths.py +++ b/dw/probe_paths.py @@ -16,10 +16,6 @@ from .locations import is_http_url, validate_media_path from .runs import fetch_output, is_output_reference -# Left to the run-time check: not yet resolved to a real file at the point -# validation walks the expanded definition. -UNRESOLVED_PREFIXES = references.UNRESOLVED - def resolve_probe_path(value, base_dir, what="a media argument"): """The local file `value` names, or None when it is not yet resolvable, @@ -38,7 +34,9 @@ def resolve_probe_path(value, base_dir, what="a media argument"): value = value.get("location") if not isinstance(value, str) or not value: return None - if value.startswith(UNRESOLVED_PREFIXES) or is_http_url(value): + # Left to the run-time check: not yet resolved to a real file at the point + # validation walks the expanded definition + if references.is_ref(references.UNRESOLVED, value) or is_http_url(value): return None if is_asset_reference(value) or is_output_reference(value): try: @@ -63,4 +61,4 @@ def resolve_probe_path(value, base_dir, what="a media argument"): return value if os.path.isfile(value) else None -__all__ = ["UNRESOLVED_PREFIXES", "resolve_probe_path"] +__all__ = ["resolve_probe_path"] diff --git a/dw/prompts.py b/dw/prompts.py index 1bb0da76..1fdcb3a8 100644 --- a/dw/prompts.py +++ b/dw/prompts.py @@ -22,23 +22,6 @@ logger = logging.getLogger("dw") -# The prefix marking a value as a reference to a stored prompt. The name after it -# is rooted at the prompt directory, not the workflow file - prompts are a shared -# library, and the same reference means the same text from every workflow -PROMPT_PREFIX = references.PROMPT - -# The prefixes a stored prompt's text may not begin with. Resolved text is -# substituted where the reference stood, so text that itself looks like a -# reference would be resolved again - or worse, expand a step's iterations -RESERVED_TEXT_PREFIXES = ( - references.PREVIOUS_RESULT, - references.VARIABLE, - references.CONSTANT, - references.ASSET, - references.OUTPUT, - PROMPT_PREFIX, -) - def get_prompt_dir(base_dir=None): """The directory stored prompts are rooted at. @@ -94,7 +77,12 @@ def resolve_prompt_reference(reference, prompt_dir=None, base_dir=None, library= ValueError: If no prompt file exists under that name in any directory on the search path """ - name = validate_prompt_reference(reference.removeprefix(PROMPT_PREFIX).strip()) + name = references.ref_name(references.PROMPT, reference) + if name is None: + # A bare name resolves as written: callers guard with is_prompt_reference, + # a direct caller need not + name = reference + name = validate_prompt_reference(name.strip()) library = library or prompt_library(prompt_dir, base_dir) found = library.find(name, refuse=True) if found: @@ -153,10 +141,12 @@ def fetch_prompt(reference, prompt_dir=None, base_dir=None): # Arguments are realized more than once, and iteration expansion scans the # realized template - text that begins like a reference would be treated # as one on the next pass, so it is data that may not masquerade as syntax - if text.startswith(RESERVED_TEXT_PREFIXES): + # (resolved text is substituted where the reference stood, so text that + # itself looks like one would be resolved again) + if references.is_ref(references.RESERVED_TEXT, text): raise ValueError( f"Prompt '{reference}' has text beginning with a reference prefix " - f"({', '.join(RESERVED_TEXT_PREFIXES)}) - a prompt's text may not " + f"({', '.join(references.RESERVED_TEXT)}) - a prompt's text may not " f"itself be a reference" ) diff --git a/dw/realize.py b/dw/realize.py index 6f878268..eaa8f5c6 100644 --- a/dw/realize.py +++ b/dw/realize.py @@ -24,10 +24,9 @@ import os from . import references -from .prompts import PROMPT_PREFIX, fetch_prompt +from .prompts import fetch_prompt from .runs import ( LATEST, - OUTPUT_PREFIX, is_output_reference, output_root as default_output_root, resolve_output_reference, @@ -39,9 +38,6 @@ logger = logging.getLogger("dw") -BUILTIN_PREFIX = references.BUILTIN -VARIABLE_PREFIX = references.VARIABLE - def realize_workflow( definition, @@ -101,8 +97,8 @@ def realize_workflow( # of this realized copy with no seed argument would put that null # default back over the pinned integer everywhere but the top level. definition_seed = definition.get("seed") - if isinstance(definition_seed, str) and definition_seed.startswith(VARIABLE_PREFIX): - seed_variable = definition_seed.removeprefix(VARIABLE_PREFIX) + seed_variable = references.ref_name(references.VARIABLE, definition_seed) + if seed_variable is not None: if isinstance(variables, dict) and seed_variable in variables: variables[seed_variable] = seed realized = _pin( @@ -118,7 +114,7 @@ def strings_with_prefix(tree, prefix): found = [] def collect(value): - if value.startswith(prefix) and value not in found: + if references.is_ref(prefix, value) and value not in found: found.append(value) return value @@ -149,7 +145,7 @@ def _pin(value, annotations, base_dir, prompt_dir, output_root, pin_outputs=True `pin_outputs`, output references pinned.""" def transform(string): - if string.startswith(PROMPT_PREFIX): + if references.is_ref(references.PROMPT, string): return _inline_prompt(string, annotations, prompt_dir, base_dir) if pin_outputs and is_output_reference(string): return _pin_output(string, output_root) @@ -170,7 +166,7 @@ def _inline_prompt(reference, annotations, prompt_dir, base_dir): except (SecurityError, OSError, ValueError) as e: logger.warning(f"Realization kept {reference} as written: {e}") return reference - name = reference.removeprefix(PROMPT_PREFIX).strip() + name = references.ref_name(references.PROMPT, reference).strip() if name not in annotations["prompts"]: annotations["prompts"].append(name) return text @@ -186,7 +182,7 @@ def _pin_output(reference, output_root): disk - realizing must not fail on a reference the run has not reached yet. """ - name = reference.removeprefix(OUTPUT_PREFIX).strip() + name = references.ref_name(references.OUTPUT, reference).strip() if not any( part == LATEST or version_selector(part) is not None for part in name.split("/") ): @@ -198,7 +194,7 @@ def _pin_output(reference, output_root): except (SecurityError, OSError, ValueError) as e: logger.warning(f"Realization kept {reference} as written: {e}") return reference - return f"{OUTPUT_PREFIX}{relative}" + return references.make_ref(references.OUTPUT, relative) def _record_sub_workflows(steps, annotations, base_dir, workflow_dir): @@ -215,7 +211,9 @@ def scan(value): reference = value.get("workflow") if isinstance(reference, dict): path = reference.get("path") - if isinstance(path, str) and not path.startswith(BUILTIN_PREFIX): + if isinstance(path, str) and not references.is_ref( + references.BUILTIN, path + ): annotations["sub_workflows"][path] = _digest( path, base_dir, workflow_dir ) diff --git a/dw/reference_limits.py b/dw/reference_limits.py index 4ff82583..5add5387 100644 --- a/dw/reference_limits.py +++ b/dw/reference_limits.py @@ -44,10 +44,6 @@ REFERENCE_TYPE_KEY = "reference_type" -# Values substitution resolves before this pass runs; one still spelled out -# is another pass's complaint, not this one's -_UNRESOLVED_PREFIXES = references.UNRESOLVED - def _family(module_name): """The REFERENCE_LIMIT_BLOCKS key a class's module belongs to, or None.""" @@ -99,7 +95,9 @@ def _reference_class(value): if not isinstance(value, dict): return None name = value.get(REFERENCE_TYPE_KEY) - if not isinstance(name, str) or name.startswith(_UNRESOLVED_PREFIXES): + # A value substitution resolves before this pass, still spelled out, is + # another pass's complaint + if not isinstance(name, str) or references.is_ref(references.UNRESOLVED, name): return None if "." not in name: return None diff --git a/dw/reference_names.py b/dw/reference_names.py index dafbe830..08492f16 100644 --- a/dw/reference_names.py +++ b/dw/reference_names.py @@ -18,10 +18,7 @@ """ from . import references -from .assets import ASSET_PREFIX from .for_each import MEMBER_SEPARATOR, render_path -from .prompts import PROMPT_PREFIX -from .runs import OUTPUT_PREFIX from .security import ( InvalidInputError, validate_asset_reference, @@ -29,11 +26,6 @@ validate_prompt_reference, ) -# Substitution and expansion run before this pass, so every string reaching -# it is literal. One still spelled with a deferred prefix is nothing this -# pass resolved, and the undeclared-variable pass owns that complaint -_UNRESOLVED_PREFIXES = references.UNRESOLVED - def _output_name(reference): """The part of an `output:` reference the name rule applies to. @@ -41,14 +33,17 @@ def _output_name(reference): `latest` in the run-id position is expanded before the path is joined, so it is checked as the ordinary segment it looks like. """ - return reference.removeprefix(OUTPUT_PREFIX).strip() + return references.ref_name(references.OUTPUT, reference).strip() -_KINDS = ( - (OUTPUT_PREFIX, validate_output_reference, _output_name), - (ASSET_PREFIX, validate_asset_reference, None), - (PROMPT_PREFIX, validate_prompt_reference, None), -) +# The name rule each reference kind holds its name to: the validator, and how +# to read the name out of the reference when it is not simply what follows the +# prefix +_KINDS = { + references.OUTPUT: (validate_output_reference, _output_name), + references.ASSET: (validate_asset_reference, None), + references.PROMPT: (validate_prompt_reference, None), +} def reference_name_errors(workflow_definition, source_indices=None): @@ -93,11 +88,16 @@ def reference_fault(value): """ if not isinstance(value, str): return None - for prefix, check, extract in _KINDS: - if not value.startswith(prefix): + for prefix, (check, extract) in _KINDS.items(): + rest = references.ref_name(prefix, value) + if rest is None: continue - rest = value[len(prefix) :].strip() - if not rest or rest.startswith(_UNRESOLVED_PREFIXES): + rest = rest.strip() + # Substitution and expansion run before this pass, so every string + # reaching it is literal. One still spelled with a deferred prefix is + # nothing this pass resolved, and the undeclared-variable pass owns + # that complaint + if not rest or references.is_ref(references.UNRESOLVED, rest): return None try: check(extract(value) if extract else rest) diff --git a/dw/references.py b/dw/references.py index bc528dac..82d4ddfa 100644 --- a/dw/references.py +++ b/dw/references.py @@ -47,6 +47,13 @@ # Anything a value may still hold before realization: the unresolved # prefixes plus the ones realize_args fetches or looks up DEFERRED = UNRESOLVED + (ASSET, OUTPUT, PROMPT, CONSTANT, BUILTIN) +# What a stored prompt's text may not begin with: it would be resolved a +# second time (or expand into iteration) once the prompt is substituted in. +# The order is the order the refusal message lists them in +RESERVED_TEXT = (PREVIOUS_RESULT, VARIABLE, CONSTANT, ASSET, OUTPUT, PROMPT) +# A media argument's value that names a file only once the run reaches the +# step: an earlier step's result or a variable +LAZY_MEDIA = (PREVIOUS_RESULT, VARIABLE) def is_ref(kind, value): diff --git a/dw/runs.py b/dw/runs.py index 0e511b34..5f0ad9cc 100644 --- a/dw/runs.py +++ b/dw/runs.py @@ -51,11 +51,6 @@ # inside it is the whole reproduction story REALIZED_FILE_NAME = "workflow.json" -# The prefix marking a value as a reference to a file an earlier run wrote. -# Like 'asset:', it stands for a path - what a previous run made is an input -# like any other, and multi-stage work is what a workflow engine is for -OUTPUT_PREFIX = references.OUTPUT - # The segment that means "the newest run of this workflow that has the # file", so a workflow can name the stage before it without being edited # after every run - see _resolve_segments for why it is not simply the @@ -233,7 +228,12 @@ def resolve_output_reference(reference, root=None): """ from .security import validate_output_reference, validate_path - name = validate_output_reference(reference.removeprefix(OUTPUT_PREFIX).strip()) + name = references.ref_name(references.OUTPUT, reference) + if name is None: + # A bare name resolves as written: callers guard with is_output_reference, + # a direct caller need not + name = reference + name = validate_output_reference(name.strip()) root = root or output_root() resolved = _resolve_segments(root, name.split("/"), reference, root) diff --git a/dw/server/admission.py b/dw/server/admission.py index cc966078..a284e496 100644 --- a/dw/server/admission.py +++ b/dw/server/admission.py @@ -20,15 +20,14 @@ from pydantic import BaseModel, Field from ..assets import ( - ASSET_PREFIX, activate_asset_dir, deactivate_asset_dir, is_asset_reference, resolve_asset_reference, ) -from .. import validation +from .. import references, validation from ..plan import build_plan -from ..prompts import PROMPT_PREFIX, resolve_prompt_reference +from ..prompts import resolve_prompt_reference from ..runs import is_output_reference, resolve_output_reference from ..validation import WARNING, run_checks, to_warnings from ..variables import argument_errors @@ -267,7 +266,7 @@ def _string_leaves(value, path): # A server configured with no asset library has # no root to fail against: the resolver would # name no directory it searched - name = leaf.removeprefix(ASSET_PREFIX).strip() + name = references.ref_name(references.ASSET, leaf).strip() raise ValueError( f"Unknown asset {name!r}: " "this workspace has no asset library" @@ -276,7 +275,7 @@ def _string_leaves(value, path): # a miss names every root it looked in, the workspace's # own first resolve_asset_reference(leaf, library=asset_library) - elif leaf.startswith(PROMPT_PREFIX): + elif references.is_ref(references.PROMPT, leaf): resolve_prompt_reference(leaf, library=prompt_library) elif is_output_reference(leaf): resolve_output_reference(leaf, root=outputs) diff --git a/dw/server/catalog.py b/dw/server/catalog.py index dd85e915..c8eee39c 100644 --- a/dw/server/catalog.py +++ b/dw/server/catalog.py @@ -10,7 +10,7 @@ from fastapi import HTTPException -from ..prompts import PROMPT_PREFIX +from .. import references from ..security import ( InvalidInputError, SecurityError, @@ -45,17 +45,18 @@ def _prune_missing(cache): def collect_prompt_references(value): """Every stored-prompt name a definition references, at any depth - so deleting a prompt can warn which workflows would break.""" - references = set() + found = set() if isinstance(value, str): - if value.startswith(PROMPT_PREFIX): - references.add(value.removeprefix(PROMPT_PREFIX).strip()) + name = references.ref_name(references.PROMPT, value) + if name is not None: + found.add(name.strip()) elif isinstance(value, dict): for item in value.values(): - references |= collect_prompt_references(item) + found |= collect_prompt_references(item) elif isinstance(value, list): for item in value: - references |= collect_prompt_references(item) - return references + found |= collect_prompt_references(item) + return found def catalog_name_from_root(path, root): diff --git a/dw/server/enhancers.py b/dw/server/enhancers.py index fbdab84a..057c914e 100644 --- a/dw/server/enhancers.py +++ b/dw/server/enhancers.py @@ -9,6 +9,8 @@ import uuid +from .. import references + # A generic system prompt for expanding an idea into an image-generation # prompt - the counterpart of the H3 preset's Context-IR spec, which lives # in the builtin workflow rather than here @@ -26,7 +28,7 @@ PRESETS = { "h3": { "label": "MiniMax-H3 Context-IR", - "workflow": "builtin:h3_context_ir.json", + "workflow": references.make_ref(references.BUILTIN, "h3_context_ir.json"), "default_model": "Qwen/Qwen3-4B-Instruct-2507", "models": [ "Qwen/Qwen3-4B-Instruct-2507", diff --git a/dw/server/exports.py b/dw/server/exports.py index 43c19fe7..2808014d 100644 --- a/dw/server/exports.py +++ b/dw/server/exports.py @@ -26,11 +26,10 @@ import shutil from dataclasses import dataclass, field -from ..assets import ASSET_PREFIX +from .. import references from ..realize import strings_with_prefix from ..runs import ( MANIFEST_FILE_NAME, - OUTPUT_PREFIX, is_output_reference, resolve_output_reference, ) @@ -278,10 +277,10 @@ def _run_manifest(output_root, run_dir): def _copy_assets(summary, workflow, target, asset_library): """Every 'asset:' the workflow names, under its own name in assets/.""" - for reference in strings_with_prefix(workflow, ASSET_PREFIX): + for reference in strings_with_prefix(workflow, references.ASSET): try: name = validate_asset_reference( - reference.removeprefix(ASSET_PREFIX).strip() + references.ref_name(references.ASSET, reference).strip() ) except SecurityError: summary.missing.append(reference) @@ -303,10 +302,10 @@ def _copy_inputs(summary, workflow, target, output_root): immutable record of the run - so the directory name is the reference's own name, and the README says where each one came from. """ - for reference in strings_with_prefix(workflow, OUTPUT_PREFIX): + for reference in strings_with_prefix(workflow, references.OUTPUT): if not is_output_reference(reference): continue - name = reference.removeprefix(OUTPUT_PREFIX).strip() + name = references.ref_name(references.OUTPUT, reference).strip() try: source = resolve_output_reference(reference, output_root) except (SecurityError, OSError, ValueError): diff --git a/dw/server/jobs.py b/dw/server/jobs.py index 017d9495..b07c1340 100644 --- a/dw/server/jobs.py +++ b/dw/server/jobs.py @@ -43,7 +43,7 @@ validate_path, validate_workflow_path, ) -from ..realize import VARIABLE_PREFIX +from .. import references from ..runs import REALIZED_FILE_NAME from ..settings import resolve_path from ..workspace import DEFAULT_WORKSPACE_NAME @@ -307,9 +307,9 @@ def seed_variable(self, job_id): """ definition = self.definition(job_id) seed = (definition or {}).get("seed") - if not isinstance(seed, str) or not seed.startswith(VARIABLE_PREFIX): + name = references.ref_name(references.VARIABLE, seed) + if name is None: return None - name = seed.removeprefix(VARIABLE_PREFIX) return name if name in (definition.get("variables") or {}) else None def rerun_spec(self, job_id, new_seed=False): diff --git a/dw/server/outputs.py b/dw/server/outputs.py index 3d038270..dd85cc50 100644 --- a/dw/server/outputs.py +++ b/dw/server/outputs.py @@ -20,7 +20,7 @@ from starlette.background import BackgroundTask from .. import settings -from ..assets import ASSET_PREFIX +from .. import references from ..library import ( ASSETS_KIND, WORKSPACE_ORIGIN, @@ -28,7 +28,7 @@ LibraryRoot, library_path, ) -from ..runs import OUTPUT_PREFIX, is_output_reference, run_versions, split_run_path +from ..runs import is_output_reference, run_versions, split_run_path from ..security import ( ALLOWED_AUDIO_EXTENSIONS, ALLOWED_IMAGE_EXTENSIONS, @@ -170,7 +170,7 @@ def strip_output_prefix(name): lookup keyed on the untouched string quietly misses. """ if is_output_reference(name): - return name.removeprefix(OUTPUT_PREFIX).strip() + return references.ref_name(references.OUTPUT, name).strip() return name @@ -242,7 +242,7 @@ def asset_file(state, reference, ws): (often an examples directory they never wrote to). """ return asset_in( - reference.removeprefix(ASSET_PREFIX).strip(), + references.ref_name(references.ASSET, reference).strip(), resolution_library(state, ws), ) diff --git a/dw/server/routes/assets.py b/dw/server/routes/assets.py index 9cc584c9..fb0cfa5d 100644 --- a/dw/server/routes/assets.py +++ b/dw/server/routes/assets.py @@ -360,7 +360,7 @@ def keep_output_as_asset( provenance, ) - logger.info(f"Kept output {body.name} as asset:{asset_name}") + logger.info(f"Kept output {body.name} as {make_ref(ASSET, asset_name)}") return { "reference": make_ref(ASSET, asset_name), "name": asset_name, @@ -430,7 +430,7 @@ def delete_asset( except ReadOnlyLibraryError as refusal: raise HTTPException(status_code=403, detail=str(refusal)) os.remove(path) - logger.info(f"Deleted asset:{relative} ({path})") + logger.info(f"Deleted {make_ref(ASSET, relative)} ({path})") forget_workspace_usage() return { "name": relative, diff --git a/dw/server/routes/library.py b/dw/server/routes/library.py index 5ec4212a..b45dd23d 100644 --- a/dw/server/routes/library.py +++ b/dw/server/routes/library.py @@ -17,9 +17,8 @@ from pydantic import BaseModel, Field from ...argument_warnings import workflow_argument_warnings -from ...prompts import RESERVED_TEXT_PREFIXES from ...schema import format_validation_errors, load_schema, validate_data -from ... import validation +from ... import references, validation from ...security import InvalidInputError, SecurityError, validate_prompt_reference from ...workflow import Workflow from ...library import ( @@ -569,11 +568,11 @@ def save_prompt(http_request: Request, name: str, request: PromptRequest): status, message = validate_data(request.prompt, load_schema("prompt")) if not status: raise HTTPException(status_code=400, detail=message) - if str(request.prompt.get("text", "")).startswith(RESERVED_TEXT_PREFIXES): + if references.is_ref(references.RESERVED_TEXT, str(request.prompt.get("text", ""))): raise HTTPException( status_code=400, detail="A prompt's text may not itself begin with a reference " - f"prefix ({', '.join(RESERVED_TEXT_PREFIXES)})", + f"prefix ({', '.join(references.RESERVED_TEXT)})", ) path = resolve_prompt_name( writable_prompt_directory(state), name, allow_create=True diff --git a/dw/shots.py b/dw/shots.py index f4a8dd33..de1e6479 100644 --- a/dw/shots.py +++ b/dw/shots.py @@ -35,11 +35,11 @@ import copy -from . import references as ref_prefixes +from .references import MEMBER_SEPARATOR, PREVIOUS_RESULT, is_ref, make_ref, ref_name # The step names a for_each member as `@`; only a member of the # group conventionally called `shot` names a shot -SHOT_REFERENCE_PREFIX = f"{ref_prefixes.PREVIOUS_RESULT}shot@" +SHOT_REFERENCE_PREFIX = make_ref(PREVIOUS_RESULT, "shot" + MEMBER_SEPARATOR) def shot_record( @@ -252,15 +252,12 @@ def shot_reference_names(references): return None names = [] for reference in references: - if isinstance(reference, str) and reference.startswith(SHOT_REFERENCE_PREFIX): + step = ref_name(PREVIOUS_RESULT, reference) + if is_ref(SHOT_REFERENCE_PREFIX, reference): # `previous_result:shot@x.field` names the member, not the field - member = reference[len(ref_prefixes.PREVIOUS_RESULT) :] - names.append(member.split(".", 1)[0]) - elif isinstance(reference, str) and reference.startswith( - ref_prefixes.PREVIOUS_RESULT - ): + names.append(step.split(".", 1)[0]) + elif step is not None: # `previous_result:step.field` names the step, not the field - step = reference[len(ref_prefixes.PREVIOUS_RESULT) :] names.append(step.split(".", 1)[0]) else: names.append(None) diff --git a/dw/step_value_checks.py b/dw/step_value_checks.py index 933bb4b1..6cf05f96 100644 --- a/dw/step_value_checks.py +++ b/dw/step_value_checks.py @@ -43,11 +43,6 @@ FPS_KEY = "fps" -# Reference prefixes substitution resolves before this pass runs. One still -# spelled out here is one nothing resolved, and that is the undeclared- -# variable pass's complaint rather than a shape error -_UNRESOLVED_PREFIXES = references.SUBSTITUTED - def fps_errors(workflow_definition, source_indices=None): """Every result 'fps' that cannot be written, as [{path, message}]. @@ -71,7 +66,9 @@ def fps_errors(workflow_definition, source_indices=None): if not isinstance(result, dict) or FPS_KEY not in result: continue value = result[FPS_KEY] - if isinstance(value, str) and value.startswith(_UNRESOLVED_PREFIXES): + # A prefix substitution resolves before this pass: one still spelled + # out is the undeclared-variable pass's complaint, not a shape error + if references.is_ref(references.SUBSTITUTED, value): continue source = references.author_index(source_indices, index) diff --git a/dw/subfolders.py b/dw/subfolders.py index 01f74318..be21f39c 100644 --- a/dw/subfolders.py +++ b/dw/subfolders.py @@ -25,11 +25,6 @@ SUBFOLDER_KEY = "subfolder" FILE_BASE_NAME_KEY = "file_base_name" -# Reference prefixes substitution resolves before this pass runs. One still -# spelled out here is one nothing resolved, and that is the undeclared- -# variable pass's complaint rather than a shape error -_UNRESOLVED_PREFIXES = references.SUBSTITUTED - def step_subfolder(step_definition): """The validated subfolder a step's result names, or '' when it names @@ -86,7 +81,10 @@ def subfolder_errors(workflow_definition, source_indices=None): if key not in result: continue value = result[key] - if isinstance(value, str) and value.startswith(_UNRESOLVED_PREFIXES): + # A prefix substitution resolves before this pass: one still + # spelled out is the undeclared-variable pass's complaint, not a + # shape error + if references.is_ref(references.SUBSTITUTED, value): continue try: if not isinstance(value, str): diff --git a/dw/variable_constraints.py b/dw/variable_constraints.py index ce85ccf0..732cdb0d 100644 --- a/dw/variable_constraints.py +++ b/dw/variable_constraints.py @@ -420,10 +420,9 @@ def walk(node): if not isinstance(node, dict): return reference = node.get("frame_snap") - if isinstance(reference, str) and reference.startswith(references.CONSTRAINT): - node["frame_snap"] = snap_block( - constraints[reference[len(references.CONSTRAINT) :]] - ) + name = references.ref_name(references.CONSTRAINT, reference) + if name is not None: + node["frame_snap"] = snap_block(constraints[name]) for value in node.values(): walk(value) @@ -447,12 +446,12 @@ def walk(node, path): return for key, value in node.items(): where = f"{path}.{key}" if path else key - if ( - key == "frame_snap" - and isinstance(value, str) - and value.startswith(references.CONSTRAINT) - and value[len(references.CONSTRAINT) :] not in constraints - ): + name = ( + references.ref_name(references.CONSTRAINT, value) + if key == "frame_snap" + else None + ) + if name is not None and name not in constraints: errors.append( { "path": where, diff --git a/dw/video_extensions.py b/dw/video_extensions.py index ccecd5f1..621a9d04 100644 --- a/dw/video_extensions.py +++ b/dw/video_extensions.py @@ -25,10 +25,6 @@ from .for_each import MEMBER_SEPARATOR, render_path from .security import ALLOWED_IMAGE_EXTENSIONS, ALLOWED_VIDEO_EXTENSIONS -# Left to the run-time check: not yet resolved to anything an extension can -# be read off, at the point validation walks the expanded definition -_UNRESOLVED_PREFIXES = references.UNRESOLVED - def _is_video_key(key): return isinstance(key, str) and (key == "video" or key.endswith("_video")) @@ -41,7 +37,9 @@ def _extension_problem(value): return None if value.startswith(("http://", "https://")): return None - if value.startswith(_UNRESOLVED_PREFIXES): + # Left to the run-time check: not yet resolved to anything an extension + # can be read off, at the point validation walks the expanded definition + if references.is_ref(references.UNRESOLVED, value): return None if references.is_ref((references.CONSTANT, references.PROMPT), value): return None @@ -56,7 +54,8 @@ def _extension_problem(value): f"'{value}' is a still image, and a video argument loads video " f"files - pass it as " f'{{"media_type": "image", "location": "{value}"}} to load it as ' - f"a still, or reference a prior image step with previous_result:" + f"a still, or reference a prior image step with " + f"{references.make_ref(references.PREVIOUS_RESULT, '')}" ) return f"Video file extension not allowed: {ext}" diff --git a/dw/vram_estimate.py b/dw/vram_estimate.py index 43115adc..174027f6 100644 --- a/dw/vram_estimate.py +++ b/dw/vram_estimate.py @@ -46,8 +46,7 @@ import numbers -from . import references as ref_prefixes -from .references import FROM_FILE_KEY, FROM_PREVIOUS_RESULT_KEY +from .references import FROM_FILE_KEY, FROM_PREVIOUS_RESULT_KEY, VARIABLE, ref_name from .for_each import FOR_EACH_KEY, MEMBER_SEPARATOR, render_path KEY = "vram_estimate" @@ -58,8 +57,9 @@ def _resolved(value, variables): """A `variable:` reference resolved against a template's own defaults - the index reads templates as written, not substituted.""" - if isinstance(value, str) and value.startswith(ref_prefixes.VARIABLE): - return variables.get(value[len(ref_prefixes.VARIABLE) :]) + name = ref_name(VARIABLE, value) + if name is not None: + return variables.get(name) return value @@ -253,8 +253,9 @@ def _projections(definition, estimate, arguments): def _variable_names(value): """Every name a `variable:` reference inside `value` spells.""" if isinstance(value, str): - if value.startswith(ref_prefixes.VARIABLE): - yield value[len(ref_prefixes.VARIABLE) :] + name = ref_name(VARIABLE, value) + if name is not None: + yield name elif isinstance(value, dict): for item in value.values(): yield from _variable_names(item) @@ -295,8 +296,8 @@ def _where(estimate, index, step, supplied, source_indices, written): _entry_position(source_indices, index) if source_indices is not None else 0 ) entries = written_step[FOR_EACH_KEY] - if isinstance(entries, str) and entries.startswith(ref_prefixes.VARIABLE): - name = entries[len(ref_prefixes.VARIABLE) :] + name = ref_name(VARIABLE, entries) + if name is not None: root = "arguments" if name in supplied else "variables" return f"{root}.{name}[{position}]" return render_path(("steps", source, FOR_EACH_KEY, position)) diff --git a/dw/workflow_run.py b/dw/workflow_run.py index 394ad396..ac7e779c 100644 --- a/dw/workflow_run.py +++ b/dw/workflow_run.py @@ -175,10 +175,9 @@ def selected_field(step_data, selected): position = selected.get("position") if isinstance(position, int) and 0 <= position < len(candidates): candidate = candidates[position] - if isinstance(candidate, str) and candidate.startswith( - references.PREVIOUS_RESULT - ): - field["entry"] = candidate[len(references.PREVIOUS_RESULT) :] + entry = references.ref_name(references.PREVIOUS_RESULT, candidate) + if entry is not None: + field["entry"] = entry return field diff --git a/tests/test_arguments.py b/tests/test_arguments.py index 6df50649..0b6b29c2 100644 --- a/tests/test_arguments.py +++ b/tests/test_arguments.py @@ -1261,6 +1261,14 @@ def test_a_function_is_not_a_constant(self): with pytest.raises(ValueError, match="not a constant"): fetch_constant("constant:os.system") + def test_a_bare_name_reads_as_written(self): + """`fetch_constant` strips a `constant:` prefix and leaves a bare + dotted name alone; the guard (is_constant_reference) is the caller's, + so a direct caller reaches the strip unprefixed.""" + import os + + assert fetch_constant("os.sep") == os.sep == fetch_constant("constant:os.sep") + def test_an_unknown_constant_names_itself(self): with pytest.raises(ValueError, match="no_such_module.NAME"): fetch_constant("constant:no_such_module.NAME") diff --git a/tests/test_assets.py b/tests/test_assets.py index c9a6b535..a79d462b 100644 --- a/tests/test_assets.py +++ b/tests/test_assets.py @@ -72,6 +72,16 @@ def test_a_nested_reference_resolves(self, asset_dir): asset_dir / "gyre" / "frames" / "web.png" ) + def test_a_bare_name_resolves_as_written(self, asset_dir): + """The prefix is stripped when present and a bare name is left alone: + callers guard with is_asset_reference, a direct caller need not.""" + assert resolve_asset_reference("iris.png") == resolve_asset_reference( + "asset:iris.png" + ) + assert resolve_asset_reference(" iris.png ") == resolve_asset_reference( + "asset:iris.png" + ) + def test_a_missing_asset_says_so(self, asset_dir): with pytest.raises(ValueError, match="not found"): resolve_asset_reference("asset:nothing.png") diff --git a/tests/test_output_references.py b/tests/test_output_references.py index a77d9b9d..21d473e9 100644 --- a/tests/test_output_references.py +++ b/tests/test_output_references.py @@ -39,6 +39,14 @@ def outputs(tmp_path): class TestReferences: + def test_a_bare_name_resolves_as_written(self, outputs): + """The prefix is stripped when present and a bare name is left alone: + callers guard with is_output_reference, a direct caller need not.""" + name = "ltx2/Gyre/20260905-101500-aaaaaaaa/still.png" + assert resolve_output_reference(name) == resolve_output_reference( + "output:" + name + ) + def test_a_run_and_file_resolve(self, outputs): assert resolve_output_reference( "output:ltx2/Gyre/20260905-101500-aaaaaaaa/still.png" diff --git a/tests/test_prompt_references.py b/tests/test_prompt_references.py index 7e717cb7..09cbe32a 100644 --- a/tests/test_prompt_references.py +++ b/tests/test_prompt_references.py @@ -12,7 +12,8 @@ import pytest -from dw.prompts import PROMPT_PREFIX, resolve_prompt_reference +from dw.prompts import resolve_prompt_reference +from dw.references import PROMPT, make_ref from dw.server.catalog import collect_prompt_references from tests.test_examples import REPO_ROOT, get_example_files @@ -34,8 +35,7 @@ def get_builtin_files(): def prompt_references(definition): """Every 'prompt:' reference a workflow makes, as written.""" return [ - f"{PROMPT_PREFIX}{name}" - for name in sorted(collect_prompt_references(definition)) + make_ref(PROMPT, name) for name in sorted(collect_prompt_references(definition)) ] diff --git a/tests/test_prompts.py b/tests/test_prompts.py index 61b19bd4..a991fd1f 100644 --- a/tests/test_prompts.py +++ b/tests/test_prompts.py @@ -8,11 +8,11 @@ from dw.arguments import is_prompt_reference, realize_args from dw.prompts import ( - RESERVED_TEXT_PREFIXES, fetch_prompt, get_prompt_dir, load_prompt, ) +from dw.references import RESERVED_TEXT from dw.security import InvalidInputError, SecurityError @@ -166,6 +166,22 @@ def test_text_that_is_itself_a_reference_is_rejected(self, prompt_dir): with pytest.raises(ValueError, match="may not itself be a reference"): fetch_prompt("prompt:sneaky") + def test_the_refusal_lists_the_reserved_prefixes_in_order(self, prompt_dir): + (prompt_dir / "sneaky.json").write_text(json.dumps({"text": "asset:x.png"})) + with pytest.raises(ValueError) as refused: + fetch_prompt("prompt:sneaky") + assert ( + "(previous_result:, variable:, constant:, asset:, output:, prompt:)" + in str(refused.value) + ) + + def test_a_bare_name_resolves_as_written(self, prompt_dir): + """`resolve_prompt_reference` strips a `prompt:` prefix and leaves a + bare name alone: its callers guard with is_prompt_reference, a direct + caller need not. Pinned so moving the strip onto `ref_name` (None for an + unprefixed value) keeps the leniency.""" + assert fetch_prompt("minimax/fox") == fetch_prompt("prompt:minimax/fox") + def test_load_prompt_returns_the_whole_definition(self, prompt_dir): definition = load_prompt(str(prompt_dir / "minimax" / "fox.json")) assert definition["intended_model"] == "minimax-h3" @@ -193,4 +209,4 @@ class TestShippedLibrary: @pytest.mark.parametrize("path", shipped_prompt_files()) def test_shipped_prompt_is_valid(self, path): prompt = load_prompt(os.path.join(REPO_ROOT, path)) - assert not prompt["text"].startswith(RESERVED_TEXT_PREFIXES) + assert not prompt["text"].startswith(RESERVED_TEXT) diff --git a/tests/test_reference_sets.py b/tests/test_reference_sets.py deleted file mode 100644 index 452a0b70..00000000 --- a/tests/test_reference_sets.py +++ /dev/null @@ -1,36 +0,0 @@ -"""Each checker skips exactly the references its pass cannot see yet - the -set it uses is references. itself, not a copy that can drift.""" - -import pytest - -from dw import ( - adapter_compatibility, - content_types, - kernel_availability, - probe_paths, - reference_limits, - reference_names, - references, - step_value_checks, - subfolders, - video_extensions, -) - - -@pytest.mark.parametrize( - "module", - [adapter_compatibility, reference_names, reference_limits, video_extensions], -) -def test_the_unresolved_checkers_share_one_set(module): - assert module._UNRESOLVED_PREFIXES is references.UNRESOLVED - - -@pytest.mark.parametrize( - "module", [content_types, kernel_availability, step_value_checks, subfolders] -) -def test_the_substituted_checkers_still_check_step_results(module): - assert module._UNRESOLVED_PREFIXES is references.SUBSTITUTED - - -def test_probe_paths_public_name_is_the_shared_set(): - assert probe_paths.UNRESOLVED_PREFIXES is references.UNRESOLVED diff --git a/tests/test_references.py b/tests/test_references.py index 29a33f11..957b1f8e 100644 --- a/tests/test_references.py +++ b/tests/test_references.py @@ -1,6 +1,7 @@ """dw/references.py - the one place a reference prefix is spelled.""" import ast +import json import pathlib import pytest @@ -8,6 +9,7 @@ from dw import references from dw.references import ( ASSET, + CONSTRAINT, DEFERRED, GATHER, ITEM, @@ -88,3 +90,31 @@ def test_references_imports_nothing_from_dw(): def test_render_path_writes_a_json_path_the_way_schema_errors_do(): path = ["steps", 3, "task", "arguments", "videos", 1] assert render_path(path) == "steps[3].task.arguments.videos[1]" + + +def test_the_schema_patterns_spell_the_prefixes_references_owns(): + """Characterization: the JSON schema is data and keeps its own regexes, so + this fails only when a `^variable:` or `^constraint:` pattern drifts from + the constant it stands for (or one is added that begins another way).""" + schema = pathlib.Path(references.__file__).with_name("workflow_schema.json") + patterns = [] + + def collect(node): + if isinstance(node, dict): + for key, value in node.items(): + if key == "pattern" and isinstance(value, str): + patterns.append(value) + else: + collect(value) + elif isinstance(node, list): + for item in node: + collect(item) + + collect(json.loads(schema.read_text(encoding="utf-8"))) + spelled = 0 + for pattern in patterns: + for kind in (VARIABLE, CONSTRAINT): + if pattern.startswith("^" + kind.rstrip(":")): + spelled += 1 + assert pattern.startswith("^" + kind), pattern + assert spelled, "no schema pattern names a reference prefix any more" diff --git a/tests/test_server_guides.py b/tests/test_server_guides.py index 8638239a..c978034c 100644 --- a/tests/test_server_guides.py +++ b/tests/test_server_guides.py @@ -264,13 +264,13 @@ def test_the_authoring_section_names_every_reference_prefix(self): """The prefixes the engine reserves are the ones the section has to explain; a new prefix added to the engine fails here until it is written up.""" - from dw.prompts import RESERVED_TEXT_PREFIXES + from dw.references import RESERVED_TEXT content = guides.get_guide( "workflows", section="Authoring a workflow from an agent" )["content"] - for prefix in RESERVED_TEXT_PREFIXES: + for prefix in RESERVED_TEXT: assert f"`{prefix}`" in content, prefix def test_the_authoring_section_states_the_cartesian_rule_and_the_loop(self): From f4e80d9ea63624575d4dca2fbe0f9f92347f63b2 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 13:00:34 -0500 Subject: [PATCH 13/41] refactor(references): prose messages stay literal, skip-set test, drop SHOT_REFERENCE_PREFIX Co-Authored-By: Claude Opus 5.5 --- docs/stabilization/phase-4.md | 2 +- dw/arguments.py | 4 +- dw/shots.py | 14 ++-- dw/video_extensions.py | 3 +- scripts/arch_metrics.py | 4 +- tests/test_reference_skip_sets.py | 108 ++++++++++++++++++++++++++++++ 6 files changed, 119 insertions(+), 16 deletions(-) create mode 100644 tests/test_reference_skip_sets.py diff --git a/docs/stabilization/phase-4.md b/docs/stabilization/phase-4.md index 118d081c..f37cc83b 100644 --- a/docs/stabilization/phase-4.md +++ b/docs/stabilization/phase-4.md @@ -144,7 +144,7 @@ Work on branch `stabilization/phase-4a` in the worktree, from `develop` at `e768 - The cache snapshot is the cache's *key*. A shared leaf mutated after the snapshot changes the stored key with it, and `deep_equal`'s `a is b` shortcut then reports a hit for a changed input. - What staying costs: memory and time per member and per cached step. That is measured, not assumed: the gate 4 lem timing runs a `for_each` over media. If it shows a real cost, sharing becomes its own design with a mutation audit, after the freeze. - **References are built and read only through `references.py`.** - - New in `references.py`: `RESERVED_TEXT` (the six-tuple `prompts.py` owns today) and `LAZY_MEDIA = (PREVIOUS_RESULT, VARIABLE)` (the pair spelled inline three times). `SHOT_REFERENCE_PREFIX` stays in `shots.py` as `make_ref(PREVIOUS_RESULT, "shot" + MEMBER_SEPARATOR)`: it is a shots concept built from the owned parts. + - New in `references.py`: `RESERVED_TEXT` (the six-tuple `prompts.py` owns today) and `LAZY_MEDIA = (PREVIOUS_RESULT, VARIABLE)` (the pair spelled inline three times). `SHOT_REFERENCE_PREFIX` is deleted: its branch in `shots.py` had a body identical to the plain `previous_result:` branch (a `shot@` member and a step are both read as `ref_name(...).split(".", 1)[0]`), so the branches merged and the constant went. - Every alias constant is deleted, and every caller uses `references.` or a helper. `_UNRESOLVED_PREFIXES` goes away with them, so no module names a tuple after something it isn't. - `x.removeprefix(K).strip()` becomes `ref_name(K, x).strip()`. `ref_name` gets no strip variant: whitespace hygiene is the caller's rule. Where a site relied on `removeprefix` returning an unprefixed value unchanged, the task checks its guard and keeps the semantics. `ref_name` returns `None` there. - The JSON schema's patterns stay (a schema is data). A new test pins each `pattern` that begins `^variable:` or `^constraint:` to `references.VARIABLE` / `references.CONSTRAINT`. diff --git a/dw/arguments.py b/dw/arguments.py index c61a716d..9471bf13 100644 --- a/dw/arguments.py +++ b/dw/arguments.py @@ -330,8 +330,8 @@ def fetch_constant(reference): if callable(value): raise ValueError( f"'{name}' is a {type(value).__name__}, not a constant - " - f"'{references.make_ref(references.CONSTANT, '')}' reads a value, " - f"and a type is named with a '_type' argument instead" + f"'constant:' reads a value, and a type is named with a " + f"'_type' argument instead" ) logger.info(f"Reading constant {name}") diff --git a/dw/shots.py b/dw/shots.py index de1e6479..89b8633b 100644 --- a/dw/shots.py +++ b/dw/shots.py @@ -35,11 +35,7 @@ import copy -from .references import MEMBER_SEPARATOR, PREVIOUS_RESULT, is_ref, make_ref, ref_name - -# The step names a for_each member as `@`; only a member of the -# group conventionally called `shot` names a shot -SHOT_REFERENCE_PREFIX = make_ref(PREVIOUS_RESULT, "shot" + MEMBER_SEPARATOR) +from .references import PREVIOUS_RESULT, ref_name def shot_record( @@ -253,11 +249,9 @@ def shot_reference_names(references): names = [] for reference in references: step = ref_name(PREVIOUS_RESULT, reference) - if is_ref(SHOT_REFERENCE_PREFIX, reference): - # `previous_result:shot@x.field` names the member, not the field - names.append(step.split(".", 1)[0]) - elif step is not None: - # `previous_result:step.field` names the step, not the field + if step is not None: + # `previous_result:step.field` and `previous_result:shot@x.field` + # name the step or member, not the field names.append(step.split(".", 1)[0]) else: names.append(None) diff --git a/dw/video_extensions.py b/dw/video_extensions.py index 621a9d04..ba14f386 100644 --- a/dw/video_extensions.py +++ b/dw/video_extensions.py @@ -54,8 +54,7 @@ def _extension_problem(value): f"'{value}' is a still image, and a video argument loads video " f"files - pass it as " f'{{"media_type": "image", "location": "{value}"}} to load it as ' - f"a still, or reference a prior image step with " - f"{references.make_ref(references.PREVIOUS_RESULT, '')}" + f"a still, or reference a prior image step with 'previous_result:'" ) return f"Video file extension not allowed: {ext}" diff --git a/scripts/arch_metrics.py b/scripts/arch_metrics.py index e3419878..16c95401 100644 --- a/scripts/arch_metrics.py +++ b/scripts/arch_metrics.py @@ -27,7 +27,9 @@ with one as an operand, or an f-string splicing one in; (d) a module-level assignment of one (or a tuple holding one); (e) a non-docstring string that starts with a prefix and is longer than it, or an f-string fragment that - ends with a prefix. Prose naming a bare prefix is not counted. + ends with a prefix. Prose naming a bare prefix is not counted. A table keyed by reference + kinds (e.g. reference_names._KINDS, a dict) is not an alias and is deliberately + not counted under form (d). """ import argparse diff --git a/tests/test_reference_skip_sets.py b/tests/test_reference_skip_sets.py new file mode 100644 index 00000000..5441f483 --- /dev/null +++ b/tests/test_reference_skip_sets.py @@ -0,0 +1,108 @@ +"""Which reference set each checker skips, observed through its real entry +point rather than by comparing a constant. + +A checker that runs on the substituted definition skips values still spelled +`variable:`/`item:` (SUBSTITUTED) or, when it also cannot see an earlier +step's result, `previous_result:`/`gather:` (UNRESOLVED). Each probe answers +whether the checker acted on a value, so swapping one site's set for the +other changes an answer here. +""" + +import pytest + +from dw import ( + adapter_compatibility, + content_types, + kernel_availability, + probe_paths, + reference_limits, + reference_names, + step_value_checks, + subfolders, + type_helpers, + video_extensions, +) + +_STEP_CHECKS = { + "content_type": content_types.content_type_errors, + "fps": step_value_checks.fps_errors, + "subfolder": subfolders.subfolder_errors, +} + + +def _step_result(key): + def probe(v, **_): + return bool(_STEP_CHECKS[key]({"steps": [{"name": "s", "result": {key: v}}]})) + + return probe + + +def _kernel(v, monkeypatch, **_): + monkeypatch.setattr(kernel_availability, "_fault_for_name", lambda name: "fault") + return bool(kernel_availability.kernel_availability_fault(v)) + + +def _adapter(v, **_): + step = { + "name": "s", + "pipeline": { + "from_pretrained_arguments": {"workflow": "ref2va"}, + "loras": [{"model_name": "org/lora", "weight_name": v}], + }, + } + return bool(adapter_compatibility.adapter_warnings({"steps": [step]})) + + +def _names(v, **_): + return reference_names.reference_fault("output:" + v) is not None + + +def _limits(v, monkeypatch, **_): + monkeypatch.setattr(type_helpers, "load_type_from_full_name", lambda *a: object) + return ( + reference_limits._reference_class({"reference_type": v + ".Kind"}) is not None + ) + + +def _video(v, **_): + step = {"name": "s", "task": {"arguments": {"video": v + "/clip.zip"}}} + return bool(video_extensions.video_extension_errors({"steps": [step]})) + + +def _probe(v, tmp_path, **_): + (tmp_path / (v + ".png")).write_bytes(b"x") + return probe_paths.resolve_probe_path(v + ".png", str(tmp_path)) is not None + + +SUBSTITUTED_ONLY = { + "content_types": _step_result("content_type"), + "step_value_checks": _step_result("fps"), + "subfolders": _step_result("subfolder"), + "kernel_availability": _kernel, +} +UNRESOLVED_TOO = { + "adapter_compatibility": _adapter, + "reference_names": _names, + "reference_limits": _limits, + "video_extensions": _video, + "probe_paths": _probe, +} + + +@pytest.mark.parametrize("checker", sorted(SUBSTITUTED_ONLY)) +def test_a_substituted_only_checker_still_reports_an_earlier_steps_reference( + checker, monkeypatch, tmp_path +): + probe = SUBSTITUTED_ONLY[checker] + assert probe("previous_result:x", monkeypatch=monkeypatch, tmp_path=tmp_path) + assert not probe("variable:x", monkeypatch=monkeypatch, tmp_path=tmp_path) + + +@pytest.mark.parametrize("checker", sorted(UNRESOLVED_TOO)) +def test_an_unresolved_checker_leaves_an_earlier_steps_reference_alone( + checker, monkeypatch, tmp_path +): + probe = UNRESOLVED_TOO[checker] + assert not probe("previous_result:x", monkeypatch=monkeypatch, tmp_path=tmp_path) + assert not probe("gather:x", monkeypatch=monkeypatch, tmp_path=tmp_path) + assert not probe("variable:x", monkeypatch=monkeypatch, tmp_path=tmp_path) From 3975e8de13f118a71243509423118967f811cbc0 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 13:10:44 -0500 Subject: [PATCH 14/41] chore(stabilization): stage 4a merge prep Final-review minors: validation docstrings no longer describe the deleted Workflow delegators; dead library.catalog_root_dir removed; CLAUDE.md names workflow_run.cache_hits; the gain_audio release line says 'one sample off', not 'short'. Hot zone back to the standing entries; ROADMAP Phase 4 row says 4a merged. Co-Authored-By: Claude Opus 5.5 --- CLAUDE.md | 2 +- docs/RELEASING.md | 2 +- docs/stabilization/ROADMAP.md | 2 +- docs/stabilization/hot-zone.txt | 46 +-------------------------------- dw/library.py | 17 +----------- dw/validation.py | 21 ++++++++------- scripts/surface_snapshot.py | 2 +- 7 files changed, 17 insertions(+), 75 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 177896f6..eae3d8ed 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -341,7 +341,7 @@ same reason - default setup cannot load a pack. with the current plan when the fingerprint or the required downloads changed; `minutes` never compared), and the job records `acknowledged: none | boolean | bound`. `cached_steps` is the worker's answer to a - `probe_cache` command (`Workflow.cache_hits`, which shares + `probe_cache` command (`workflow_run.cache_hits`, which shares `prepare_definition` / `cache_lookup` (`dw/workflow_run.py`) with `run` so the two cannot drift). The web UI reads the fields only: the editor lists the plan under a valid verdict (`describePlan`, `ui/src/lib/plan.ts`), and a job queued `bound` diff --git a/docs/RELEASING.md b/docs/RELEASING.md index 84793b1c..b5aee411 100644 --- a/docs/RELEASING.md +++ b/docs/RELEASING.md @@ -16,7 +16,7 @@ when a release ships. - A run that fails before it opens its run directory no longer rewrites the previous run's `manifest.json` when the same workflow instance is reused: `Workflow.run` resets the directory and version it carried. - The per-variant lines of a kernels "Cannot find a build variant" error are sorted by dw (`kernel_availability.stable_message`), so the message no longer varies by process. - The `argument_template` schema description now says what the code does: handed arguments are held on the child at run time, never written into the definition, and an authored value is the fallback. -- `gain_audio` rounds a frame-addressed region's end once, as `slice_audio` does, so a region can no longer end one sample short of the matching slice. +- `gain_audio` rounds a frame-addressed region's end once, as `slice_audio` does, so a region's end can no longer be one sample off the matching slice's. - `concat_videos` refuses a track with no sample rate (`concat_videos: '' has audio with no sample rate`) instead of joining it unresampled at the wrong speed and pitch; `dissolve_videos` gives the same message in place of the resample error. Save that step with `audio_sample_rate` in its result and join the saved file through an `output:` reference. An unpinned `dissolve_videos` with such a track now raises this `ValueError` rather than a `TypeError`. ### 0.6.0 diff --git a/docs/stabilization/ROADMAP.md b/docs/stabilization/ROADMAP.md index 71df3aa8..70b32f19 100644 --- a/docs/stabilization/ROADMAP.md +++ b/docs/stabilization/ROADMAP.md @@ -14,7 +14,7 @@ phase works on is what the earlier phases leave behind. | 1 | Metrics v2 first (see below); remove the REPL; one prepare pipeline; one admission service; `dw.run` becomes a thin client of `dw.serve` | Validation sees the definition the run sees; the server admits a request once (one `Workflow`, one expansion); every entry point reaches the worker through the server; ratchets re-baselined | [phase-1.md](phase-1.md) | done 2026-09-28 (`stabilization-gate-1`) | | 2 | Seams in place: `references.py`, validation context + check registry, shared task rules, step cache, typed worker protocol | `validation_errors` is a registry loop; no prefix literals outside `references.py` | [phase-2.md](phase-2.md) (staged: 2a-2d) | done 2026-09-30 (`stabilization-gate-2`) | | 3 | Structural moves: `app.py` routers + services, `LibraryPath`, split `result.py` / `pipeline.py`, one media + dsp module | No module over 1,000 lines, no function over 150; suite and lem smoke green | [phase-3.md](phase-3.md) (staged: 3a-3e) | done 2026-10-01 (`stabilization-gate-3`) | -| 4 | Carried fixes; guardrails installed; context diet (root CLAUDE.md <= 150 lines, every CLAUDE.md triaged; was "<= 250 total", Don 2026-10-01) | Guardrails live in dw CI and the harness; freeze lifted | [phase-4.md](phase-4.md) (staged: 4a-4d) | 4a in progress | +| 4 | Carried fixes; guardrails installed; context diet (root CLAUDE.md <= 150 lines, every CLAUDE.md triaged; was "<= 250 total", Don 2026-10-01) | Guardrails live in dw CI and the harness; freeze lifted | [phase-4.md](phase-4.md) (staged: 4a-4d) | 4a merged 2026-10-01 | ## Metrics diff --git a/docs/stabilization/hot-zone.txt b/docs/stabilization/hot-zone.txt index c15fd484..929f02a4 100644 --- a/docs/stabilization/hot-zone.txt +++ b/docs/stabilization/hot-zone.txt @@ -2,50 +2,6 @@ # The harness implementer must not change these; a field bug whose fix # needs one is labelled `stabilization` and handed to owner:don. # One path (or directory ending in /) per line; '#' starts a comment. -# Stage 4a (carried code) live from 2026-10-01; plan: docs/stabilization/phase-4.md -dw/workflow.py -dw/workflow_run.py -dw/validation.py -dw/library.py -dw/realize.py -dw/worker.py -dw/kernel_availability.py -dw/workflow_schema.json -dw/references.py -dw/tasks/audio_utils.py -dw/tasks/joins.py -dw/tasks/concat_videos.py -dw/tasks/dissolve_videos.py -dw/server/admission.py -dw/server/routes/library.py -dw/server/routes/jobs.py -dw/server/routes/assets.py -dw/server/catalog.py -dw/server/enhancers.py -dw/server/exports.py -dw/server/jobs.py -dw/server/outputs.py -dw/adapter_compatibility.py -dw/argument_media.py -dw/arguments.py -dw/assets.py -dw/content_types.py -dw/elision.py -dw/for_each.py -dw/locations.py -dw/plan.py -dw/probe_paths.py -dw/prompts.py -dw/reference_limits.py -dw/reference_names.py -dw/runs.py -dw/shots.py -dw/step_value_checks.py -dw/subfolders.py -dw/variable_constraints.py -dw/video_extensions.py -dw/vram_estimate.py -dw/server/catalog_shape.py +# Stage 4a merged (2026-10-01); stage 4b's list goes live when it is detailed. scripts/arch_metrics.py -scripts/surface_snapshot.py docs/stabilization/ diff --git a/dw/library.py b/dw/library.py index 4b89f561..82b6893a 100644 --- a/dw/library.py +++ b/dw/library.py @@ -81,8 +81,7 @@ def catalog_root(directory): The root a run with no workflow_dir of its own confines a relative sub-workflow reference to, so a template under templates/ can still climb - to a sibling models/ without leaving the catalog. `catalog_root_dir` - (below) is this rule asked for a file rather than a directory. + to a sibling models/ without leaving the catalog. """ directory = os.path.normpath(os.path.abspath(directory)) parts = directory.split(os.sep) @@ -118,20 +117,6 @@ def workflow_output_subfolder(file_spec): return os.path.join(*parts[index + 1 :]) if index + 1 < len(parts) else "" -def catalog_root_dir(file_spec): - """The nearest ancestor directory literally named 'workflows' of - file_spec, else file_spec's own directory. - - Used to confine a relative sub-workflow reference when a run carries no - workflow_dir of its own (an unconfined CLI run) - the same "last - 'workflows' segment" rule workflow_output_subfolder uses for output - naming, but returning the directory itself rather than what sits under - it. It is `catalog_root` asked for a file rather than a directory, so - the resolver (resolve_sub_workflow) confines to exactly this root. - """ - return catalog_root(os.path.dirname(os.path.abspath(file_spec))) - - class LibraryRoot: """One root on a library's search path.""" diff --git a/dw/validation.py b/dw/validation.py index ca1182ba..d2f4f686 100644 --- a/dw/validation.py +++ b/dw/validation.py @@ -17,10 +17,11 @@ it was built for (B9). It is never stored on the Workflow, so a replaced asset is probed afresh by the next request. -The Workflow's validation methods are one-line calls into the functions at the -end of this module (`workflow_errors`, `workflow_context`, -`run_warning_check`, `undeclared_variable_errors`, `sub_workflow_errors`, -`sub_workflow_argument_warnings`). The three step-value checkers the error +Callers use the functions at the end of this module directly +(`workflow_errors`, `workflow_context`, `run_warning_check`, +`undeclared_variable_errors`, `sub_workflow_errors`, +`sub_workflow_argument_warnings`); only `Workflow.validation_errors` and +`Workflow.validate` remain as methods. The three step-value checkers the error registry runs (`fps_errors`, `null_media_errors`, `select_errors`) are in dw/step_value_checks.py. """ @@ -480,8 +481,8 @@ def _is_seeded(definition, arguments): # Every warning source admit() reports that does not need the plan, in the # order admit() listed them - the response's order. Each returns today's # "path: message" strings. A source that raises is one internal warning and -# never refuses (B10); the six the Workflow also answers as methods run the -# same entry, so a direct call and a request agree. +# never refuses (B10); `run_warning_check` runs the same entry, so a direct +# call and a request agree. # # Four read the definition as written, with the caller's arguments over its # defaults; the rest read the request's one expansion. `arguments` may be @@ -591,14 +592,14 @@ def _null_variable_argument_warnings(context): def warning_check(name): """The warning registry's entry called `name`, looked up at call time so - a Workflow method runs whatever the registry currently holds.""" + a direct call runs whatever the registry currently holds.""" check = next((check for check in WARNING_CHECKS if check.name == name), None) if check is None: raise KeyError(name) return check -# --- What Workflow's validation methods call --------------------------------- +# --- Entry points over one Workflow ------------------------------------------- # # `workflow` is a Workflow handed in as a value: this module never imports # dw.workflow. A composed child is opened through `Workflow.open_sub_workflow`. @@ -706,8 +707,8 @@ def workflow_errors(workflow, arguments=None, composing=None, context=None): def run_warning_check(workflow, name, arguments, **context_fields): """The registry's warning check `name` over a context of this - call's own - how the Workflow's warning methods answer when called - directly rather than through admit(). An expansion that fails + call's own - how one check answers when called directly rather + than through admit(). An expansion that fails raises inside the check, which makes it one internal warning.""" context = workflow_context(workflow, arguments, **context_fields) check = warning_check(name) diff --git a/scripts/surface_snapshot.py b/scripts/surface_snapshot.py index 092b5843..a4eda840 100644 --- a/scripts/surface_snapshot.py +++ b/scripts/surface_snapshot.py @@ -214,7 +214,7 @@ def catalog_validation(root): """`validation_errors()` and the warning checks that need no server state (no `ceiling_index`, no observed costs) for every JSON under `workflows/` and `dw/workflows/`. Keyed by repo-relative path. A same-machine - comparison: `validation_context` reads the device. A file that will not + comparison: `workflow_context` reads the device. A file that will not load, or a check that raises, is recorded as its error string.""" repo = Path(__file__).resolve().parent.parent files = sorted( From 3bf93a09fd959992931daba251f271a7b660b198 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 13:15:42 -0500 Subject: [PATCH 15/41] docs(stabilization): stage 4b detail and hot zone Co-Authored-By: Claude Opus 5.5 --- docs/stabilization/ROADMAP.md | 2 +- docs/stabilization/hot-zone.txt | 6 ++- docs/stabilization/phase-4.md | 83 +++++++++++++++++++++++++++++++-- 3 files changed, 86 insertions(+), 5 deletions(-) diff --git a/docs/stabilization/ROADMAP.md b/docs/stabilization/ROADMAP.md index 70b32f19..4c7acdd3 100644 --- a/docs/stabilization/ROADMAP.md +++ b/docs/stabilization/ROADMAP.md @@ -14,7 +14,7 @@ phase works on is what the earlier phases leave behind. | 1 | Metrics v2 first (see below); remove the REPL; one prepare pipeline; one admission service; `dw.run` becomes a thin client of `dw.serve` | Validation sees the definition the run sees; the server admits a request once (one `Workflow`, one expansion); every entry point reaches the worker through the server; ratchets re-baselined | [phase-1.md](phase-1.md) | done 2026-09-28 (`stabilization-gate-1`) | | 2 | Seams in place: `references.py`, validation context + check registry, shared task rules, step cache, typed worker protocol | `validation_errors` is a registry loop; no prefix literals outside `references.py` | [phase-2.md](phase-2.md) (staged: 2a-2d) | done 2026-09-30 (`stabilization-gate-2`) | | 3 | Structural moves: `app.py` routers + services, `LibraryPath`, split `result.py` / `pipeline.py`, one media + dsp module | No module over 1,000 lines, no function over 150; suite and lem smoke green | [phase-3.md](phase-3.md) (staged: 3a-3e) | done 2026-10-01 (`stabilization-gate-3`) | -| 4 | Carried fixes; guardrails installed; context diet (root CLAUDE.md <= 150 lines, every CLAUDE.md triaged; was "<= 250 total", Don 2026-10-01) | Guardrails live in dw CI and the harness; freeze lifted | [phase-4.md](phase-4.md) (staged: 4a-4d) | 4a merged 2026-10-01 | +| 4 | Carried fixes; guardrails installed; context diet (root CLAUDE.md <= 150 lines, every CLAUDE.md triaged; was "<= 250 total", Don 2026-10-01) | Guardrails live in dw CI and the harness; freeze lifted | [phase-4.md](phase-4.md) (staged: 4a-4d) | 4a merged 2026-10-01; 4b in progress | ## Metrics diff --git a/docs/stabilization/hot-zone.txt b/docs/stabilization/hot-zone.txt index 929f02a4..5ec181d4 100644 --- a/docs/stabilization/hot-zone.txt +++ b/docs/stabilization/hot-zone.txt @@ -2,6 +2,10 @@ # The harness implementer must not change these; a field bug whose fix # needs one is labelled `stabilization` and handed to owner:don. # One path (or directory ending in /) per line; '#' starts a comment. -# Stage 4a merged (2026-10-01); stage 4b's list goes live when it is detailed. +# Stage 4b (guardrails in dw) live from 2026-10-01; plan: docs/stabilization/phase-4.md scripts/arch_metrics.py +scripts/arch_report.py +scripts/preflight.sh +.github/workflows/ci.yml +tests/test_arch_metrics.py docs/stabilization/ diff --git a/docs/stabilization/phase-4.md b/docs/stabilization/phase-4.md index f37cc83b..99887a37 100644 --- a/docs/stabilization/phase-4.md +++ b/docs/stabilization/phase-4.md @@ -27,7 +27,7 @@ | Stage | Scope | Done when | | --- | --- | --- | | 4a | **Carried code.** The behaviour items the freeze parked (each with a failing-first test and a 0.7.0 release note); `Workflow`'s 8 delegators replaced by their module functions; the remaining prefix spellings moved onto `references.py`'s helpers, with a metric that counts them. | Every "Carried to Phase 4" code item in ROADMAP.md, Gate 3, is fixed or ruled out in Decisions; `Workflow` LCOM4 is 1; the new prefix metric is 0 and ratcheted | -| 4b | **Guardrails in dw.** The size rule gets its warn band; `arch_metrics.py --check` runs in CI's `backend` job and in `preflight.sh`; the re-baseline rule is written down where the check prints it. | A PR that raises any ratchet fails CI; a module growing past 1,000 lines warns and does not fail until the ceiling | +| 4b | **Guardrails in dw.** The size rule gets its warn band; `arch_metrics.py --check` runs in CI's `backend` job and in `preflight.sh`; the re-baseline rule is written down where the check prints it. | A PR that raises any ratchet fails CI (a direct push to `develop` goes red after landing; the harness's gate is what blocks those); a module growing past 1,000 lines warns and does not fail until the ceiling | | 4c | **Seam map, then the context diet.** `docs/ARCHITECTURE.md`: concept → owning module → the rule, one row each. Then every CLAUDE.md is cut to a map: each paragraph is deleted (the knowledge is already in a doc, a docstring or a test), moved into the owning module's docstring, or moved into the seam map. `.github/copilot-instructions.md` becomes a pointer. | the root CLAUDE.md <= 150 lines, every CLAUDE.md through the triage, `claude_md_lines` re-baselined; every plugin skill still named in the root CLAUDE.md; no paragraph moved to `docs/` verbatim without a ruling | | 4d | **Gate 4.** The harness stage C prompt (`harness/stage-c-guardrails.md`), which Don runs; then `FREEZE` deleted and `hot-zone.txt` emptied; the metrics report, tag, re-baseline, lem deploy, ROADMAP and ASSESSMENT refreshed. | Stage C is committed in the harness, *then* the freeze is lifted on `develop` | @@ -74,7 +74,7 @@ From ROADMAP.md, Gate 3, "Carried to Phase 4", plus 3e's carried list. Each item - `--check` (and the default output) also prints a non-failing `warning:` line for every module between 1,001 and 1,100 lines, naming it and its length. A warning is not a metric, so it never enters `baseline.json` and the harness's key-by-key comparison never sees it. - `functions_over_150_lines` stays a cliff. No ruling asked for a band there, and a 150-line function is already four screens. - Cost if wrong: a module can sit at 1,099 indefinitely. The warning makes that visible on every run; tightening is a one-line change. - - The key rename is safe for stage B's harness ratchet: `regressions()` skips a key missing on either side, so the comparison across the rename sees neither key for one session and both sides of it afterwards. + - The key rename is safe for stage B's harness ratchet: stage B measures both trees with `origin/develop`'s script, so both sides carry the new key from the moment the rename lands. - **The re-baseline rule.** `baseline.json` may be lowered by any commit that improves a ratchet, and is rewritten at every gate. It is raised only by a commit that names the rise and why in its message, and from 4d on the harness refuses that unless the issue carries `arch-approved` (stage C). The check prints this rule when it fails. - **Release: open.** Phase 4 changes behaviour (4a) and develop is 0.7.0-alpha.1. Whether gate 4 ships 0.7.0 is Don's call at the gate; 4a collects the notes under `### 0.7.0` either way. @@ -288,7 +288,84 @@ It is a long list because the prefix migration touches a line or two in 31 modul ## Stage 4b: guardrails in dw -Detailed when 4a merges. +Work on branch `stabilization/phase-4b` in the worktree, from `develop` at `4bb997c5` (the 4a merge) or later. + +**What exists (at `4bb997c5`).** +- `scripts/arch_metrics.py`: + - `measure()` counts `modules_over_1000_lines` with a bare `> 1000`; + - `regressions(current, baseline)` is a pure key-by-key "number rose", skipping a key missing on either side; + - `main()` prints the metrics JSON, then each regression on `--check`, and exits 1 on any. + - Nothing else is printed: no warnings, and no hint of what to do on failure. +- Other places that name `modules_over_1000_lines`: `scripts/arch_report.py:32` (a row label) and `tests/test_arch_metrics.py:42`. The harness's stage B compares whatever keys both sides have. +- `.github/workflows/ci.yml`'s `backend` job installs `requirements.txt` + `requirements-test.txt` (`-e .[dev]`, which carries grimp via import-linter, networkx, pylint and pygount), then runs ruff format/check on `dw dw_mcp tests` and `pytest -q`. It runs on push to `master`/`develop` and on every PR. +- `scripts/preflight.sh` runs `ruff format .` and `ruff check . --fix` over the whole repo, then pytest, integration and the UI preflight. Whole-repo `ruff format` rewrites the Python code blocks in `docs/**/*.md` (gate 3 had to revert those by hand). CI formats only `dw dw_mcp tests`. +- After 4a, the modules nearest the line are `workflow.py` (about 945), `workflow_run.py` 972, `arguments.py` 954 and `tasks/task.py` 947. + +### Decisions (4b) + +- **`modules_over_size_ceiling` replaces `modules_over_1000_lines`,** at `SIZE_CEILING = 1100`, with `SIZE_WARNING = 1000` beside it (the warn band; frame Decisions). `measure()` stays pure numbers. A new `size_warnings(root)` returns `[(path, lines)]` for modules in the band, and `main()` prints each as `warning: dw/x.py is 1,043 lines (warn above 1,000, fail above 1,100)` on every run, `--check` included, without changing the exit code. `arch_report.py`'s row is renamed to match; it measures every column with today's script, so earlier gates read on the new key. +- **The check says what to do when it fails.** After the regressions, `--check` prints dw's re-baseline rule in two lines: + - lower `baseline.json` freely when a ratchet improves; + - raise it only in a commit whose message names the rise and why. + + The harness enforces `arch-approved` itself; dw's script doesn't name it. +- **CI: one step in `backend`, after the tests:** `python scripts/arch_metrics.py --check docs/stabilization/baseline.json`. + - It fails closed: `import_graph` runs grimp in a subprocess with `check=True`, so an ImportError there raises `CalledProcessError`, the script exits non-zero and the step fails. Task 1 pins that with a test. The subprocess runs with `cwd=root`, so a `grimp.py` that raises, dropped in the fixture tree's root, shadows the real one. + - The step adds no install, because the dev extra is already there. + - Cost: `--check` takes 7 s on the real tree (measured 2026-10-01), small next to the tests. + - What it blocks and what it doesn't: a PR that raises a ratchet goes red before it merges. A direct push to `develop` (the harness and Don push it directly) goes red *after* it lands, so CI reports there rather than blocks. The blocker for direct pushes is the harness's stage B hand-off gate, and stage C after it. +- **Preflight runs the same check, and formats what CI formats.** + - It gains a `run_step "architecture ratchet"` step. + - `ruff format` / `ruff check --fix` move from `.` to `dw dw_mcp tests scripts`, so preflight stops rewriting the docs' code blocks. CI's own format and lint steps gain `scripts` to match. +- **`baseline.json` stays at `docs/stabilization/baseline.json`.** Moving it after the freeze would break the harness's stage B, which reads it by that path. 4d may revisit, with stage C. + +### Review Focus (4b) + +1. **A module in the band warns and passes, and one over the ceiling fails.** Fixture modules at 1,000 lines (no warning), 1,001 (warning, exit 0) and 1,101 (`modules_over_size_ceiling: 0 -> 1`, exit 1). The test reads `main()`'s output and exit code. +2. **The check fails closed.** With a `grimp.py` that raises ImportError in the fixture tree's root (the subprocess runs there, so it shadows the real one), `--check` exits non-zero. It is never a pass with a skipped metric. +3. **CI's step can actually fail.** On the branch, before the merge, one throwaway commit raises a ratchet (an extra module) and is pushed to a draft PR, and the run is red at that step. A plain revert commit follows; no force-push. + +### Before Task 1: the hot zone goes live + +Commit the 4b list below into `docs/stabilization/hot-zone.txt` on `develop` with this section, push, and branch `stabilization/phase-4b` from that commit. + +### Task 1: The size band and the failing check's message + +- [ ] **Step 1:** Write the Review Focus 1 and 2 tests in `tests/test_arch_metrics.py`, plus one asserting that `--check` prints the two-line re-baseline rule only when it fails. + - The existing `test_a_long_function_and_a_long_module_are_counted` builds about 1,052 lines, which is inside the new band. Grow it past 1,100 so it stays the "counted" case. Add a separate 1,001-1,100-line fixture for "warns, exit 0". Run them; they fail on today's script (no key, no warning, no rule text). +- [ ] **Step 2:** Implement the Decisions (4b) items in `scripts/arch_metrics.py`: the key rename, `SIZE_CEILING` and `SIZE_WARNING`, `size_warnings`, and the printing in `main()`. Rename the row in `arch_report.py`, and update `tests/test_arch_metrics.py:42` to the new key. Update the script's docstring. +- [ ] **Step 3:** Run it on the worktree. `modules_over_size_ceiling` is 0 and no module is in the band. Re-baseline (`--write`). The diff is the key rename only, with the same value 0. +- [ ] **Step 4:** ROADMAP.md's Metrics paragraph: one line for the warn band. The suite and ruff pass. Commit, with a message that names the key rename (the harness's stage B sees it). + +### Task 2: CI and preflight + +- [ ] **Step 1:** `ci.yml`'s `backend` job: + - the new step after `Tests`; + - `Format check` and `Lint` widened to `dw dw_mcp tests scripts`. Run `ruff format --check scripts` locally first and fix anything it reports in the same commit. +- [ ] **Step 2:** `preflight.sh`: the ruff steps scoped to `dw dw_mcp tests scripts`, and a `run_step "architecture ratchet" python scripts/arch_metrics.py --check docs/stabilization/baseline.json` step after the pytest steps. Its header comment names the new step. +- [ ] **Step 3:** Time `arch_metrics.py --check` on the worktree and record it in the report. +- [ ] **Step 4:** Run `scripts/preflight.sh` in full and confirm it leaves no edits under `docs/` (`git status`). +- [ ] **Step 5:** Commit. Push the branch and open a draft PR against `develop`. Confirm the new CI step runs and passes. +- [ ] **Step 6:** Review Focus 3: push one throwaway commit that adds an empty `dw/_ratchet_probe.py`. Confirm the run fails at the ratchet step with `modules: 165 -> 166`. Then push a plain `git revert` of it and confirm green. Record both run URLs in the report. The merge in Task 3 carries the probe and its revert. That's harmless, but say so in the merge commit. + +### Task 3: Stage 4b merge + +- [ ] **Step 1:** `arch_metrics.py --check` passes; the baseline diff since 4a is the key rename only. +- [ ] **Step 2:** Docs: `git grep -n "modules_over_1000_lines" -- ':!docs/stabilization/'` returns nothing. RELEASING.md gets no user-facing line; this is tooling only. +- [ ] **Step 3:** Put the hot zone back to the standing entries. ROADMAP Phase 4 row: "4b merged". +- [ ] **Step 4:** Close the draft PR. Merge `stabilization/phase-4b` to `develop` with `--no-ff` and push. The `develop` CI run is green, including the ratchet step. Report to Don. +- [ ] **Step 5:** Detail stage 4c in this file. Cross-check the design with Fable before Task 1 of 4c. + +### Hot zone (4b) + +``` +scripts/arch_metrics.py +scripts/arch_report.py +scripts/preflight.sh +.github/workflows/ci.yml +tests/test_arch_metrics.py +docs/stabilization/ +``` ## Stage 4c: seam map, then the context diet From 136605238214d11ca053a0a0836d8f635574fb7f Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 13:20:08 -0500 Subject: [PATCH 16/41] feat(arch): module-size warn band; rename modules_over_1000_lines to modules_over_size_ceiling Key rename in baseline.json (value 0 -> 0): modules_over_1000_lines is now modules_over_size_ceiling, ceiling 1,100. Modules in 1,001-1,100 print a non-failing warning line; a failing --check prints the re-baseline rule. Co-Authored-By: Claude Opus 5.5 --- docs/stabilization/ROADMAP.md | 1 + docs/stabilization/baseline.json | 2 +- scripts/arch_metrics.py | 38 ++++++++++++- scripts/arch_report.py | 2 +- tests/test_arch_metrics.py | 98 +++++++++++++++++++++++++++++++- 5 files changed, 134 insertions(+), 7 deletions(-) diff --git a/docs/stabilization/ROADMAP.md b/docs/stabilization/ROADMAP.md index 4c7acdd3..dea17876 100644 --- a/docs/stabilization/ROADMAP.md +++ b/docs/stabilization/ROADMAP.md @@ -22,6 +22,7 @@ Two kinds, kept small on purpose. **Ratchets.** These live in `scripts/arch_metrics.py`. Each is lower-is-better, and `--check` against `baseline.json` fails the build when one gets worse. +- Module size is a band (Phase 4b): the ratchet `modules_over_size_ceiling` counts modules over 1,100 lines, and a module between 1,001 and 1,100 prints a non-failing `warning:` line on every run (never in `baseline.json`). - Phase 0 set: modules, modules over 1,000 lines, functions over 150 lines, reference-prefix literals, test `patch("dw...")` targets, CLAUDE.md lines, duplicate blocks. - Added at the start of Phase 1 ("metrics v2"), re-baselined in the same commit: - **Cyclomatic complexity:** functions above 15, by ruff's C901 (already a dev dependency). Baseline 21. diff --git a/docs/stabilization/baseline.json b/docs/stabilization/baseline.json index 2e014edf..55241310 100644 --- a/docs/stabilization/baseline.json +++ b/docs/stabilization/baseline.json @@ -1,6 +1,6 @@ { "modules": 165, - "modules_over_1000_lines": 0, + "modules_over_size_ceiling": 0, "functions_over_150_lines": 0, "prefix_literals": 0, "prefix_handling": 0, diff --git a/scripts/arch_metrics.py b/scripts/arch_metrics.py index 16c95401..9b15ddb2 100644 --- a/scripts/arch_metrics.py +++ b/scripts/arch_metrics.py @@ -7,6 +7,10 @@ Counting rules, fixed so any commit measures the same way: - Sources are dw/ and dw_mcp/, minus EXCLUDED (vendored community pipelines). +- Module size is a band: modules_over_size_ceiling counts modules above + SIZE_CEILING (1,100 lines, ratcheted); a module above SIZE_WARNING (1,000) + and at or under the ceiling is printed as a `warning:` line on every run, + which is not a metric and never enters the baseline. - Cyclomatic complexity is ruff's C901 (mccabe) as ruff reports it: each function scored on its own body, nested functions counted into it too. - An import cycle is a strongly connected component of more than one module @@ -68,6 +72,13 @@ ("startswith", "removeprefix", "removesuffix", "replace", "split", "partition") ) PATCH_TARGET = re.compile(r"""patch\(\s*["']dw[._]""") +# Module size is a band: warning above SIZE_WARNING, failing above SIZE_CEILING +SIZE_WARNING = 1000 +SIZE_CEILING = 1100 +RERUN_RULE = ( + "Re-baseline rule: lower baseline.json freely when a ratchet improves;", + "raise it only in a commit whose message names the rise and why.", +) COMPLEXITY_LIMIT = 15 COMPLEXITY_MESSAGE = re.compile(r"^`(?P.+)` is too complex \((?P\d+) > 0\)$") @@ -405,7 +416,7 @@ def measure(root): engine = list(_sources(root, *PACKAGES)) metrics = { "modules": len(engine), - "modules_over_1000_lines": 0, + "modules_over_size_ceiling": 0, "functions_over_150_lines": 0, "prefix_literals": 0, "prefix_handling": 0, @@ -413,8 +424,8 @@ def measure(root): constants, prefixes = _reference_names(root.resolve()) for path in engine: text = path.read_text(encoding="utf-8") - if len(text.splitlines()) > 1000: - metrics["modules_over_1000_lines"] += 1 + if len(text.splitlines()) > SIZE_CEILING: + metrics["modules_over_size_ceiling"] += 1 tree = ast.parse(text) if path.relative_to(root).as_posix() not in PREFIX_OWNERS: metrics["prefix_handling"] += len( @@ -449,6 +460,20 @@ def measure(root): return metrics +def size_warnings(root): + """[(path relative to root, lines)] for each measured module in the warn + band: above SIZE_WARNING and at most SIZE_CEILING. Counted the way + measure() counts, so a module is in the band or over the ceiling, never + both.""" + root = pathlib.Path(root) + found = [] + for path in _sources(root, *PACKAGES): + lines = len(path.read_text(encoding="utf-8").splitlines()) + if SIZE_WARNING < lines <= SIZE_CEILING: + found.append((path.relative_to(root).as_posix(), lines)) + return found + + def regressions(current, baseline): worse = [] for name, before in baseline.items(): @@ -473,12 +498,19 @@ def main(argv=None): args = parser.parse_args(argv) current = measure(args.root) print(json.dumps(current, indent=2)) + for path, lines in size_warnings(args.root): + print( + f"warning: {path} is {lines:,} lines " + f"(warn above {SIZE_WARNING:,}, fail above {SIZE_CEILING:,})" + ) if args.write: pathlib.Path(args.write).write_text(json.dumps(current, indent=2) + "\n") if args.check: worse = regressions(current, json.loads(pathlib.Path(args.check).read_text())) for line in worse: print(line) + if worse: + print("\n".join(RERUN_RULE)) return 1 if worse else 0 return 0 diff --git a/scripts/arch_report.py b/scripts/arch_report.py index acddd42d..a4c25781 100644 --- a/scripts/arch_report.py +++ b/scripts/arch_report.py @@ -29,7 +29,7 @@ ARCHIVED = ["dw", "dw_mcp", "tests", "ui/src", ":(glob)**/CLAUDE.md"] RATCHETS = [ ("modules", "Engine + MCP modules"), - ("modules_over_1000_lines", "Modules over 1,000 lines"), + ("modules_over_size_ceiling", "Modules over 1,100 lines (ceiling)"), ("functions_over_150_lines", "Functions over 150 lines"), ("complex_functions", "Functions over cyclomatic complexity 15"), ("import_cycles", "Import cycles"), diff --git a/tests/test_arch_metrics.py b/tests/test_arch_metrics.py index 1f08763b..2c9d3c87 100644 --- a/tests/test_arch_metrics.py +++ b/tests/test_arch_metrics.py @@ -35,11 +35,11 @@ def test_a_long_function_and_a_long_module_are_counted(tmp_path): metrics = _load().measure( _tree( tmp_path, - {"dw/a.py": _long_function(151) + "\n" * 900, "dw/b.py": "x = 1\n"}, + {"dw/a.py": _long_function(151) + "\n" * 1000, "dw/b.py": "x = 1\n"}, ) ) assert metrics["functions_over_150_lines"] == 1 - assert metrics["modules_over_1000_lines"] == 1 + assert metrics["modules_over_size_ceiling"] == 1 assert metrics["modules"] == 2 @@ -95,6 +95,100 @@ def test_check_mode_exits_nonzero_on_a_regression(tmp_path): assert "modules: 0 ->" in result.stdout +def _module_of(lines): + # distinct lines: the duplicate-block scan is slow on a repeated one + return "".join(f"x{i} = {i}\n" for i in range(lines)) + + +def _run_script(root, *args): + return subprocess.run( + [sys.executable, str(SCRIPT), "--root", str(root), *args], + capture_output=True, + text=True, + ) + + +def test_a_module_in_the_warn_band_warns_and_passes(tmp_path): + root = _tree(tmp_path / "repo", {"dw/a.py": _module_of(1001)}) + baseline = tmp_path / "baseline.json" + baseline.write_text(json.dumps({"modules_over_size_ceiling": 0})) + result = _run_script(root, "--check", str(baseline)) + assert result.returncode == 0 + assert ( + "warning: dw/a.py is 1,001 lines (warn above 1,000, fail above 1,100)" + in result.stdout + ) + # the warning follows the JSON, which stays parseable up to it + assert result.stdout.index("{") < result.stdout.index("warning:") + + +def test_size_warnings_cover_the_band_and_nothing_else(tmp_path): + root = _tree( + tmp_path, + { + "dw/at_1000.py": _module_of(1000), + "dw/at_1100.py": _module_of(1100), + "dw/at_1101.py": _module_of(1101), + "dw/x/small.py": "x = 1\n", + }, + ) + assert _load().size_warnings(root) == [("dw/at_1100.py", 1100)] + assert _load().measure(root)["modules_over_size_ceiling"] == 1 + + +def test_a_module_at_1000_lines_does_not_warn(tmp_path): + root = _tree(tmp_path / "repo", {"dw/a.py": _module_of(1000)}) + assert "warning:" not in _run_script(root).stdout + + +def test_a_module_over_the_ceiling_fails_the_check(tmp_path): + root = _tree(tmp_path / "repo", {"dw/a.py": _module_of(1101)}) + baseline = tmp_path / "baseline.json" + baseline.write_text(json.dumps({"modules_over_size_ceiling": 0})) + result = _run_script(root, "--check", str(baseline)) + assert result.returncode == 1 + assert "modules_over_size_ceiling: 0 -> 1" in result.stdout + assert "warning:" not in result.stdout + + +_RULE = "raise it only in a commit whose message names the rise and why" + + +def test_a_failing_check_prints_the_rebaseline_rule_and_a_passing_one_does_not( + tmp_path, +): + root = _tree(tmp_path / "repo", {"dw/a.py": "x = 1\n"}) + baseline = tmp_path / "baseline.json" + baseline.write_text(json.dumps({"modules": 0})) + failing = _run_script(root, "--check", str(baseline)) + assert failing.returncode == 1 + assert "lower baseline.json freely when a ratchet improves" in failing.stdout + assert _RULE in failing.stdout + assert "arch-approved" not in failing.stdout + baseline.write_text(json.dumps({"modules": 5})) + passing = _run_script(root, "--check", str(baseline)) + assert passing.returncode == 0 + assert _RULE not in passing.stdout + + +def test_the_check_fails_closed_when_the_import_graph_cannot_be_built(tmp_path): + # the graph subprocess runs in the tree's root, so this shadows grimp + root = _tree( + tmp_path / "repo", + { + "dw/__init__.py": "", + "dw/a.py": "x = 1\n", + "grimp.py": "raise ImportError('grimp is gone')\n", + }, + ) + baseline = tmp_path / "baseline.json" + baseline.write_text(json.dumps({"modules": 99})) + result = _run_script(root, "--check", str(baseline)) + assert result.returncode != 0 + assert "CalledProcessError" in result.stderr + assert '"modules"' not in result.stdout # no metrics were reported as a pass + + def _branches(count): body = "".join(f" if x == {i}:\n return {i}\n" for i in range(count)) return f"def branchy(x):\n{body} return -1\n" From e546e4ea3165ed7cbe44bd54164801aa5f81f391 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 13:24:48 -0500 Subject: [PATCH 17/41] ci: run the architecture ratchet; scope ruff to the Python packages and scripts Co-Authored-By: Claude Opus 5.5 --- .github/workflows/ci.yml | 8 ++++++-- scripts/preflight.sh | 12 ++++++++---- 2 files changed, 14 insertions(+), 6 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 794ed17a..a0ca65f2 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -45,11 +45,15 @@ jobs: pip install -r requirements.txt -r requirements-test.txt pip install git+https://github.com/huggingface/diffusers - name: Format check - run: ruff format --check dw dw_mcp tests + run: ruff format --check dw dw_mcp tests scripts - name: Lint - run: ruff check dw dw_mcp tests + run: ruff check dw dw_mcp tests scripts - name: Tests run: pytest -q + # Guards the architecture metrics against drifting past the committed + # baseline (modules, size ceiling, cycles); see docs/stabilization/ + - name: Architecture ratchet + run: python scripts/arch_metrics.py --check docs/stabilization/baseline.json ui: runs-on: ubuntu-latest diff --git a/scripts/preflight.sh b/scripts/preflight.sh index 500dbe7d..37ab9720 100755 --- a/scripts/preflight.sh +++ b/scripts/preflight.sh @@ -1,6 +1,9 @@ #!/usr/bin/env bash -# Local pre-merge checks: ruff (format + fix), pytest (unit, then the -# real-model integration tests, which skip without an accelerator), and the UI's own +# Local pre-merge checks: ruff (format + fix, scoped to the Python packages so +# docs/*.md code blocks are left alone), pytest (unit, then the +# real-model integration tests, which skip without an accelerator), the +# architecture ratchet (scripts/arch_metrics.py --check against +# docs/stabilization/baseline.json), and the UI's own # preflight (check, lint, format, build, unit and e2e tests). Runs every step # and lists the ones that failed. Run it from anywhere: # @@ -22,10 +25,11 @@ run_step() { fi } -run_step "ruff format" ruff format . -run_step "ruff check" ruff check . --fix +run_step "ruff format" ruff format dw dw_mcp tests scripts +run_step "ruff check" ruff check dw dw_mcp tests scripts --fix run_step "pytest" python -m pytest run_step "pytest integration" python -m pytest -m integration -n0 +run_step "architecture ratchet" python scripts/arch_metrics.py --check docs/stabilization/baseline.json # e2e starts the fixture server on the same interpreter pytest just used, # wherever its venv lives (a worktree, .venv) - see ui/playwright.config.ts export DW_E2E_PYTHON="${DW_E2E_PYTHON:-$(command -v python)}" From fe7acd1fe6de2f51f38b67b85a42a3e7db964e51 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 13:29:46 -0500 Subject: [PATCH 18/41] probe: empty module to trip the ratchet (reverted next) Co-Authored-By: Claude Opus 5.5 --- dw/_ratchet_probe.py | 0 1 file changed, 0 insertions(+), 0 deletions(-) create mode 100644 dw/_ratchet_probe.py diff --git a/dw/_ratchet_probe.py b/dw/_ratchet_probe.py new file mode 100644 index 00000000..e69de29b From 0f928e3de7485d6a8009a34587f7426c7f186581 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 13:34:34 -0500 Subject: [PATCH 19/41] Revert "probe: empty module to trip the ratchet (reverted next)" Co-Authored-By: Claude Opus 5.5 --- dw/_ratchet_probe.py | 0 1 file changed, 0 insertions(+), 0 deletions(-) delete mode 100644 dw/_ratchet_probe.py diff --git a/dw/_ratchet_probe.py b/dw/_ratchet_probe.py deleted file mode 100644 index e69de29b..00000000 From ccbde376770654ddbfe41e7f63a695f0252a7c08 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 13:40:03 -0500 Subject: [PATCH 20/41] chore(stabilization): stage 4b merge prep Co-Authored-By: Claude Opus 5.5 --- docs/stabilization/ROADMAP.md | 2 +- docs/stabilization/hot-zone.txt | 6 +----- 2 files changed, 2 insertions(+), 6 deletions(-) diff --git a/docs/stabilization/ROADMAP.md b/docs/stabilization/ROADMAP.md index dea17876..44cc35f8 100644 --- a/docs/stabilization/ROADMAP.md +++ b/docs/stabilization/ROADMAP.md @@ -14,7 +14,7 @@ phase works on is what the earlier phases leave behind. | 1 | Metrics v2 first (see below); remove the REPL; one prepare pipeline; one admission service; `dw.run` becomes a thin client of `dw.serve` | Validation sees the definition the run sees; the server admits a request once (one `Workflow`, one expansion); every entry point reaches the worker through the server; ratchets re-baselined | [phase-1.md](phase-1.md) | done 2026-09-28 (`stabilization-gate-1`) | | 2 | Seams in place: `references.py`, validation context + check registry, shared task rules, step cache, typed worker protocol | `validation_errors` is a registry loop; no prefix literals outside `references.py` | [phase-2.md](phase-2.md) (staged: 2a-2d) | done 2026-09-30 (`stabilization-gate-2`) | | 3 | Structural moves: `app.py` routers + services, `LibraryPath`, split `result.py` / `pipeline.py`, one media + dsp module | No module over 1,000 lines, no function over 150; suite and lem smoke green | [phase-3.md](phase-3.md) (staged: 3a-3e) | done 2026-10-01 (`stabilization-gate-3`) | -| 4 | Carried fixes; guardrails installed; context diet (root CLAUDE.md <= 150 lines, every CLAUDE.md triaged; was "<= 250 total", Don 2026-10-01) | Guardrails live in dw CI and the harness; freeze lifted | [phase-4.md](phase-4.md) (staged: 4a-4d) | 4a merged 2026-10-01; 4b in progress | +| 4 | Carried fixes; guardrails installed; context diet (root CLAUDE.md <= 150 lines, every CLAUDE.md triaged; was "<= 250 total", Don 2026-10-01) | Guardrails live in dw CI and the harness; freeze lifted | [phase-4.md](phase-4.md) (staged: 4a-4d) | 4a, 4b merged 2026-10-01 | ## Metrics diff --git a/docs/stabilization/hot-zone.txt b/docs/stabilization/hot-zone.txt index 5ec181d4..a4af1659 100644 --- a/docs/stabilization/hot-zone.txt +++ b/docs/stabilization/hot-zone.txt @@ -2,10 +2,6 @@ # The harness implementer must not change these; a field bug whose fix # needs one is labelled `stabilization` and handed to owner:don. # One path (or directory ending in /) per line; '#' starts a comment. -# Stage 4b (guardrails in dw) live from 2026-10-01; plan: docs/stabilization/phase-4.md +# Stage 4b merged (2026-10-01); stage 4c's list goes live when it is detailed. scripts/arch_metrics.py -scripts/arch_report.py -scripts/preflight.sh -.github/workflows/ci.yml -tests/test_arch_metrics.py docs/stabilization/ From 2270d730947c2b4286bb82fbabdbadee68efd240 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 13:44:11 -0500 Subject: [PATCH 21/41] docs(stabilization): stage 4c detail and hot zone Co-Authored-By: Claude Opus 5.5 --- docs/stabilization/ROADMAP.md | 2 +- docs/stabilization/hot-zone.txt | 12 +++- docs/stabilization/phase-4.md | 120 +++++++++++++++++++++++++++++++- 3 files changed, 131 insertions(+), 3 deletions(-) diff --git a/docs/stabilization/ROADMAP.md b/docs/stabilization/ROADMAP.md index 44cc35f8..a4c604f5 100644 --- a/docs/stabilization/ROADMAP.md +++ b/docs/stabilization/ROADMAP.md @@ -14,7 +14,7 @@ phase works on is what the earlier phases leave behind. | 1 | Metrics v2 first (see below); remove the REPL; one prepare pipeline; one admission service; `dw.run` becomes a thin client of `dw.serve` | Validation sees the definition the run sees; the server admits a request once (one `Workflow`, one expansion); every entry point reaches the worker through the server; ratchets re-baselined | [phase-1.md](phase-1.md) | done 2026-09-28 (`stabilization-gate-1`) | | 2 | Seams in place: `references.py`, validation context + check registry, shared task rules, step cache, typed worker protocol | `validation_errors` is a registry loop; no prefix literals outside `references.py` | [phase-2.md](phase-2.md) (staged: 2a-2d) | done 2026-09-30 (`stabilization-gate-2`) | | 3 | Structural moves: `app.py` routers + services, `LibraryPath`, split `result.py` / `pipeline.py`, one media + dsp module | No module over 1,000 lines, no function over 150; suite and lem smoke green | [phase-3.md](phase-3.md) (staged: 3a-3e) | done 2026-10-01 (`stabilization-gate-3`) | -| 4 | Carried fixes; guardrails installed; context diet (root CLAUDE.md <= 150 lines, every CLAUDE.md triaged; was "<= 250 total", Don 2026-10-01) | Guardrails live in dw CI and the harness; freeze lifted | [phase-4.md](phase-4.md) (staged: 4a-4d) | 4a, 4b merged 2026-10-01 | +| 4 | Carried fixes; guardrails installed; context diet (root CLAUDE.md <= 150 lines, every CLAUDE.md triaged; was "<= 250 total", Don 2026-10-01) | Guardrails live in dw CI and the harness; freeze lifted | [phase-4.md](phase-4.md) (staged: 4a-4d) | 4a, 4b merged 2026-10-01; 4c in progress | ## Metrics diff --git a/docs/stabilization/hot-zone.txt b/docs/stabilization/hot-zone.txt index a4af1659..3d04f63e 100644 --- a/docs/stabilization/hot-zone.txt +++ b/docs/stabilization/hot-zone.txt @@ -2,6 +2,16 @@ # The harness implementer must not change these; a field bug whose fix # needs one is labelled `stabilization` and handed to owner:don. # One path (or directory ending in /) per line; '#' starts a comment. -# Stage 4b merged (2026-10-01); stage 4c's list goes live when it is detailed. +# Stage 4c (seam map + context diet) live from 2026-10-01; plan: docs/stabilization/phase-4.md +CLAUDE.md +ui/CLAUDE.md +dw_mcp/CLAUDE.md +dw/server/CLAUDE.md +.github/copilot-instructions.md +docs/ARCHITECTURE.md +docs/WORKFLOW_GUIDE.md +docs/AGENT_LOOP.md +.claude/skills/model-family-onboarding/references/cold-drill-example.md +tests/test_architecture_map.py scripts/arch_metrics.py docs/stabilization/ diff --git a/docs/stabilization/phase-4.md b/docs/stabilization/phase-4.md index 99887a37..73e5e110 100644 --- a/docs/stabilization/phase-4.md +++ b/docs/stabilization/phase-4.md @@ -369,7 +369,125 @@ docs/stabilization/ ## Stage 4c: seam map, then the context diet -Detailed when 4b merges. +Work on branch `stabilization/phase-4c` in the worktree, from `develop` at `edb6e443` (the 4b merge) or later. + +**What exists (at `edb6e443`).** +- Four CLAUDE.md files, 957 lines (`claude_md_lines`, ratcheted): + - root, 674 lines and 51 KB, loaded by every session. "Critical Gotchas" is 395 of those lines, and "Type System" 69. + - `ui/CLAUDE.md` 154 (design system, assets); + - `dw_mcp/CLAUDE.md` 95; + - `dw/server/CLAUDE.md` 34. +- `.github/copilot-instructions.md`, 114 lines. It is a parallel description of the architecture, partly stale (it still describes a 5-minute execution timeout and memory warnings), and the metric does not count it. +- What reads a CLAUDE.md mechanically: + - `tests/test_plugin_skills.py::test_a_skill_is_enumerated_where_the_plugin_describes_itself`, which needs every plugin skill's name in backticks in the root file; + - `arch_metrics.py` and `arch_report.py`, for counting. +- Docs that point into CLAUDE.md: `docs/AGENT_LOOP.md`, `docs/WORKFLOW_GUIDE.md` (its authoring section and the root's Type System say "change both when one changes") and `.claude/skills/model-family-onboarding/references/cold-drill-example.md`. +- No architecture map exists. An agent learns which module owns a concept from CLAUDE.md prose or by grepping. + +### Decisions (4c) + +- **The triage comes first, and it is a committed table:** `phase-4-surveys/claude-md-triage.md`. It has one row per paragraph or bullet of each CLAUDE.md and of `copilot-instructions.md`, with four columns: + - where the paragraph is; + - its first words; + - the verdict: **delete**, **docstring**, **map** or **keep**; + - the evidence. + - Evidence by verdict: + - delete: the file:line that already says it (a doc, a docstring, or a test whose name or docstring states the rule); + - docstring: the module it moves to; + - map: the seam-map row it becomes; + - keep: why an agent needs it before it knows which module to open. + + A paragraph the triage can't place stays as **keep**, with that said. The diet then argues from the table, and the review checks the table, not 700 lines of prose. +- **What stays in the root CLAUDE.md (<= 150 lines):** + - the project in two sentences, and the common commands; + - a "where things are" block that points at the seam map; + - the security rules, which an agent must know before it opens any file; + - the plugin-skill names (pinned by the test); + - the few gotchas that bite at the moment of editing and that no test or check catches. Each is one or two lines, and each ends with a pointer to its owner. + - Everything that describes how a subsystem works goes to its module's docstring, or to a seam-map row. +- **The seam map is `docs/ARCHITECTURE.md`:** a table of concept, owning module(s), the rule in one sentence, and the test or check that enforces it, if any. It covers the concepts the triage sends there, plus every owner the stabilization created: `references`, `library`, the validation registry, `workflow_run`, `step_cache`, `worker_protocol`, `media`/`dsp`, `trust`/`security`, `plan`/`observed_cost`, `runs` versions, the server routers and the MCP tools. + - It is not loaded into every session, so its length is not context cost. It must still stay a map, a sentence per row; a row that needs a paragraph links to the docstring that holds it. + - A test (`tests/test_architecture_map.py`) checks that every backticked `dw/...` / `dw_mcp/...` / `ui/...` path in it exists. A map that names a deleted module is worse than none, and this is the drift that would happen silently. It fails first, against a fixture map naming a missing file. +- **A docstring move never changes what agents are served.** Destinations are module docstrings and internal-function docstrings only. Never an MCP tool's docstring, a `register_command` task description, or the schema: those are the served surface, and `scripts/surface_snapshot.py` pins them byte for byte. The snapshot does not cover the served guides, so anything touching `docs/WORKFLOW_GUIDE.md` or another `docs/*.md` is a pointer added, never content moved and never a heading changed. Tasks 3-5 each end with the snapshot byte-identical. +- **Delete needs evidence an agent reads before editing:** a doc or a docstring. A rule stated only by a test, in its name or docstring, becomes a **map** row with that test in the "enforced by" column. Nobody opens `tests/` first, so the map has to send them there. +- **Docstring moves keep the code's line budget in view.** A module receiving a moved rule must stay under the size warning (1,000 lines). If one would cross it, the rule goes to the map row instead, and the triage says so. +- **`copilot-instructions.md` becomes a pointer of 10 lines or fewer:** read `CLAUDE.md` and `docs/ARCHITECTURE.md`. Its stale facts go, and it is not counted by the metric either way. +- **"Change both when one changes" pairs are resolved, not kept.** Where the root CLAUDE.md and `docs/WORKFLOW_GUIDE.md` both describe the reference conventions, the guide is the owner (agents read it over MCP); CLAUDE.md points at it. +- **Sub-files get the same triage, with no quota** (frame Decisions). `ui/CLAUDE.md`'s design-system rules are read only by UI work and stay where that work reads them, unless the triage finds them already said in code or a test. + +### Review Focus (4c) + +1. **A "delete" whose evidence does not say it.** The reviewer samples at least 15 delete rows, weighted to Critical Gotchas, and reads each cited file:line. A row whose evidence is weaker than the paragraph (it names the module but not the rule) is a finding, and the paragraph is re-triaged. +2. **A rule that vanished.** Every paragraph of the old files appears in the triage table (counted against `git show :CLAUDE.md`), and every docstring and map row the table promises exists after Task 3. +3. **The map names things that exist,** by the map test, and its owners are the real ones: the reviewer spot-checks 10 rows against the code. +4. **The mechanical readers still pass:** `test_plugin_skills` without loosening, `arch_metrics --check` after re-baselining `claude_md_lines`, and the docs that point into CLAUDE.md (AGENT_LOOP, WORKFLOW_GUIDE, the onboarding skill's drill) still point at text that exists. +5. **A cold agent can still find its way.** After Task 4, one fresh subagent with no session context gets only the new root CLAUDE.md and is asked three questions a harness implementer meets: + - where `asset:` references are resolved, and what confines them; + - what to change to add a validation check; + - why a seeded rerun generates nothing. + + It records the files it opened, in order. It passes only when the chain is CLAUDE.md, then `docs/ARCHITECTURE.md` (or a named docstring), then the module, with no grep before the map. A right answer reached by grep is still a finding against the pointers. + +### Before Task 1: the hot zone goes live + +Commit the 4c list below into `hot-zone.txt` on `develop` with this section, push, and branch `stabilization/phase-4c`. Task 1's table names the modules that will receive docstrings; those are added to the hot zone in Task 1's commit. + +### Task 1: The triage + +- [ ] **Step 1:** Build the table in two dispatches on the most capable model: one for the root's "Critical Gotchas" (395 lines), one for the rest of the root plus the three sub-files and `copilot-instructions.md`. It is judgement work, and its mistakes are what the diet ships; its reviewer is on the same model. Cheap evidence sources: the long descriptive test names and docstrings, and the `docs/*.md` list. +- [ ] **Step 2:** For every delete row, open the cited evidence and confirm it states the rule, not just the topic. Downgrade to docstring or map where it doesn't. +- [ ] **Step 3:** Summarise the table at its top: rows per verdict per file; the projected root line count; the modules that will receive docstrings, with their current line counts against 1,000; the seam-map rows to be written. +- [ ] **Step 4:** Commit the table. Put the hot-zone additions (the docstring destinations) on `develop` too, pushed as a docs-only commit as the 4a and 4b lists were. The harness reads `hot-zone.txt` from `develop`, not from this branch. +- [ ] **Step 5:** Send Don the table's summary (counts per verdict per file, the keep rows, the projected root count) with "say if a delete should stay". Don't wait for an answer, but make it visible before Task 4 rewrites the file he works in. + +### Task 2: The seam map + +- [ ] **Step 1:** The map test, against a fixture map naming one missing path; it fails, since the test file does not exist yet. +- [ ] **Step 2:** Write `docs/ARCHITECTURE.md` from the table's map rows plus the owners listed in Decisions (4c). Each row's module path is checked by the test, and each rule sentence by reading the code. +- [ ] **Step 3:** Run the map test on the real file. Commit. + +### Task 3: The docstring moves + +- [ ] **Step 1:** For each docstring row, put the rule in the module's docstring (or the owning function's), rewritten to the module's voice, not pasted. Keep each module under 1,000 lines; `arch_metrics` prints a warning otherwise. +- [ ] **Step 2:** The suite, ruff and `arch_metrics --check` pass, with no new warnings, and the surface snapshot is byte-identical. Commit. + +### Task 4: The root CLAUDE.md + +- [ ] **Step 1:** Rewrite it to the Decisions (4c) shape from the keep rows, <= 150 lines, on the most capable model. Match the existing voice: short and specific, every claim naming a file. Don reads and edits this file himself. Point `docs/WORKFLOW_GUIDE.md`'s "change both" note, `docs/AGENT_LOOP.md` and `.claude/skills/model-family-onboarding/references/cold-drill-example.md` at whatever they referred to. +- [ ] **Step 2:** `test_plugin_skills` and the map test pass, and the surface snapshot is byte-identical. List each kept gotcha as a candidate check: ASSESSMENT's principle is mechanical over prose, so these become 4d / stage C follow-ups, not the end state. +- [ ] **Step 3:** Review Focus 5, the cold-agent check. Record its answers in the report, then fix the map or the root file for any miss. +- [ ] **Step 4:** Commit. + +### Task 5: The sub-files and copilot-instructions + +- [ ] **Step 1:** Apply the triage to `ui/CLAUDE.md`, `dw_mcp/CLAUDE.md` and `dw/server/CLAUDE.md`. Turn `copilot-instructions.md` into a pointer. +- [ ] **Step 2:** The UI preflight is unaffected (docs only); spot-check that `ui/CLAUDE.md`'s kept design rules still name real files. The surface snapshot is byte-identical. Commit. + +### Task 6: Stage 4c merge + +- [ ] **Step 1:** Re-baseline. `claude_md_lines` falls from 957 to the new total; nothing else moves. +- [ ] **Step 2:** ROADMAP Phase 4 row: record the final line counts per file. ASSESSMENT's "CLAUDE.md files total 964 lines" finding gets its gate-4 number at the gate, not here. +- [ ] **Step 3:** Hot zone back to the standing entries. Merge with `--no-ff`, push, and check CI. Report to Don with the per-file counts and the cold-agent answers. +- [ ] **Step 4:** Detail stage 4d. It includes writing the harness stage C prompt, which goes to Don before anything else in 4d. + +### Hot zone (4c) + +``` +CLAUDE.md +ui/CLAUDE.md +dw_mcp/CLAUDE.md +dw/server/CLAUDE.md +.github/copilot-instructions.md +docs/ARCHITECTURE.md +docs/WORKFLOW_GUIDE.md +docs/AGENT_LOOP.md +.claude/skills/model-family-onboarding/references/cold-drill-example.md +tests/test_architecture_map.py +scripts/arch_metrics.py +docs/stabilization/ +``` + +The harness implementer often adds a CLAUDE.md line with a field fix. While 4c runs, a fix that needs one is labelled `stabilization` and handed to Don, and its note goes into the triage instead. ## Stage 4d: gate 4 From 91b39c2fe0b25e92914715d914f61c504365167a Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 14:09:35 -0500 Subject: [PATCH 22/41] docs(stabilization): 4c CLAUDE.md triage table (221 rows, reviewed) Co-Authored-By: Claude Opus 5.5 --- .../phase-4-surveys/claude-md-triage.md | 419 ++++++++++++++++++ docs/stabilization/phase-4.md | 2 +- 2 files changed, 420 insertions(+), 1 deletion(-) create mode 100644 docs/stabilization/phase-4-surveys/claude-md-triage.md diff --git a/docs/stabilization/phase-4-surveys/claude-md-triage.md b/docs/stabilization/phase-4-surveys/claude-md-triage.md new file mode 100644 index 00000000..4588d550 --- /dev/null +++ b/docs/stabilization/phase-4-surveys/claude-md-triage.md @@ -0,0 +1,419 @@ +# CLAUDE.md triage (stage 4c, Task 1) + +Base `2270d730`. One row per paragraph or bullet of each CLAUDE.md and of `.github/copilot-instructions.md`, with a verdict and its evidence (rules: [phase-4.md](../phase-4.md), Decisions (4c)). Built in two halves on 2026-10-01: the root's Critical Gotchas section, then everything else. + +**Summary (221 rows):** + +| File | delete | docstring | map | keep | +| --- | --- | --- | --- | --- | +| root CLAUDE.md, Critical Gotchas (274-669) | 42 | 7 | 10 | 1 | +| root CLAUDE.md, the rest (1-273, 670-674) | 56 | 4 | 9 | 12 | +| dw/server/CLAUDE.md | 5 | 0 | 3 | 1 | +| dw_mcp/CLAUDE.md | 15 | 1 | 2 | 4 | +| ui/CLAUDE.md | 18 | 0 | 0 | 7 | +| .github/copilot-instructions.md | 22 | 0 | 2 | 0 | + +- **Projected kept lines:** + - root about 60, before the new "where things are" block (target <= 150); + - `dw/server` about 3, `dw_mcp` about 10, `ui` about 22; + - copilot becomes a pointer of 10 lines or fewer. +- **Docstring destinations:** `dw/subfolders.py` (amendment 1), `dw/reference_names.py`, `dw/pipeline_processors/placement.py`, `dw/previous_results.py`, `dw/runs.py` (894, about +6), `dw/step_cache.py`, `dw/variable_constraints.py`, `dw/shots.py`, `dw/__init__.py`, `dw/settings.py`, `dw/server/job_history.py`, `dw/realize.py`, `dw_mcp/__init__.py`. All stay under 1,000 lines. +- **Rulings (controller, 2026-10-01):** + - A README or SKILL.md beside the code it describes (`workflows/templates/ltx2/README.md`, `plugins/dw/**`), or a header comment in the file an agent edits (`ui/src/app.css`), counts as "a doc an agent reads before editing" for a delete. + - `dw/variables.py` gets an owner row on the seam map (Task 2). + - The ruling on comments is widened: a block comment in the file that implements the behaviour (for example `ui/src/lib/pages/AssetsPage.svelte`) also counts, because whoever edits that behaviour reads that file. This covers UI rows 17, 18, 22, 23, 24 and root row 20. + - Stale text the triage found (root 109 `builtin:` in `workflow.py`, now `library.py`; 529 and 548 on the audio QC warnings; copilot's 5-minute timeout, SHA-256 change detection, JPEG default and schema-per-pipeline note) is deleted, not carried forward. + +**Amendments from the table review (2026-10-01; they override the rows they name):** +1. Gotchas row 32 (CLAUDE.md:407): split. The `output:` seven-segment ceiling clause becomes a **docstring** in `dw/subfolders.py`'s module docstring. `SUBFOLDER_PATTERN` has no depth bound, and `OUTPUT_REFERENCE_PATTERN` allows 1-6 segments after the identity, so a deep subfolder under a nested identity cannot be named. The rest of the row stays delete. +2. Root row 67 (CLAUDE.md:243): split. The ordering rule (device translation happens before anything reads the backend, so the MPS accommodations fire for a translated device too) becomes a **docstring** on `place_component` (`dw/pipeline_processors/placement.py`). Today it is only a `#` comment. +3. Copilot row 12: the evidence is `docs/TESTING.md:6-7`, not the root's Common Commands. Root row 3 (keep) gains the unit-test command line (`pytest tests/ -v`; `-n auto` with xdist). +4. Root rows 12, 26 and 81 become **keep**, as lines in the new "where things are" block: + - `dw.serve` runs every job in a persistent spawned worker that keeps models cached, so engine code changes need a server restart; + - the packaged `dw/workflows/` (what `builtin:` names) is not the top-level `workflows/` examples folder. +5. The stale note on 529 is corrected: a muxed video holds the pre-encode headroom prediction and falls back to it only when the post-write probe cannot measure the file (`audio_qc.py:58-62`). Dropping CLAUDE.md's "and a muxed video" is still right. +6. Gotchas row 57's "enforced by" names a cache-hit / `reused: true` test (Task 2 picks the exact one), not `test_full_cleanup_clears_step_cache`. MCP row 16's evidence is the handler docstring (`dw_mcp/diagnose.py:89-98`), not the `#` comment. +7. Copilot's memory-growth warning (>500 MB) is not stale (`dw/worker.py:86`). The table was right; the plan's "What exists" label was wrong. + +## Root CLAUDE.md - Critical Gotchas (lines 274-669) + +Base: worktree `stabilization/phase-4b` at `2270d730` (CLAUDE.md 674 lines; section header 274, next section 670). +30 top-level bullets (276, 277, 285, 295, 303, 317, 318, 319, 320, 321, 322, 323, 324, 350, 354, 369, 407, 435, 453, 482, 498, 512, 520, 529, 548, 563, 592, 626, 641, 644), split into 60 rows where parts take different verdicts. Every row carries its source line. + +**Counts per verdict (60 rows):** delete 42 · docstring 7 · map 10 · keep 1 + +**Docstring destinations (current `wc -l`, all stay well under 1,000):** +- `dw/reference_names.py` - 122 (module docstring) +- `dw/pipeline_processors/placement.py` - 369 (`place_component` docstring; module has none) +- `dw/previous_results.py` - 396 (`_not_found` docstring) +- `dw/runs.py` - 894 (`run_versions` docstring, ~6 lines -> ~900) +- `dw/step_cache.py` - 609 (module docstring) +- `dw/variable_constraints.py` - 468 (module docstring) +- `dw/shots.py` - 411 (module docstring) + +Not used as destinations on purpose: `dw/workflow_run.py` (971), `dw/arguments.py` (957), `dw/plan.py` (922): close to the band. Their rules here are already in their docstrings, so those rows are deletes. + +**Seam-map rows this section creates (10):** list variables from the CLI; failed-run reply manifest; job-to-run link; run-version surfaces; template subfolder roles; observed-cost residue; template audio-level placement; H3 Ref2VA VRAM numbers; inherited-VRAM index; step cache. + +**Keep (1):** row 13 (`{}` escaping). It duplicates root CLAUDE.md:133 (Type System), so keep it once, in one line. + +**Stale text found (do not carry it forward):** +- 529 says the headroom warning fires "for both a saved audio file and a muxed video". The code (`warn_without_headroom(emit=False)` for a mux, audio_qc.py:48-62) and TASKS.md:746 say a mux holds that prediction and reports only `audio_clipped`. +- 548 says the post-write probe is "silent when `warn_without_headroom` already spoke". `warn_if_written_above_full_scale` (audio_qc.py:124) narrows that to a plain audio save into a lossy container, citing #174 and #295. + +| # | CLAUDE.md:line | first words | verdict | evidence | +|---|---|---|---|---| +| 1 | 276 | Schema validation runs before variable substitution | delete | docs/WORKFLOW_GUIDE.md:274-275: "Schema validation runs before substitution, so a default must already be the JSON type" | +| 2 | 277 | `previous_result:` references are checked statically too | delete | dw/previous_results.py:326 `previous_result_reference_errors`: "Every 'previous_result:' reference that names no earlier step." The same docstring says the definition is "already been substituted and expanded". Also WORKFLOW_GUIDE.md:278-281 | +| 3 | 285 | `for_each` expands before the reference check (expansion order, source-index paths) | delete | dw/for_each.py module docstring: "before the reference check (validation_errors)". dw/for_each.py:61 `expand_for_each`: "report an error at a path in the file the author wrote" | +| 4 | 285 (cont.) | An undeclared `variable:` is a validation error at its path | delete | dw/variables.py:187 `undeclared_variable_references`: "the ones `replace_variables` will refuse at run time, found before anything loads". Also WORKFLOW_GUIDE.md:560 | +| 5 | 295 | The two MiniMax cut templates take one `shots` list (entry shapes, `shot@` members) | delete | dw/for_each.py:299 `list_fields`: "What an entry of each list-driven variable has to carry" (the catalog's `lists`). WORKFLOW_GUIDE.md:536-539: members are "named `shot@wide_open`, `shot@closeup`" | +| 6 | 295 (cont.) | The CLI only takes `name=value` strings ... comma-split | map | List variables from the CLI \| `dw/run.py`, `dw/variables.py` (`get_value`) \| a `name=value` string for a list variable is comma-split, so a list of objects (`shots`) can only be passed over the API/MCP \| none. The only statement is a `#` comment at variables.py:352, which does not count as evidence | +| 7 | 303 | A reference name is checked for its shape before the queue (shape now, existence later) | delete | dw/reference_names.py module docstring: "So the shape is checked here, in `validation_errors` ... while *existence* stays where it was" | +| 8 | 303 (cont.) | ... and `@` is part of it (accepted, never leading; `_name_fault` names the character) | docstring | `dw/reference_names.py` module docstring (122). `_name_fault` (security.py:323) already says it names the character. The `@` rule exists only as `#` comments at security.py:425-427 and 466-471 | +| 9 | 317 | Cartesian product explosion | delete | docs/WORKFLOW_GUIDE.md:489-491: "runs that step once for every combination — a cartesian product. Four images and three masks is twelve iterations" | +| 10 | 318 | Component sharing requires exact key matching | delete | docs/WORKFLOW_GUIDE.md:1399: "The names must match exactly between the two steps." | +| 11 | 319 | Built-in workflows need explicit argument mapping | delete | docs/WORKFLOW_GUIDE.md:886: "with `arguments` handed down as that workflow's variables". :902: "warns about an argument the composed workflow declares no variable for" | +| 12 | 320 | MPS differences from CUDA (no bnb/flash_attn/triton/compile; sequential -> model) | delete | docs/ACCELERATION.md:289: "No flash-attn, no Triton, no bitsandbytes". :290: "`"offload": "sequential"` becomes `"model"`". WORKFLOW_GUIDE.md:1127 | +| 13 | 320 (cont.) | `exclude_from_cpu_offload` is sequential-only and does not survive the downgrade | docstring | `dw/pipeline_processors/placement.py` (369), `place_component` docstring. Today it is only in a log string at placement.py:331 | +| 14 | 321 | `{}`-escaped strings | keep | WORKFLOW_GUIDE.md:365-367: a wrong value "fails at load time, after validation has already passed". No check catches it, and it bites while hand-editing JSON. It duplicates CLAUDE.md:133, so keep one line, once | +| 15 | 322 | A stored prompt's `text` may not begin with a reference prefix | delete | docs/WORKFLOW_GUIDE.md:314-315: "That text may not itself begin with any of these prefixes; the engine rejects such a prompt". Also :1900-1901, MCP.md:296 | +| 16 | 323 | Audio+video muxing | delete | docs/WORKFLOW_GUIDE.md:1069-1071: "a video with its own audio track ... is muxed into one `video/mp4` file with PyAV" | +| 17 | 324 | A caller's `arguments` are checked before anything is queued (`argument_errors`; no variables = no arguments) | delete | dw/variables.py:414 `argument_errors`: "Exactly the check `set_variables` makes at the top of a run". The same docstring: "A workflow that declares no variables at all takes no arguments". SERVER.md:294-296 (400 from `POST /api/jobs`) | +| 18 | 324 (cont.) | A valid `POST /api/validate` answer also carries `plan` (fingerprint, counts, downloads, estimate/basis, null) | delete | docs/SERVER.md:312-360: "A valid answer also carries `plan`, what the run will execute for those arguments". `basis` values are at :346-355 and "`plan` is `null` when it could not be built" at :359 | +| 19 | 324 (cont.) | `acknowledged_cost` on `POST /api/jobs` / `rerun` takes `true` or the plan's object | delete | docs/SERVER.md:167: "answers **409** when the fingerprint differs ... `minutes` is recorded, never compared". MCP.md:356-363 | +| 20 | 324 (cont.) | `cached_steps` is the worker's answer to `probe_cache` (shares `prepare_definition`/`cache_lookup`) | delete | dw/workflow_run.py module docstring: "The step cache probe (`cache_hits`) shares `prepare_definition` and `cache_lookup` with the run". `cache_hits`:438. SERVER.md:318 | +| 21 | 324 (cont.) | The web UI reads the fields only (`describePlan`, `bound` shown) | delete | ui/src/lib/plan.ts:15-17 doc comment: "under the verdict: the work, the figure with its basis, what the step cache would serve". SERVER.md:167: "the web UI and every caller that sends nothing are `acknowledged: none`" | +| 22 | 350 | A failed run still reports what it wrote: worker carries partial manifest on error/cancelled | map | Failed-run reporting \| `dw/worker.py`, `dw/worker_protocol.py` (error/cancelled replies' `manifest`) \| a failed or cancelled run's reply still carries the manifest of the steps that ran \| tests/test_worker_execute.py::test_failure_carries_the_manifest_of_the_steps_that_ran, ::test_cancellation_carries_the_manifest_too. The test is the only evidence (WORKSPACES.md:210 covers the on-disk manifest, not the reply) | +| 23 | 350 (cont.) | "Previous result not found" names the steps that ran after `release_unreferenced_results` | docstring | `dw/previous_results.py` (396), `_not_found` docstring (:300). The docstring today says only "Names what is there as well as what was asked for", and the released-steps clause is in the code alone | +| 24 | 354 | Run directories: layout, identity, run id + `-N`, sub-workflow shares the parent's, flat layout | delete | docs/WORKSPACES.md:197-213: "The folder is the workflow's identity"; "A sub-workflow is part of its parent's run". :233-235 flat. dw/runs.py module docstring and `workflow_identity`:296, `new_run_id`:330 | +| 25 | 354 (cont.) | The gallery groups a workflow's runs by stripping the run id | delete | dw/runs.py:381 `strip_run_id`: "The workflow identity a run-relative path belongs to". SERVER.md:103: "The folder filter groups a workflow's runs together" | +| 26 | 354 (cont.) | The realized workflow is written as `workflow.json`; manifest `workflow` block (`realized`, `prompts`, `sub_workflows`) | delete | dw/realize.py module docstring: "for a local sub-workflow, digested into the manifest instead". WORKSPACES.md:204-207: "`manifest.json` points at it and lists which prompts were inlined" | +| 27 | 354 (cont.) | A job records the run it was (`run_id`/`run_dir`); `JobManager.realized` | map | Job-to-run link \| `dw/server/` JobManager, `jobs.sqlite` \| a job records `run_id`/`run_dir`/`run_version`, and `JobManager.realized` reads that run's `workflow.json` \| none. SERVER.md:170 describes only the route's answer | +| 28 | 354 (cont.) | `exports` is a reserved workspace name; export route and zip | delete | docs/SERVER.md:171: "Gather one finished job into `/exports//`". :172 zip. MCP.md:308 | +| 29 | 369 | A run has a number: assigned at open, never recomputed, unrecorded runs ranked, pinned on the two write paths | delete | dw/runs.py:519 `run_versions`: "A run that recorded a version keeps it verbatim". :537 `record_run_versions`: "a new run opening, or a run directory being deleted. The listing never writes". :597 `open_run`: stub manifest before the lock is released | +| 30 | 369 (cont.) | `max(recorded)+1` over every sibling; deleting the newest frees its number; flat layout has `version` null | docstring | `dw/runs.py` (894), `run_versions` docstring, ~6 lines. Neither of the two deliberate limits is stated in any docstring today | +| 31 | 369 (cont.) | Surfaces: gallery `version`/`run_id`, `?folder=&version=`, `output:.../v4/`, `run_start`, `run_version` column, export name, MCP, UI chip | map | Run-version surfaces \| `dw/runs.py` (owner), the server gallery/jobs routes, `dw_mcp` `list_gallery`, `ui` (reads only) \| the ordinal is a name everywhere it appears and is computed only in `runs.py` \| none. Partial docs: WORKFLOW_GUIDE.md:309-310 (`v`), MCP.md:229 | +| 32 | 407 | Result subfolders: convention, no default, computed once, `SUBFOLDER_PATTERN`, checked statically and at run time, containment | delete | dw/subfolders.py module docstring: "by convention 'final' or 'intermediate', though the engine treats no name specially". dw/workflow.py:292 `step_output_dir`: "Computed here, once ... both have to land in the same place". WORKFLOW_GUIDE.md:844-861 | +| 33 | 407 (cont.) | Manifest/`step_end`/gallery carry `subfolder`; `split_run_path`; `?subfolder=` filter; UI sections | delete | docs/SERVER.md:169, :193 (`step_end` "`files` and `subfolder` at the end"). dw/runs.py:355 `split_run_path`: "The run id is found wherever it sits". ui/src/lib/results.ts:65 `sectionBySubfolder` doc comment. MCP.md:229 | +| 34 | 407 (cont.) | `file_base_name` may not contain a separator; replaces the base; collision gets `-2` | delete | docs/WORKFLOW_GUIDE.md:862-866: "`file_base_name` is a name, not a path ... the second collides and picks up a `-2`" | +| 35 | 407 (cont.) | Every `workflows/templates/**` file marks each saving step `final`/`intermediate`; builtins stay unmarked | map | Template subfolder roles \| `workflows/templates/**`, `dw/workflows/` \| every template saving step names its role, at least one `final`, and builtins are left unmarked \| tests/test_template_subfolders.py::test_every_saving_step_of_a_template_names_its_role, ::test_the_packaged_builtins_stay_unmarked. WORKFLOW_GUIDE.md:852 states a weaker rule ("two or more saving steps") | +| 36 | 407 (cont.) | The step-cache key includes `result`, so changing `subfolder` misses the cache | docstring | `dw/step_cache.py` (609), module docstring, one sentence. It is implied by "step_data" in condition 1 but never said | +| 37 | 435 | A task argument's numeric domain is declared, not inferred | delete | dw/task_domains.py module docstring: "The domains are declared here, once, and checked in two places". TASKS.md:28-36. audio_utils.py:112 `as_track`: "A rate that is not a rate stops here." | +| 38 | 453 | `cost` is curated, `observed` is derived: four rules, one query, watermark cache, compact vs full listing, plan quotes cold only | delete | dw/server/observed_cost.py module docstring: "Four rules, each of which is a way the naive median would lie". SERVER.md:383-401: "cached against the jobs table's high-water mark rather than a file mtime". dw/plan.py:533 `estimate`: "quoted only from the *cold* median ... only for the bucket" | +| 39 | 453 (cont.) | Raw `GET /api/workflows/{name}` left verbatim; a `cost_drivers` entry naming no variable is dropped | map | Observed-cost residue \| `dw/server/observed_cost.py`, workflow routes \| the raw workflow GET is served verbatim because the editor saves what it reads, and a driver naming no declared variable is dropped \| tests/test_observed_cost.py::test_a_driver_naming_no_variable_is_dropped, ::test_every_declared_driver_is_a_variable_of_its_workflow | +| 40 | 482 | An observed figure below three runs ... says so (`_tempered`) | delete | dw/plan.py:258 `_tempered`: "one run counts for a third of the blend". Also "a `low_confidence` flag is added instead" | +| 41 | 498 | An H3 adapter is checked against the partition its step denoises on | delete | dw/adapter_compatibility.py module docstring: "an unrecognised name is a warning naming the rule rather than a refusal". It also says `tests/test_h3_adapters.py` pins the names. SERVER.md:303-310 | +| 42 | 512 | An elided step says whether anyone decided it | delete | dw/elision.py module docstring, 4th guardrail: "A step nothing ever read still gets the diagnosis." :172 `overriding_variables`. SERVER.md:198 (`step_elided`: "`overridden_by` when a supplied argument") | +| 43 | 520 | A null `model_name` switches a lora off | delete | docs/WORKFLOW_GUIDE.md:1358-1361: "`null` switches the entry off ... Validation and the run both warn (`lora_disabled`)". :1365-1366 defaults for `adapter_name`/`scale`. adapters.py:92 `active_loras` | +| 44 | 529 | A deliverable with no audio headroom warns (−0.5 dBFS; warning, not gain change) | delete | dw/audio_qc.py:48 `warn_without_headroom`: "A warning rather than a change to the mix". TASKS.md:738-740. Note: CLAUDE.md's "and a muxed video" is stale (see header) | +| 45 | 529 (cont.) | `music-video` normalizes only the mux track; `music`'s deliverable is `balanced` | map | Template audio-level placement \| `workflows/templates/minimax/music-video.json`, `music.json` \| the deliverable's normalize sits on the muxed track only, so the slices that condition shots are untouched \| none. No doc states it, and it lives in the template JSON | +| 46 | 529 (cont.) | `normalize_audio(limit=true)`: true-peak limiter, gain search to 0.1 LU, 12 dB cap, warnings; `compress_audio` limit is sample-peak | delete | docs/TASKS.md:719: "a true-peak look-ahead limiter ... until the limited track lands within 0.1 LU". :1077: "`mode: "limit"` is a sample-peak limiter with no look-ahead" | +| 47 | 548 | A deliverable is measured as written (`audio_clipped`, material-dependent overshoot) | delete | dw/audio_qc.py:124 `warn_if_written_above_full_scale`: "no amount of headroom chosen up front can be known to be enough". TASKS.md:741-745. Note: CLAUDE.md's suppression clause is stale (see header) | +| 48 | 548 (cont.) | Templates ending in a `pair_audio` mux, and `music`, normalize to −3 dBFS | delete | docs/WORKFLOW_GUIDE.md:787-789: "`normalize_audio` to -3 dBFS, then `pair_audio`, as every template that muxes". TASKS.md:436-438. RELEASING.md:350 (`music`) | +| 49 | 563 | A variable's bound is declared by the author, checked three times (frame_snap shape, `snap: "up"`, effective value, entry fields, paths, catalog) | delete | dw/variable_constraints.py module docstring: "Checked in three places". :138 `aligned`: "108 is accepted (it becomes 124)". :356 `apply_constraints`: "emits a warning for each rounded value". :261 `constraint_errors` paths. :60 `entry_constraint_fields` | +| 50 | 563 (cont.) | LTX-2.5 declares `8n+1` with no `snap` because those pipelines floor | docstring | `dw/variable_constraints.py` (468), module docstring, one sentence. Pinned by tests/test_variable_constraints.py::test_the_ltx2_frame_rule_is_the_pipelines. Today the rationale is in no docstring | +| 51 | 592 | `vram_estimate`'s ceiling is projected per pipeline step after expansion (own args, `gb_per_reference`, null refs free, largest reported, entry path, validate + run time) | delete | dw/vram_estimate.py module docstring: "a `for_each` member is projected with its own `num_frames`". :324 `vram_estimate_errors`: "One step is reported - the largest". :400 `apply_vram_estimate` (run-time backstop). :274 `_where` | +| 52 | 592 (cont.) | H3 Ref2VA numbers: 16.0 / 28.71 / 1.0 and 243/209/175/141 frames | map | H3 Ref2VA VRAM numbers \| `workflows/templates/minimax/*` `vram_estimate` \| every Ref2VA template declares base 16.0 GB, 28.71 B/voxel and 1.0 GB per reference, a field classification rather than a fit \| tests/test_h3_vram_ceiling.py::test_every_ref2va_minimax_template_declares_gb_per_reference | +| 53 | 592 (cont.) | A workflow with no `vram_estimate` inherits the catalog's (identity match, warns never refuses, no run-time backstop) | delete | dw/vram_inheritance.py module docstring: "An inherited ceiling warns and never refuses ... There is no run-time backstop" | +| 54 | 592 (cont.) | `ceiling_index` in `dw/server/deps.py` cached against listing mtimes; same-identity templates agree | map | Inherited-VRAM index \| `dw/vram_inheritance.py`, `dw/server/deps.py` (`ceiling_index`) \| built from single-identity templates, cached on the listing's mtimes, and templates of one identity must declare the same numbers \| tests/test_vram_inheritance.py::test_every_template_declaring_one_identity_declares_the_same_numbers | +| 55 | 626 | A joined video records its shots, measured (constructors decided) | delete | dw/shots.py module docstring: "*measured* from the waveform the join built rather than derived". "`tests/test_shots.py` fails on a constructor site nobody decided for" | +| 56 | 626 (cont.) | `saved_shots` survives the step cache's stripped copy; manifest/`step_end` renamed `shot@`; mp4 carries nothing | docstring | `dw/shots.py` (411), module docstring, ~3 lines. `recorded_shots` readback is already at runs.py:739 | +| 57 | 641 | Step cache: singleton, keyed `(workflow id, step name)` + borrow chain, root-validated, off without `seed`, `reused: true`, rerun `new_seed` | map | Step cache \| `dw/step_cache.py`, `dw/workflow_run.py` (`cache_lookup`) \| a seeded rerun with unchanged inputs is served from the cache (`reused: true`, nothing written), and `rerun(new_seed=true)` is the way to a different result \| tests/test_worker_execute.py::test_full_cleanup_clears_step_cache. The rule is already in step_cache.py's module docstring and `step_pipeline_keys`:145, WORKSPACES.md:215, SERVER.md:176. The map row is needed for Review Focus 4c item 5 ("why a seeded rerun generates nothing") | +| 58 | 644 | Assessment probes measure and decide nothing (streaming, `assessment=True`, `list_tasks` group, shot-boundary order, `hard_cut`, rules table pinned) | delete | dw/tasks/assess.py module docstring: "Nothing acts on a finding". dw/assessment_rules.py module docstring: "`seam_frame_jump` does not fire at a seam whose incoming shot is marked `hard_cut`". TASKS.md:1144-1148 (`assessment=True`, `attribute_voices` "is not one") | +| 59 | 644 (cont.) | A probe step must save `application/json` | delete | dw/scalar_result_validation.py module docstring: "So a `result` on one is allowed and must say `application/json`." | +| 60 | 644 (cont.) | `GET /api/gallery/{name}/assess`: sync route, one decode, merged findings, `not_applicable`, `probe=` whitelisted | delete | dw/server/assess.py module docstring: "the route is a sync `def` ... One request is one decode". :84 `assess`. :35 `unknown_probe` | + +# CLAUDE.md triage, part 2: everything except the root's "Critical Gotchas" + +Base: worktree `stabilization/phase-4c` at `2270d730`. Scope: root `CLAUDE.md` 1-273 and 670-674; +`ui/CLAUDE.md`; `dw_mcp/CLAUDE.md`; `dw/server/CLAUDE.md`; `.github/copilot-instructions.md`. +Line ranges tile every non-blank line of each in-scope range; a heading is folded into the first row under it. +A long paragraph is split into its distinct rules where they get different verdicts (a single line may appear in +several rows, each naming its clause). + +## Summary + +### Rows per verdict per file + +| File | rows | delete | docstring | map | keep | +| --- | --- | --- | --- | --- | --- | +| root `CLAUDE.md` (1-273, 670-674) | 81 | 56 | 4 | 9 | 12 | +| `dw/server/CLAUDE.md` (34) | 9 | 5 | 0 | 3 | 1 | +| `dw_mcp/CLAUDE.md` (95) | 22 | 15 | 1 | 2 | 4 | +| `ui/CLAUDE.md` (154) | 25 | 18 | 0 | 0 | 7 | +| `.github/copilot-instructions.md` (114) | 24 | 22 (4 of them stale) | 0 | 2 | 0 (becomes the <=10-line pointer) | +| **total** | **161** | **116** | **5** | **16** | **24** | + +(copilot "delete" counts both "delete (redundant: ...)" and "delete (stale: ...)" rows.) + +### Projected kept lines + +| File | now | projected (this scope) | +| --- | --- | --- | +| root, lines 1-273 + 670-674 (278 lines) | 278 | ~55: header 3, overview 3, commands ~22 (25 today; the long comments can lose 3), server/MCP/UI pointers 3, plugin skills ~7, `get_device_type` gotcha 1, Security Rules 9 + CodeQL 2, schema pointer 2, headings ~3. Plus the "where things are" block (new, ~10-15) and whatever the Critical Gotchas half keeps. | +| `dw/server/CLAUDE.md` | 34 | ~3 (heading + pointer to docs/SERVER.md and the seam map) | +| `dw_mcp/CLAUDE.md` | 95 | ~10 (heading, two surface-text authoring rules, pointer) | +| `ui/CLAUDE.md` | 154 | ~22 (build/check commands, WCAG, `--live` whitelist, proof principle, read-field-only rule, pointer to `src/app.css` header and docs/SERVER.md) | +| `.github/copilot-instructions.md` | 114 | <=10 pointer (per ruling; not counted by the metric) | + +### Docstring destinations (line counts against 1,000) + +| Module | now | receives | +| --- | --- | --- | +| `dw/__init__.py` | 509 (no module docstring) | root #64: the default torch device is deliberately never set (today a comment in `startup()`) | +| `dw/settings.py` | 98 (no module docstring) | root #74: the standing settings keys and what overrides each | +| `dw/server/job_history.py` | 437 | root #16: the `workspace` column, backfilled to `default` | +| `dw/realize.py` | 257 | root #44: the realized workflow keeps `for_each` unexpanded; the manifest names the members | +| `dw_mcp/__init__.py` | 6 | mcp #3: what the MCP surface does not cover (SSE stream, the two bulk zips, the SPA mount) | + +None comes near 1,000. Routed to map instead of a docstring because of the line budget: the Type System owner +row (#31) and the explicit-reference ordering (#37) would belong in `dw/arguments.py`, which is 957 lines with no +module docstring. + +### Seam-map rows this half asks for + +| # | concept | owner module(s) | rule (one sentence) | enforced by | +| --- | --- | --- | --- | --- | +| root 31 | Type conversion (`*_type`, `{}` escape, dotted names) | `dw/arguments.py` (`realize_args`), `dw/type_helpers.py` | Keys ending `_type`/`_dtype` or named `dtype` load a Python object at realize time; the rules are WORKFLOW_GUIDE "Types and escaping". | `tests/test_type_helpers.py::TestGetType::test_get_type_from_diffusers` | +| root 37 | Explicit references resolve first | `dw/arguments.py` (`_realize_explicit_reference`), `dw/assets.py`, `dw/runs.py`, `dw/prompts.py`, `dw/references.py` | `asset:`/`output:`/`constant:`/`prompt:` resolve in `realize_args` before any key-name convention, whatever `apply_key_conventions` says; `asset:` is confined to the asset library root and `output:` to the output root, each through a `dw/security.py` validator (`validate_path` against the root). | `tests/test_assets.py::TestReferences::test_a_symlink_out_of_the_library_is_refused`, `tests/test_security_symlinks.py::TestOutputs::test_an_output_reference_does_not_follow_a_linked_run_directory` | +| root 52 | IC-LoRA reference scale | `workflows/templates/ltx2/*` | The three conditioning templates run `reference_downscale_factor: 1`, the two upscalers 2. | `tests/test_ltx2_ic_loras.py::TestEachTemplateMatchesItsCard::test_the_reference_is_encoded_at_the_output_resolution`, `::TestTheGenerativeUpscaleMatchesTheSameCard::test_it_loads_the_upscaler_at_factor_two` | +| root 56 | IC-LoRA numbers | `workflows/templates/ltx2/*` | Every number in the IC-LoRA templates is the vendor card's. | `tests/test_ltx2_ic_loras.py::TestEachTemplateMatchesItsCard::test_the_strength_is_the_cards_default`, `::TestEachTemplateMatchesItsCard::test_the_defaults_are_the_trained_bucket` | +| root 57 | IC-LoRA prompt genre | `prompts/ltx2/*` (tag `ic-lora`) | An IC-LoRA stored prompt is checked against its trained form, not the 150-220-word T2V caption rule. | `tests/test_ltx_prompt_library.py::test_an_ic_lora_prompt_is_in_its_trained_form` | +| root 58 | Per-repo gating | `dw/plan.py` (`downloads_required`) | Each gated repo (IC-LoRAs are `gated: auto`) is granted separately, so a box with one can 403 on another; `downloads_required` names each. | - | +| root 73 | On-demand placement wrappers | `dw/pipeline_processors/placement.py` (`apply_on_demand_placement`) | The wrappers keep the wrapped signature (`functools.wraps`; H3's denoiser reads `signature(transformer.forward)`), and on-demand/group-offload components suppress the wholesale `pipeline.to(device)`. | `tests/test_modular_pipeline.py::TestOnDemandResidency::test_the_wrapped_signature_survives`, `::TestOnDemandResidency::test_group_offload_and_on_demand_together_are_rejected`, `::TestOffloadDevice::test_component_group_offload_is_not_moved_to_device` | +| root 76, 79 | CodeQL path-injection model | `.github/codeql/dw-security/`, `dw/security.py` | The local pack models `dw/security.py`'s validators as sanitizers (advanced setup, since default setup cannot load a pack); a moved validator is re-modelled in the same commit. | CodeQL `dw/path-injection` (`.github/workflows/codeql.yml`) | +| server 1 | Server package layout | `dw/server/app.py`, `routes/`, `deps.py`, `outputs.py`, `catalog.py`, `http_security.py`, `admission.py`, `jobs.py`, `job_history.py`, `job_record.py` | `create_app` builds state and registers one router per resource in `ROUTERS` order; routers call the services. | `routes/__init__.py` order (tests on route order) | +| server 6 | Engine/server import direction | `dw/` vs `dw/server/` (`EXPORTS_SUBDIR` in `dw/workspace.py`) | The engine never imports `dw.server`; a constant both need lives engine-side and the server re-imports it. | `scripts/arch_metrics.py` `import_cycles` = 0 | +| server 8 | Archives never follow links | `dw/server/outputs.py` (`zip_download`), `dw/security.py` (`contained`) | Listings drop and archives skip a symlink leaving its root. | `tests/test_security_symlinks.py::TestOutputs::test_the_archive_route_does_not_follow_the_link`, `::TestOutputs::test_the_gallery_listing_does_not_enumerate_the_link` | +| mcp 14 | `dw_mcp` stays torch-free | `dw_mcp/` | `dw_mcp` is a top-level package and imports no `dw.*` module, since `dw/__init__.py` pulls in torch. | `tests/test_mcp_server.py::TestStartupWeight::test_the_server_starts_without_importing_the_engine` | +| mcp 21 | MCP surface text budget | `dw_mcp/server.py`, `dw_mcp/tools_*.py` | The instructions and each tool description stay <= 2,048 characters (Claude Code truncates past that), and the surface total within `SURFACE_BUDGET`. | `tests/test_mcp_server.py::test_no_text_the_agent_reads_is_cut_off_by_the_client`, `::test_the_tool_surface_fits_the_budget` | +| copilot 7 | Core engine owners | `dw/workflow.py`, `dw/workflow_run.py`, `dw/validation.py`, `dw/step.py`, `dw/pipeline_processors/` (`pipeline.py`, `placement.py`, `components.py`, `adapters.py`, `progress.py`), `dw/previous_results.py` | One row each in the seam map (the plan's owner list already covers them). | - | +| copilot 13 | Adding a task | `dw/tasks/task.py` (`register_command`) | A task is a function registered with `@register_command`; its implementation's signature is its argument schema. | `tests/test_task_discovery.py::TestDescribeTask::test_signature_becomes_the_schema` | + +### Concerns + +1. **Non-`docs/` evidence.** Several deletes cite docs that are not `docs/*.md` or a Python docstring: + `workflows/templates/ltx2/README.md` and `plugins/dw/skills/ltx-2.5/SKILL.md` (root #51, 53-55), + `plugins/dw/README.md` (root #9-11), and for `ui/CLAUDE.md` the header comment of `ui/src/app.css` and + `//`/`` comments in `AssetsPage.svelte` (UI has no docstrings; the brief lists ui/src as evidence). Each + such cell says so. If the reviewer rules those out, the ltx2 rows become map rows on `workflows/templates/ltx2/` + and the UI rows become keep. +2. **The design system is already in code.** Contrary to the plan's expectation that `ui/CLAUDE.md`'s design rules + stay, `src/app.css:1-24` states the contact-sheet, colour-means-state and mono-for-the-engine rules almost word + for word, so they triage as delete. The `--live` whitelist (UI 13) and WCAG AA (UI 7) are not in app.css and + have no check, so they stay. The kept file should point at the app.css header. +3. **Copilot "stale" claims, checked.** The 5-minute timeout is stale (`docs/WORKER_GUIDE.md:40`: "There is no + execution timeout"). The memory-growth warning (>500MB) is **not** stale: `dw/worker.py:86,604`. Also stale: + SHA-256 workflow change detection, "Adding Pipeline Types: update workflow_schema.json" (`component_type` is a + free-form string, `dw/workflow_schema.json:584-587`) and "image results default to JPEG" (no `content_type` + means nothing is saved, `dw/result.py:306-307`). +4. **Stale detail in the root.** Line 109-110 says `builtin:` is resolved in `dw/workflow.py`; it is now + `dw/library.py` (`builtin_root`, `resolve_sub_workflow`). Nothing in the diet should carry it forward. +5. **"Change both" (root #50).** The ruling resolves it (phase-4.md:415). `docs/WORKFLOW_GUIDE.md` has no reciprocal + "change both" sentence, so Task 5 has no pointer edit to make there. +6. **`dw/variables.py` has no owner row.** Root #45-46 delete on WORKFLOW_GUIDE evidence, but the owner (`resolve_variable_values`, `undeclared_variable_references`; no module docstring, 459 lines) is not in the plan's seam-map owner list. Flag for the Critical Gotchas half / Task 2 (`argument_errors` lives there too). +7. **No doc points into an in-scope section.** `grep -rn CLAUDE.md .claude/skills docs plugins` hits only cold-drill-example.md:10,37 and AGENT_LOOP.md:94 (generic mentions) and WORKFLOW_GUIDE.md:372 (Critical Gotchas half). Nothing to redirect from this half. +8. **Line 239 and 245 are single lines carrying 3-4 rules each.** They're split into separate rows (#63-65, + #68-71), so the row count runs higher than the paragraph count. + +--- + +## CLAUDE.md (root, lines 1-273 and 670-674) + +| # | file:line | first words | verdict | evidence | +| --- | --- | --- | --- | --- | +| 1 | CLAUDE.md:1-3 | "# CLAUDE.md / This file provides guidance" | keep | File header; 2 lines. | +| 2 | CLAUDE.md:5-7 | "## Project Overview / `diffusers-workflow` is a declarative" | keep | The two-sentence overview the ruling keeps (phase-4.md, "What stays"). | +| 3 | CLAUDE.md:9-33 | "## Common Commands / ```bash # Install" | keep | The common commands the ruling keeps. The long `dw.run` comment (18-21) can shrink to one line. | +| 4 | CLAUDE.md:35-40 | "## Architecture / ### Server & Web UI / `dw/serve.py` runs a FastAPI app" | delete | docs/WORKER_GUIDE.md:5-6 "`dw.serve` runs a persistent worker subprocess"; dw/server/jobs.py:3 "One runner thread executes jobs FIFO against the single GPU worker"; docs/SERVER.md:68 "`~/.diffusers_helper/jobs.sqlite`". | +| 5 | CLAUDE.md:41 | "See docs/SERVER.md, `dw/server/CLAUDE.md` and `ui/CLAUDE.md`." | keep | Becomes a line of the "where things are" block: an agent needs it before it knows which file to open. | +| 6 | CLAUDE.md:43-45 | "### MCP Server / The stdio MCP server lives in `dw_mcp/`" | keep | Pointer, same block. | +| 7 | CLAUDE.md:47-55 | "### Claude Code plugin / `.claude-plugin/marketplace.json` publishes" (through "pinned by the same test.") | keep | The plugin-skill names, each in backticks: `tests/test_plugin_skills.py::test_a_skill_is_enumerated_where_the_plugin_describes_itself`. | +| 8 | CLAUDE.md:55-56 | "Model knowledge lives there and in the catalog, never in engine code" | keep | An edit-time rule no test or check catches; one line, pointing at `plugins/dw/` and the catalog. | +| 9 | CLAUDE.md:56-57 | "every number a skill states is pinned to a diffusers symbol" | delete | plugins/dw/README.md:38-39 "Each skill quotes catalog names and numeric rules that `tests/test_plugin_skills.py` holds to the catalog and to the diffusers module that enforces them." (Not docs/*.md; concern 1.) | +| 10 | CLAUDE.md:57-58 | "`plugin.json`'s version is the engine's, bumped by `scripts/release.sh`" | delete | plugins/dw/README.md:39-40 "The plugin's version is the engine's; the release script bumps both." | +| 11 | CLAUDE.md:58 | "Adding or re-auditing a family is `.claude/skills/model-family-onboarding/`" | delete | plugins/dw/README.md:42-44 "The repo's `model-family-onboarding` skill (`.claude/skills/`) is the full lifecycle"; the skill is also listed in every session. | +| 12 | CLAUDE.md:60-62 | "### Worker / A **persistent worker subprocess**" | delete | docs/WORKER_GUIDE.md:5-16 "persistent worker subprocess (`dw/worker.py`, managed by `dw/worker_manager.py`)", "uses the `spawn` multiprocessing start method", "over `multiprocessing.Queue`s". | +| 13 | CLAUDE.md:64-70 | "### Workspaces on the server / `dw.serve` can hold several workspaces" (default + named + one prompt library) | delete | docs/WORKSPACES.md:255-261 "The root's own three folders are the workspace named `default`... there is one prompt library: `prompt:` is shared by reference"; dw/workspace.py:401-407 `named_workspace` docstring. | +| 14 | CLAUDE.md:70-72 | "Routes take an optional `workspace`; omitting it means the default" | delete | docs/WORKSPACES.md:293-295 "Every scoped route takes an optional `?workspace=`; omitting it means `default`". | +| 15 | CLAUDE.md:72-75 | "A job carries its own `output_dir`, `asset_dir` and `workflow_dir`" | delete | dw/server/jobs.py:142-145 (`submit`) "`output_dir`, `asset_dir` and `workspace` name which workspace this job runs in. They travel with the job"; dw/assets.py:50-51 (`get_asset_dir`) "A library activated for this run wins outright - that is the server telling the worker which workspace's assets this job uses". | +| 16 | CLAUDE.md:75-76 | "`jobs.sqlite` has a `workspace` column, backfilled to `default`" | docstring | `dw/server/job_history.py` (437 lines), module docstring. Today only code (job_history.py:58-61). | +| 17 | CLAUDE.md:76-82 | "`common/assets` at the root is the one asset library every workspace shares" | delete | docs/WORKSPACES.md:268-284 "`common/assets/` is the one place an asset can live that belongs to no single workspace", "a workspace's own name still shadows a shared one", the three `shared` write paths. | +| 18 | CLAUDE.md:83-84 | "Reserved names: `workflows`, `prompts`, `assets`, `outputs`, `exports`, `common`" | delete | docs/WORKSPACES.md:261-266 "reserved names ... Two more names are reserved ... `exports` and `common`"; dw/workspace.py:94-96 `RESERVED_WORKSPACE_NAMES`. | +| 19 | CLAUDE.md:84-86 | "The web UI is organised by workspace, with a sidebar" | delete | docs/WORKSPACES.md:299 "The sidebar lists every workspace; the selected one is named in the hash (`#/ws//...`)". | +| 20 | CLAUDE.md:88-91 | "The web UI has a page for it: `ui/src/lib/pages/AssetsPage.svelte`" | delete | ui/src/lib/pages/AssetsPage.svelte:2-7 "The input side of the gallery ... folder groups, a contact-sheet grid, a detail popout" and :53-55 "the page never builds a path" (file-level comment; concern 1); ui/CLAUDE.md:93-103 if that survives. | +| 21 | CLAUDE.md:93-99 | "### Library search paths / `dw/library.py` owns the three content libraries' search paths" | delete | dw/library.py:1-26 module docstring "Reads resolve front to back; writes only ever go to the front"; `library_path` docstring (library.py:390-401) gives the order per kind incl. "the library every workspace under the root shares (`common`...)". | +| 22 | CLAUDE.md:100-103 | "`find`/`entries` span every root front-to-back, so an earlier name shadows" | delete | dw/library.py:175-178 (`LibraryPath`) "a name resolves front to back and only inside its root (a symlink that leaves it is a miss) ... later copies as shadowed; saves go to the writable root"; library.py:16-18 "Saving over a read-only workflow ... writes a copy"; docs/WORKSPACES.md:131-132 "deleting something from a read-only root is refused with a 403". | +| 23 | CLAUDE.md:104-105 | "`dw.serve` pins the read-only tails into `DW_PROMPT_PATH` / `DW_ASSET_PATH`" | delete | dw/library.py:499-500 (`pin_library_path`) "Pin a library's read-only roots in the environment, so the worker subprocess resolves a reference exactly as the entry point would." | +| 24 | CLAUDE.md:105-107 | "A job carries the root it is confined to (`JobManager.submit(workflow_dir=...)`)" | delete | dw/server/jobs.py:136-140 (`submit`) "`workflow_dir` overrides this job's confinement root for a workflow that lives outside the writable directory - an example or a builtin". | +| 25 | CLAUDE.md:107-108 | "All three listings share one envelope" | delete | docs/WORKSPACES.md:144-149 "`libraries` (`[{origin, root, writable}]` ...) and tags every entry ... `origin` and `writable` ... listed under `shadowed`"; :172-173 "the same envelope as the workflow listing"; dw/library.py:316-318 (`shadowed_listing`). | +| 26 | CLAUDE.md:109-110 | "Packaged `dw/workflows/` is off the path" | delete | docs/WORKSPACES.md:177-180 "The packaged workflows in `dw/workflows/` are deliberately *not* on the path". Also stale: `builtin:` now resolves in `dw/library.py` (`builtin_root`, :73-74), not `dw/workflow.py`. | +| 27 | CLAUDE.md:112-119 | "### Workspaces / `dw/workspace.py` resolves the one directory" (order, rule four, `ensure()`) | delete | dw/workspace.py:18-34 module docstring: the five-step order, "Rule 4 is what keeps a checkout working unchanged", "Nothing here creates a directory ... calls ensure()". | +| 28 | CLAUDE.md:120-124 | "`set_workspace` pins the root *and* how it was chosen" | delete | dw/workspace.py:271-273 (`set_workspace`) "Pin a resolved workspace in the environment, so a spawned worker subprocess ... agree"; :326-328 (`discover_library`) "a workspace merely inferred from the working directory or fallen back to must not preempt a library a workflow already reaches". | +| 29 | CLAUDE.md:124-125 | "`--workflow-dir`, `--output-dir` and `--prompt-dir` each still override one folder" | delete | docs/WORKSPACES.md:45 "## Overriding one folder" (section states it). | +| 30 | CLAUDE.md:125-127 | "See docs/WORKSPACES.md; the later stages ... are documented above" | delete | Cross-reference inside this file; the docs/WORKSPACES.md pointer moves to the "where things are" block. | +| 31 | CLAUDE.md:129-131 | "### Type System / `arguments.py` + `type_helpers.py` handle dynamic type conversion" | map | Type conversion / `dw/arguments.py` (`realize_args`), `dw/type_helpers.py` / keys `*_type`, `*_dtype`, `dtype` load a Python object at realize time. `dw/arguments.py` is 957 lines, so no docstring move. | +| 32 | CLAUDE.md:132 | "Keys ending in `_type` or `_dtype`, or named `dtype`, are auto-converted" | delete | docs/WORKFLOW_GUIDE.md:362-364 "Any key ending in `_type` or `_dtype`, or named `dtype`, has its string value loaded as a Python object: `"FluxPipeline"` from `diffusers`". | +| 33 | CLAUDE.md:133 | "Values wrapped in `{}` are escaped" | delete | docs/WORKFLOW_GUIDE.md:365-366 "Wrapping a value in braces keeps it a plain string — `"{nf4}"` is the string `nf4`". | +| 34 | CLAUDE.md:134 | "Dotted names use full module path" | delete | docs/WORKFLOW_GUIDE.md:364-365 "a dotted name (`"torch.bfloat16"`, `"sdnq.SDNQConfig"`) by full module path". | +| 35 | CLAUDE.md:135-137 | "Values prefixed with `constant:` read a value declared in python" | delete | docs/WORKFLOW_GUIDE.md:1824-1826 "Reference it with `constant:` and its dotted python name instead of copying it"; :1847 "anything callable is refused". | +| 36 | CLAUDE.md:138-142 | "Values prefixed with `asset:` resolve to the path of a file" (rooted, confined, loaded as a path) | delete | docs/WORKFLOW_GUIDE.md:1907-1908 "An `asset:` reference is rooted at the asset library instead"; :1923-1925 "resolves to that file's path ... What loads the path is unchanged"; :1932-1933 "A reference can only name a file inside the library"; dw/assets.py:10-13. | +| 37 | CLAUDE.md:139-140 | "Resolved in `realize_args` before every other convention" | map | Explicit references resolve first / `dw/arguments.py` (`realize_args`, `_realize_explicit_reference`), `dw/assets.py`, `dw/runs.py`, `dw/prompts.py` / an explicit reference resolves before any key-name convention. Partly in `realize_args` docstring (arguments.py:104-108); arguments.py is at 957, so map. | +| 38 | CLAUDE.md:142-144 | "The library is `DW_ASSET_DIR` / `--asset-dir`, else the workspace's `assets/`" | delete | docs/WORKFLOW_GUIDE.md:1927-1930 "the `DW_ASSET_DIR` environment variable ..., then the workspace's `assets/` when a workspace was named explicitly, then `./assets` ..., then ... walking up"; dw/assets.py:52-55 (`get_asset_dir`). | +| 39 | CLAUDE.md:145-152 | "Values prefixed with `output:` resolve to the path of a file an earlier run wrote" | delete | docs/WORKFLOW_GUIDE.md:1950-1960 "Writing `latest` where the run id goes resolves to the newest run ... *that holds the file*", "`v4` not holding the file is an error, not a reason to try `v3`", "only select a run where run directories are"; :1967-1968 "cannot leave it"; dw/realize.py:8-9 "which run 'output:.../latest/...' picked". | +| 40 | CLAUDE.md:153-156 | "A generated file becomes a stable input with `POST /api/assets/keep`" | delete | docs/WORKFLOW_GUIDE.md:1974-1978 "`POST /api/assets/keep` (... MCP's `keep_output`) copies it into the workspace's asset library ... stable whatever happens to the run directory"; docs/SERVER.md:493 "a hard link where the filesystem allows one". | +| 41 | CLAUDE.md:157-162 | "Values prefixed with `prompt:` load a stored prompt's `text`" | delete | docs/WORKFLOW_GUIDE.md:1888-1895 (location order) and :1896 "References are rooted at that one directory - not at the workflow file"; dw/prompts.py:27-33 (`get_prompt_dir`) "DW_PROMPT_DIR ... a named workspace, then ./prompts, then a walk up from base_dir, then the workspace's prompts/ as the fallback". | +| 42 | CLAUDE.md:163-166 | "A step's `result.subfolder` names a subfolder of the run directory" | delete | dw/subfolders.py:5-9 "'subfolder' on a step's result block puts that step's files into a subfolder ... by convention 'final' or 'intermediate', though the engine treats no name specially"; docs/WORKFLOW_GUIDE.md:846-857 "applies no default", "a relative path of any depth (`shots/act-1`), may be a `variable:` or ... an `item:` reference". | +| 43 | CLAUDE.md:167-177 | "A step carrying `for_each` ... is expanded by `expand_for_each`" (naming, `item:`, `gather:`, pairing, `@`, 32 max, release on last) | delete | dw/for_each.py:1-27 module docstring (runs after substitution, before the reference check; `item:`; pairing "'slice' inside 'shot@open' -> 'slice@open'"; `gather:`; "'previous_result:shot' naming a group is an error"; "'@' is reserved"); docs/WORKFLOW_GUIDE.md:539-546 (names; "must be unique in its list"), :600-604 ("at most 32 entries", "`release_pipeline` on a `for_each` step releases after the *last* member"). | +| 44 | CLAUDE.md:177 | "The realized workflow keeps `for_each`; the manifest names the members" | docstring | `dw/realize.py` (257 lines), module docstring. The manifest half is in docs/WORKFLOW_GUIDE.md:540-541 ("Those are the names the manifest, the job's events and the gallery show"); the realized half is stated nowhere. | +| 45 | CLAUDE.md:177-182 | "An entry of a list-valued variable may reference another variable" | delete | docs/WORKFLOW_GUIDE.md:555-565 "An entry may name another variable ... resolved before anything in the entry is loaded, and an undeclared one is a validation error at the entry's path ... A value may not reference itself". | +| 46 | CLAUDE.md:182-185 | "The catalog derives `lists` (`list_fields`, `dw/for_each.py`)" | delete | docs/WORKFLOW_GUIDE.md:605-606 "the listing's `lists` block names the fields an entry takes and the steps over it"; :615-616 "An entry key no step reads is a validation warning at the entry's path". | +| 47 | CLAUDE.md:185-186 | "A `cost` entry may carry `per_entry` ... measured, never derived" | delete | docs/WORKFLOW_GUIDE.md:606-612 "`cost` carries `per_entry` once one entry has been measured ... without `per_entry` it extrapolates" (`basis: derived`). | +| 48 | CLAUDE.md:186-187 | "An empty `for_each` list is an error; `expanded_definition` realizes constants first" | delete | docs/WORKFLOW_GUIDE.md:600-603 "an empty list is a validation error ... Validation realizes a `constant:` default before checking it". | +| 49 | CLAUDE.md:188-193 | "Every run directory holds `workflow.json` beside its manifest" | delete | dw/realize.py:1-9 module docstring ("the arguments a caller passed, the seed a seedless workflow drew, the text a stored prompt held ... which run 'output:.../latest/...' picked", "never fails a run"); docs/WORKSPACES.md:204-206; docs/WORKFLOW_GUIDE.md:333-339 (`get_job_workflow`, `save_workflow`, `export_job`). | +| 50 | CLAUDE.md:195-197 | "The same conventions, written for an agent ... change both when one changes" | delete | Resolved by ruling: docs/stabilization/phase-4.md:415 "the guide is the owner ...; CLAUDE.md points at it". The pointer goes in the "where things are" block. | +| 51 | CLAUDE.md:199-204 | "### LTX-2.5 IC-LoRAs / The catalog's IC-LoRA templates are two upscalers" (inventory, `LTX2InContextPipeline`) | delete | workflows/templates/ltx2/README.md:43-45, 53, 66-67 (each template's row); plugins/dw/skills/ltx-2.5/SKILL.md:44-57. Not docs/*.md (concern 1). | +| 52 | CLAUDE.md:204-205 | "the three run at `reference_downscale_factor: 1` (the upscalers' is 2)" | map | IC-LoRA reference scale / `workflows/templates/ltx2/*` / conditioning templates 1, upscalers 2 / `tests/test_ltx2_ic_loras.py::TestEachTemplateMatchesItsCard::test_the_reference_is_encoded_at_the_output_resolution`, `::TestTheGenerativeUpscaleMatchesTheSameCard::test_it_loads_the_upscaler_at_factor_two`. | +| 53 | CLAUDE.md:205-208 | "`upscale-clip` runs the upscaler over the caller's own `source_video`" | delete | workflows/templates/ltx2/README.md:45 "with its soundtrack paired back on by `pair_audio`"; SKILL.md:56-60 "silent source refused". (Concern 1.) | +| 54 | CLAUDE.md:209-213 | "`reference-sheet` drives Ingredients — the family's only identity route" | delete | workflows/templates/ltx2/README.md:53 "The family's identity route ... looped into a static video by a `loop_frames` step because the LoRA reads it through a 121-frame bucket". (Concern 1.) | +| 55 | CLAUDE.md:213-214 | "`restore-deblur` and `restore-decompression` each invert one defect" | delete | workflows/templates/ltx2/README.md:66-67 "Spatial defocus only - not motion blur, not noise", "Not a deblur and not an upscale". (Concern 1.) | +| 56 | CLAUDE.md:214-215 | "Every number in the three is the vendor card's and is pinned by `tests/test_ltx2_ic_loras.py`" | map | IC-LoRA numbers / `workflows/templates/ltx2/*` / every number is the vendor card's / `tests/test_ltx2_ic_loras.py::TestEachTemplateMatchesItsCard::test_the_strength_is_the_cards_default`, `::TestEachTemplateMatchesItsCard::test_the_defaults_are_the_trained_bucket`. Test-only. | +| 57 | CLAUDE.md:215-218 | "the trained caption form is a *different* genre ... tagged `ic-lora`" | map | IC-LoRA prompt genre / `prompts/ltx2/` (tag `ic-lora`) / checked against its trained form, not the paragraph rule / `tests/test_ltx_prompt_library.py::test_an_ic_lora_prompt_is_in_its_trained_form`. Test-only. | +| 58 | CLAUDE.md:219-220 | "The weights are `gated: auto` on Hugging Face — per repo" | map | Per-repo gating / `dw/plan.py` (`downloads_required`) / each gated repo is granted separately / no check. SKILL.md:138-139 names the gated weight but not the per-repo point, too weak for delete. | +| 59 | CLAUDE.md:220-224 | "A `loras` entry counts toward `plan.downloads_required`" | delete | dw/plan.py:807-812 (`_collect_sources`) "A `loras` entry counts too. It carries its repo under `model_name` directly ... answered `downloads_required: []` and then pulled it mid-run". | +| 60 | CLAUDE.md:226-228 | "### Quantization Support / Quantization configs are defined per-component" | delete | docs/QUANTIZATION.md:3 "applied per-component in the pipeline - and any other backend with a config class (optimum-quanto, for example) works through the same dynamic `config_type` import"; :211-215. | +| 61 | CLAUDE.md:230 | "SDNQ pre-quantized models use a different pattern" | delete | docs/QUANTIZATION.md:163-175 "`pre_load_modules` — Imports sdnq before pipeline loading (registers quantization method)", "`sdnq_optimize` — ... (CUDA/XPU only, skipped on MPS/CPU)". | +| 62 | CLAUDE.md:232-237 | "### Cross-Platform Device Support / `dw/__init__.py` handles device detection" | delete | docs/WORKFLOW_GUIDE.md:1465 "Device is auto-detected (CUDA > MPS > CPU)"; docs/ACCELERATION.md:261-270 (TF32/cuDNN settings), :291-293 (attention slicing "2.4x slower ... It helps SD 1.5", MPS watermark). | +| 63 | CLAUDE.md:239 (clause 1) | "Detection is overridden by the `DW_DEVICE` environment variable" | delete | dw/__init__.py:134-137 (`get_device`) "An explicit choice wins over detection - the DW_DEVICE environment variable for a single run, or the 'device' setting". | +| 64 | CLAUDE.md:239 (clause 2) | "no default torch device is set, since that would build models directly in VRAM" | docstring | `dw/__init__.py` (509 lines, no module docstring). Today a comment in `startup()` (:432-436), not a docstring. | +| 65 | CLAUDE.md:239 (clause 3) | "Compare backends with `get_device_type()` rather than `== "cuda"`" | keep | An edit-time gotcha no test or check catches (the docstring at dw/__init__.py:162-163 is in a file nobody opens before writing a comparison). One line, pointing at `dw/__init__.py`. | +| 66 | CLAUDE.md:241 | "A step can override the device it runs on" | delete | docs/WORKFLOW_GUIDE.md:1474-1477 "A step can name a device instead, in a pipeline `configuration` ..., in a component `configuration`, or in a task's `arguments`". | +| 67 | CLAUDE.md:243 | "Every device a workflow names passes through `resolve_device()`" | delete | dw/__init__.py:209-216 (`resolve_device`) "The backend is what gets translated - an index is left alone when the backend matches ... A CPU device is never rewritten"; docs/WORKFLOW_GUIDE.md:1487-1492; docs/ACCELERATION.md:290 (the MPS adaptations). | +| 68 | CLAUDE.md:245 (clause 1) | "SDNQ `quantization_device`/`return_device` go through `resolve_device()`" (+ matmul off on MPS, streams dropped) | delete | docs/ACCELERATION.md:290 "a `cuda` device becomes `mps`, SDNQ's `quantization_device` follows it and its quantized matmul is switched off (it runs through `torch._int_mm`, ~500x slower on MPS), group-offload CUDA streams are dropped". | +| 69 | CLAUDE.md:245 (clause 2) | "`vram_estimate` checks against the serving device's own `cost` entries" | delete | dw/vram_estimate.py:42 "The entries checked are the serving device's own (see `_entries_for`)"; :153 "capacity when it can report one, and against every entry when it cannot"; docs/ACCELERATION.md:295. | +| 70 | CLAUDE.md:245 (clause 3) | "`device_memory_stats()` reports real unified-memory figures on MPS" (+ CPU-fallback warning) | delete | docs/ACCELERATION.md:295 "Memory figures are real: allocated, driver-reserved, and Metal's recommended working set"; :294 "dw logs torch's warning when one does". | +| 71 | CLAUDE.md:245 (clause 4) | "`HF_ENABLE_PARALLEL_LOADING` defaults to off on macOS" | delete | dw/__init__.py:28-35 (`_parallel_loading_default`) "that segfaulted LTX-2.5's SDNQ transformer load ... diffusers reads the variable once at import ... so this goes by platform"; docs/ACCELERATION.md:297. | +| 72 | CLAUDE.md:247 (clause 1) | "A `components` entry can additionally set `residency: "on_demand"`" (mechanics, exclusive with `group_offload`) | delete | docs/WORKFLOW_GUIDE.md:1221-1234 "rests on the CPU and is moved to `device` around whichever of `forward`, `encode` and `decode`", "Cannot be combined with `group_offload`"; dw/pipeline_processors/placement.py:21-43 (`apply_on_demand_placement`). | +| 73 | CLAUDE.md:247 (clause 2) | "The wrappers use `functools.wraps` because callers introspect the signature" (+ suppresses `pipeline.to(device)`) | map | On-demand wrappers / `dw/pipeline_processors/placement.py` / wrappers keep the signature, and on-demand/group-offload suppress the wholesale move / `tests/test_modular_pipeline.py::TestOnDemandResidency::test_the_wrapped_signature_survives`, `::TestOffloadDevice::test_component_group_offload_is_not_moved_to_device`. Only comments today (placement.py:87-89, :360-362). | +| 74 | CLAUDE.md:249 | "Settings in `~/.diffusers_helper/settings.json` (`dw/settings.py`)" | docstring | `dw/settings.py` (98 lines, no module docstring). Today only per-field comments; docs/ACCELERATION.md:261 covers the CUDA keys only. | +| 75 | CLAUDE.md:251-259 | "## Security Rules / All entry points use `dw/security.py`'s validators" | keep | The ruling keeps the security rules: an agent needs them before it opens any file. | +| 76 | CLAUDE.md:261-266 | "CodeQL knows about these validators, which is why the scan is quiet" | map | CodeQL path-injection model / `.github/codeql/dw-security/`, `dw/security.py` / validators modelled as sanitizers; a moved validator is re-modelled in the same commit / CodeQL `dw/path-injection`. The `.qll` header says it, but that is not a doc or docstring. | +| 77 | CLAUDE.md:266-269 | "but it only holds while new filesystem access goes through a validator" | keep | Edit-time: a new filesystem access outside a validator is a real alert, to be treated as a finding. 1 line under Security Rules. | +| 78 | CLAUDE.md:269-271 | "`validate_path(path, base)` is modeled as a barrier only when `base` is not `None`" | keep | Edit-time: calling with `base=None` leaves the path reportable. 1 line. | +| 79 | CLAUDE.md:271-272 | "Scanning is advanced setup (`.github/workflows/codeql.yml`)" | map | Folded into #76's row. The source is .github/workflows/codeql.yml:1-4 header comment. | +| 80 | CLAUDE.md:670-672 | "## JSON Workflow Structure / The workflow schema is at `dw/workflow_schema.json`" | keep | A "where things are" pointer. | +| 81 | CLAUDE.md:674 | "File paths in workflows are relative to the workflow file. Built-in workflows use `"builtin:filename.json"`" | delete | docs/WORKFLOW_GUIDE.md:251-252 "`location` is a path relative to the workflow"; :159 "`builtin:name.json` for the packaged fragments in `dw/workflows/`"; docs/WORKSPACES.md:177-180. | + +--- + +## dw/server/CLAUDE.md (1-34) + +| # | file:line | first words | verdict | evidence | +| --- | --- | --- | --- | --- | +| 1 | dw/server/CLAUDE.md:1-3 | "# dw/server / Guidance for the HTTP server package. `app.py` is the factory" | map | Server package layout / `app.py`, `routes/`, `deps.py`, `outputs.py`, `catalog.py`, `http_security.py`, `admission.py`, `jobs.py`, `job_history.py`, `job_record.py` / one router per resource in `ROUTERS` order, calling the services. These are the plan's "server routers" rows. Each module's own docstring already describes it (app.py:1-8, routes/__init__.py:1-9, deps.py, outputs.py, catalog.py, http_security.py:1-5). | +| 2 | dw/server/CLAUDE.md:5-8 | "`dw.serve --mcp` additionally serves the MCP tool surface at `/mcp`" | delete | dw/server/mcp_mount.py:1-8 "the tools reach the REST API over the server's own bind address with the same token"; docs/REMOTE.md:14-21 "`--mcp` is refused outright (exit 2) on a non-loopback bind with no" token, and "install the systemd unit in contrib/systemd". | +| 3 | dw/server/CLAUDE.md:9-10 | "The `Origin` check accepts the request's own `Host` hostname" | delete | docs/SERVER.md:590-597 "rejected (403) unless its hostname is a loopback name, the configured `--host`, or the hostname the request itself was addressed to (`Host`)"; dw/server/http_security.py:77-85. | +| 4 | dw/server/CLAUDE.md:12-16 | "`guides.py` serves the prose guides (`GET /api/guides`)" | delete | dw/server/guides.py:114-120 (`_guide_file`) "build_dist.sh puts under dw/docs/ ... The checkout wins ... same rule default_ui_dir applies to the SPA"; guides.py:136-138 (unknown name refused against `GUIDES`); dw_mcp/guides.py:1-14 "These proxy `GET /api/guides`". | +| 5 | dw/server/CLAUDE.md:18-21, 24-26 | "`exports.py` gathers one finished job into a standalone directory" (+ route, `zip_url`, zip built on request) | delete | dw/server/exports.py:1-20 module docstring (the tree: workflow.json, manifest.json, job.json, README, assets/inputs/outputs); docs/SERVER.md:171-172 "a `zip_url`", "built on request rather than kept as a second copy". | +| 6 | dw/server/CLAUDE.md:21-24 | "`EXPORTS_SUBDIR` lives in `dw/workspace.py` rather than here" | map | Engine/server import direction / `dw/` vs `dw/server/` / the engine never imports `dw.server`; a constant both need lives engine-side / `scripts/arch_metrics.py` `import_cycles` = 0. | +| 7 | dw/server/CLAUDE.md:28-31 | "A listing that walks a root with `os.walk` ... drops a file symlink" | delete | dw/security.py:162-168 (`contained`) "For a listing that walks a directory with os.walk: a file symlink ... Both sides are resolved, so a root that is itself a link ... still contains its own files". | +| 8 | dw/server/CLAUDE.md:32 | "and `zip_download` skips links" | map | Archives never follow links / `dw/server/outputs.py` (`zip_download`), `dw/security.py` / listings drop and archives skip an out-of-root symlink / `tests/test_security_symlinks.py::TestOutputs::test_the_archive_route_does_not_follow_the_link`, `::TestOutputs::test_the_gallery_listing_does_not_enumerate_the_link`. Only a comment today (outputs.py:349-351). | +| 9 | dw/server/CLAUDE.md:34 | "See docs/SERVER.md." | keep | Pointer; the file shrinks to it plus the seam-map pointer. | + +--- + +## dw_mcp/CLAUDE.md (1-95) + +| # | file:line | first words | verdict | evidence | +| --- | --- | --- | --- | --- | +| 1 | dw_mcp/CLAUDE.md:1-3 | "# dw_mcp / Guidance for the `dw_mcp/` stdio MCP server package." | keep | Header. | +| 2 | dw_mcp/CLAUDE.md:5-8 | "`dw_mcp/` is a stdio MCP server (`dw-mcp`, `python -m dw_mcp`) that wraps the `dw.serve` REST API" | delete | dw_mcp/__init__.py:3-5 "A stdio MCP server that is an HTTP client of a running `dw.serve` ... every tool is a call against the REST API"; docs/MCP.md:3-10. | +| 3 | dw_mcp/CLAUDE.md:9-11 | "It covers the REST surface except the SSE event stream" | docstring | `dw_mcp/__init__.py` (6 lines). The exclusions (SSE, the two bulk zips, the SPA mount) are not in docs/MCP.md or any docstring. | +| 4 | dw_mcp/CLAUDE.md:11-18 | "`POST /api/uploads` *is* covered, by `upload_asset`" (+ `content` base64, 4MB vs 200MB) | delete | dw_mcp/assets.py:1-10 "`upload_asset` is that path ... what comes back is the reference, not a path"; assets.py:20, 26 (`MAX_UPLOAD_BYTES`, `MAX_INLINE_UPLOAD_BYTES`); docs/MCP.md:488-495 "use `upload_asset(content=..., asset_name=...)` ... capped at 4MB (#203)". | +| 5 | dw_mcp/CLAUDE.md:19-23 | "`get_output_audio` is `get_output_image`'s sibling for sound" | delete | dw_mcp/media.py:151-162 (`get_output_audio`) "a whole clip over budget is refused rather than truncated (#204). The way to hear part of a long track is to *ask* for the part"; docs/MCP.md:482-487. | +| 6 | dw_mcp/CLAUDE.md:23-25 | "Video has no MCP content type, so `get_output_frames` returns frames" | delete | docs/MCP.md:481-485 "there is no video content type over MCP, so a video is seen as frames and heard as its track". | +| 7 | dw_mcp/CLAUDE.md:25-31 | "A session works in one of the server's workspaces: `--workspace` / `DW_MCP_WORKSPACE`" | delete | docs/MCP.md:35 "A *name* on the server, not a directory here - `DW_WORKSPACE` means something else"; dw_mcp/workspaces.py:1-9; dw_mcp/client.py:323-329 (`_scoped`) "One place rather than a parameter on every handler ... the default workspace sends nothing". | +| 8 | dw_mcp/CLAUDE.md:32-36 | "`list_jobs` is bounded (newest 20) rather than complete" | delete | dw_mcp/catalog.py:181-191 (`list_jobs`) "Bounded because the unbounded answer was a dead tool ... `total` is what matched before the cut". | +| 9 | dw_mcp/CLAUDE.md:37-44 | "`list_prompts` had the same disease and the same cure" | delete | dw_mcp/prompts.py:25-31 (`list_prompts`) "The bodies are left out by default ... 44 prompts and 87 KB, past a client's result cap"; docs/MCP.md:293 (`tag`, `intended_model`, `text_chars`). | +| 10 | dw_mcp/CLAUDE.md:45-50 | "`run_workflow`, `validate_workflow` and the gallery and output tools ... also take an optional per-call `workspace`" | delete | dw_mcp/client.py:331-335 (`_scoped`) "`workspace` is the per-call pin - 'for this one call, without switching the session' - and wins over the session's own". | +| 11 | dw_mcp/CLAUDE.md:51-55 | "`get_server_info` (`/api/server`) is the capability call" (+ "an HTTP client of a *running* `dw.serve`") | delete | docs/MCP.md:227 "`device` (the accelerator a run will use), `version`, the `workspace` ..."; docs/MCP.md:8-9 "an HTTP client of a **running** `dw.serve`. It owns no job state and no GPU worker". | +| 12 | dw_mcp/CLAUDE.md:55-60 | "`guides.py` proxies `GET /api/guides`" | delete | dw_mcp/guides.py:1-14 "These proxy `GET /api/guides` ... the guides an agent reads have to describe the engine that will run what it authors ... The `GUIDES` table ... now live in `dw/server/guides.py`". | +| 13 | dw_mcp/CLAUDE.md:60-63 | "Only `dw_mcp/server.py` and the `tools_*.py` modules ... import the MCP SDK" | delete | dw_mcp/server.py:3-7 "With the `tools_*` modules, the only code that imports the MCP SDK. The tools are methods of the classes there, each body a one-line call into a handler, so the handlers stay testable without a session". | +| 14 | dw_mcp/CLAUDE.md:63-66 | "It is a top-level package rather than `dw.mcp` on purpose" | map | `dw_mcp` stays torch-free / `dw_mcp/` / imports no `dw.*` module / `tests/test_mcp_server.py::TestStartupWeight::test_the_server_starts_without_importing_the_engine` (:1073). Test-only. | +| 15 | dw_mcp/CLAUDE.md:66-73 | "Seven tools require `acknowledged_cost=True`" (+ bound acknowledgement, 409 rendering) | delete | docs/MCP.md:335 "Seven tools refuse unless `acknowledged_cost=true` is passed" and the table; :358 bound `{"fingerprint": ...}`; dw_mcp/diagnose.py:44-49 (`_acknowledgement_body`) "a bound one is the dict itself, verbatim"; dw_mcp/client.py:406-411 (`_format_detail`). | +| 16 | dw_mcp/CLAUDE.md:73-80 | "The three job-queuing tools return as soon as the job is queued" (+ `wait_seconds`, `MAX_WAIT_SECONDS` 55 / env) | delete | dw_mcp/diagnose.py:3-8 "submitting returns immediately and progress is polled"; :21-27 (the 55s cap, `DW_MCP_MAX_WAIT_SECONDS`); `run_workflow` docstring :89-98 "`wait_seconds` folds the first `wait_for_job` into this call ... same clamp to MAX_WAIT_SECONDS". | +| 17 | dw_mcp/CLAUDE.md:77-78 (clause) | "interpolated into the tool descriptions, so no doc or skill quotes a number" | keep | An edit-time rule for anyone writing a tool description, doc or skill. No test checks for a quoted 55. | +| 18 | dw_mcp/CLAUDE.md:80-83 | "`delete_output(job_id=...)` is the same economy for cleanup" | delete | dw_mcp/media.py:459-462 (`delete_output`) "`job_id` is the other handle on a whole run: the job record carries the `/` its run wrote (`run_dir`...)". | +| 19 | dw_mcp/CLAUDE.md:83-85 | "Authoring has two halves: `get_schema` describes a workflow and `get_prompt_schema` a stored prompt" | delete | docs/MCP.md:295 (`get_prompt_schema` row) and the `get_schema` row in the Authoring table (MCP.md:244-282). | +| 20 | dw_mcp/CLAUDE.md:85 | "See docs/MCP.md." | keep | Pointer. | +| 21 | dw_mcp/CLAUDE.md:87-93 | "Claude Code shows an MCP server's `instructions` and each tool description only up to 2,048 characters" (+ `CLIENT_TEXT_LIMIT`, `SURFACE_BUDGET`) | map | MCP surface text budget / `dw_mcp/server.py`, `dw_mcp/tools_*.py` / <= 2,048 characters each, total within `SURFACE_BUDGET` / `tests/test_mcp_server.py::test_no_text_the_agent_reads_is_cut_off_by_the_client`, `::test_the_tool_surface_fits_the_budget`. The test comment at :1342-1344 states it; test-only. | +| 22 | dw_mcp/CLAUDE.md:93-95 | "Put what an agent must act on first, and point at a guide section rather than restating it" | keep | An authoring rule for MCP surface text. The length tests can't check ordering or restating, and the instructions and `list_workflows` sit "within a few dozen characters of the limit". 2-3 lines. | + +--- + +## ui/CLAUDE.md (1-154) + +UI files have no docstrings. Where the evidence is a file-level or block comment in `ui/src`, the cell says so (concern 1). + +| # | file:line | first words | verdict | evidence | +| --- | --- | --- | --- | --- | +| 1 | ui/CLAUDE.md:1-3 | "# ui / Guidance for the web UI single-page app." | keep | Header. | +| 2 | ui/CLAUDE.md:5-6 | "`npm run build` outputs `ui/dist`, which the server serves" | delete | ui/README.md:5-12 "## Build (what `python -m dw.serve` serves at `/`) ... npm run build # -> ui/dist, auto-detected by the server". | +| 3 | ui/CLAUDE.md:8-10 | "Front-end checks, run from `ui/`: `npm run check`, `npm run lint`, `npm test`, and `npx playwright test`" | keep | ui/README.md has only `npm run check`; lint, test, playwright and the port clash with a running `dw.serve` are said nowhere else. UI work needs them before touching a file. | +| 4 | ui/CLAUDE.md:12-23 | "Every request is scoped to the selected workspace in one place: `scoped()` in `lib/api.ts`" | delete | ui/src/lib/api.ts:45-50 (`scoped`) "Append the workspace selector ... 'default' sends nothing ... lets this live in one place instead of being threaded through every call site"; workspace.svelte.ts:8-17 "The route is the source of truth: `applyRouteWorkspace` ... a page only has to read `current` inside its load effect"; routes.ts:1-4 (legacy hashes via `legacyRedirect`). The JobPage URL correction and `outputUrl(path, version, workspace)` are visible in the code. | +| 5 | ui/CLAUDE.md:25-28 | "## Design system / The look is "contact sheet", and it rests on two rules that `src/app.css` encodes." | delete | ui/src/app.css:1-24 header comment "Contact sheet." with both rules (comment, concern 2). The kept file points at it in one line. | +| 6 | ui/CLAUDE.md:30-38 | "**Colour means machine state.** The greys are deliberately achromatic" | delete | ui/src/app.css:3-13 "Greys here are equal-RGB rather than the usual blue-tinted slate ... Colour therefore means one thing: machine state. `--live` (a darkroom safelight amber) ... never decoration. Interactive elements are ink"; :21-23 "`--accent` is kept as the name for 'interactive ink'". | +| 7 | ui/CLAUDE.md:38-39 | "Every token pair passes WCAG AA in both themes; keep it that way." | keep | Not in app.css, and no contrast check. | +| 8 | ui/CLAUDE.md:41-47 | "**Mono is what the engine reads.** `--font-mono` for anything the engine resolves literally" | delete | ui/src/app.css:15-19 "anything the ENGINE reads literally ... is set in --font-mono; anything written for a PERSON is set in --font-sans. That is why headings are mono"; :310-311 "sentence case and tight, never tracked out or capitalised into a label"; :332 (tabular figures). | +| 9 | ui/CLAUDE.md:49-52 | "Type scale is `--t-xs` .. `--t-xl` off a 15px base; radius says what a thing is" | delete | ui/src/app.css:103 "~1.25 from a 15px base"; :127-129 "Radius says what kind of thing something is rather than being one global value"; :202-203 "One filled button per view: the thing you came to do. Everything else is .quiet (outlined) or .bare (text)." | +| 10 | ui/CLAUDE.md:54 | "**Show the proof.** A list of things that produce images shows the images." | keep | A design principle for new UI. Not stated in app.css or any shared file. 1 line. | +| 11 | ui/CLAUDE.md:55-62 | "The workflows catalog and the workflow page both read `api.gallery()` and match a workflow's name" | delete | ui/src/lib/proofs.ts:3-9 (`latestProofs`) "the gallery listing reports that identity as an entry's `folder` with the run id already stripped, so a workflow's name matches its folder directly"; WorkflowsPage.svelte:262-263 "A workflow that has never run gets no frame at all rather than a grey placeholder"; :132 "Within a folder, workflows that have produced something come first". | +| 12 | ui/CLAUDE.md:64-70 | "Stripping the run id is also what makes four runs of one workflow four identical captions" (+ `v4` chip) | delete | ui/src/lib/types.ts:276-280 (`version`) "what the grid shows as `v4`. Two runs write the same `label` ... never renumbered"; JobsPage.svelte:134-138 (`run_version` chip). | +| 13 | ui/CLAUDE.md:70-71 | "The UI reads the field only; nothing here computes or orders a version." | keep | An edit-time rule for UI work, and the general one: the UI reads engine-derived fields (version, subfolder, plan) and never recomputes them. Not stated in ui/src. 1 line. | +| 14 | ui/CLAUDE.md:73-79 | "Every picture in the app sits in the global `.frame` (app.css)" | delete | ui/src/app.css:410-416 "A contact-sheet frame. Every place the app shows something a workflow produced ... uses this ... Set the size on the frame; the media inside always fills it." (The `.cardframe` exception is visible in the code.) | +| 15 | ui/CLAUDE.md:81-86 | "Where `--live` appears, and nowhere else:" | keep | The whitelist of `--live` sites exists only here; app.css:9-11 gives the categories, not the list, and no check enforces it. | +| 16 | ui/CLAUDE.md:88-89 | "The sidebar's selected entry is a heavier ink left edge (`aria-current="page"`), never a colour" | delete | ui/src/app.css:467-468 "Ticked is the user's own state, not the machine's, so it reads as a heavier ink edge rather than taking the signal colour" (the general rule); Sidebar.svelte:161, 353 implement it; headings/names mono per app.css:15-19. | +| 17 | ui/CLAUDE.md:91-103 | "## Assets / `AssetsPage.svelte` is the input side of the gallery (#165)" | delete | ui/src/lib/pages/AssetsPage.svelte:2-7 "The input side of the gallery (#165) ... The UX is the gallery's on purpose"; :53-55 "the page never builds a path"; :137-141 "a read-only library's tile carries no checkbox and would answer 403". (Comment.) docs/WORKSPACES.md:280-281 (origins). | +| 18 | ui/CLAUDE.md:105-113 | ""The same UX" is literal: the detail popout is `position: sticky; bottom: 1rem`" (+ bulk actions, Escape, archive) | delete | AssetsPage.svelte:630-632 "The gallery's popout: it rides the bottom of the viewport ... off screen for any click above the fold"; :257-259 "Escape closes the thing on top ... it clears first and the detail panel on a second press"; picks.svelte.ts:3-9 (shift-click range, select all). `POST /api/assets/archive` is in docs/SERVER.md. (Comments.) | +| 19 | ui/CLAUDE.md:115-121 | "The selection itself is not this page's: `picks.svelte.ts` holds it" (+ keep it shared rather than copy) | delete | ui/src/lib/picks.svelte.ts:5-9 "The gallery and the assets page run the same selection ... It lived twice ... the two copies had already drifted ... so it lives here instead"; app.css:440-443 "The selection logic is picks.svelte.ts and the bar is BulkBar.svelte; these are the two rules that have to reach a tile ... so they live here rather than twice". | +| 20 | ui/CLAUDE.md:121-123 | "`dialogOpen()` is the one Escape guard" | delete | ui/src/lib/picks.svelte.ts:113-118 (`dialogOpen`) "a page must not take its own selection or detail away underneath it. `alertdialog` is in the list because that is what `ConfirmDialog` actually renders." | +| 21 | ui/CLAUDE.md:123-126 | "A bulk action must never touch what the user cannot see, so `Picks.size` counts only the visible selection" | delete | ui/src/lib/picks.svelte.ts:30-31 (`size`) "a tick the filter is hiding is inert, not counted, until the filter brings it back into view"; AssetsPage.svelte:140-141 "a bulk action touching what the user cannot see". | +| 22 | ui/CLAUDE.md:128-135 | "The search path is the page's top level and folders sit inside it" | delete | AssetsPage.svelte:112-118 "One section per library on the search path, in the order the server resolves them ... An empty library still shows its header, so an empty workspace says where an upload would land"; :68-69, :108-109 (collapse persisted, a filter opens everything). (Comments.) | +| 23 | ui/CLAUDE.md:137-146 | "Upload is the section's own button rather than a destination pick" | delete | AssetsPage.svelte:62-65 "The one thing about an upload that cannot be changed afterwards, so it is the destination's own button"; :137-141 (no checkbox on read-only, select-all spans open writable sections); :171-182 (a shared delete "go[es] away for every workspace"). (Comments.) | +| 24 | ui/CLAUDE.md:148-152 | "`shadowed` is rendered rather than only described." | delete | AssetsPage.svelte:404-408 "What this library holds under a name a nearer one has taken ... 'I uploaded it and asset: still loads the old one' is otherwise unanswerable ... a tile is a dimmed label rather than a picture". (Comment.) | +| 25 | ui/CLAUDE.md:154 | "See docs/SERVER.md." | keep | Pointer. Add one for the `src/app.css` header. | + +--- + +## .github/copilot-instructions.md (1-115) + +Per the ruling it becomes a pointer of 10 lines or fewer. These rows only confirm nothing in it is uniquely true and +needed elsewhere. + +| # | file:line | first words | verdict | evidence | +| --- | --- | --- | --- | --- | +| 1 | copilot:1-3 | "# AI Coding Instructions ... This is a declarative workflow engine" | delete (redundant) | CLAUDE.md:7 overview. | +| 2 | copilot:5-8, 10-11 | "## Worker Architecture / `dw.serve` uses a **persistent worker subprocess**" (+ cleanup, full cleanup on change / `clear`) | delete (redundant) | docs/WORKER_GUIDE.md:5-8, 19-23 ("frees the old workflow's models"), 27-29 ("garbage collection + GPU cache clearing"). | +| 3 | copilot:9 | "Automatic workflow file change detection (SHA256 hash)" | delete (stale) | docs/WORKER_GUIDE.md:19 "A job runs the definition the server checked when it was submitted ... Pipelines are cached by what they load"; no hashing in dw/worker.py. | +| 4 | copilot:12 | "5-minute execution timeout with graceful shutdown" | delete (stale) | docs/WORKER_GUIDE.md:40 "There is no execution timeout". | +| 5 | copilot:13 | "Memory monitoring with growth warnings (>500MB)" | delete (redundant; **not stale**) | True: dw/worker.py:86 `MEMORY_GROWTH_THRESHOLD_MB = 500`, used at :604. Not needed elsewhere; docs/WORKER_GUIDE.md:27-29 covers memory. The plan's "stale" label for it is wrong. | +| 6 | copilot:15-20 | "**Key modules:** `dw/worker.py` ... **Worker commands:** execute, cancel, ..." | delete (redundant) | docs/WORKER_GUIDE.md:5-16; dw/worker_protocol.py:1-6 "one frozen dataclass per command and reply ... A request's reply echoes its request_id". | +| 7 | copilot:22-29 | "## Architecture Overview / **Core Components:** `dw/workflow.py`: Main orchestrator" | map | Core engine owners / `dw/workflow.py`, `dw/workflow_run.py`, `dw/validation.py`, `dw/step.py`, `dw/pipeline_processors/{pipeline,placement,components,adapters,progress}.py`, `dw/tasks/task.py`, `dw/previous_results.py`. One seam-map row each (most are already in the plan's owner list). | +| 8 | copilot:31-35 | "**Key Data Flow:** 1. JSON workflow loaded → validated" (+ `{workflow_id}-{step_name}.{index}`) | delete (redundant) | docs/WORKFLOW_GUIDE.md:173-240 (Cross-Step Data Flow); :1095 "`base_name` is `{workflow_id}-{step_name}.{step_index}`"; schema-before-substitution at :274-275. | +| 9 | copilot:37-45 | "## Critical Patterns / **Variable System** ... **Prompt References**" | delete (redundant) | docs/WORKFLOW_GUIDE.md:261-330 (References: `variable:`, `previous_result:`, `constant:`, `prompt:`). | +| 10 | copilot:47-56 | "**Pipeline Configuration:** ```json" | delete (redundant) | docs/WORKFLOW_GUIDE.md:62-84 (Pipeline Steps). | +| 11 | copilot:58-66 | "**Task Execution:** ```json" | delete (redundant) | docs/WORKFLOW_GUIDE.md:109-140 (Task Steps). | +| 12 | copilot:68-72 | "## Development Workflows / **Testing:** ... **Validation:** ... **Execution:**" | delete (redundant) | CLAUDE.md:9-33 Common Commands; docs/TESTING.md. | +| 13 | copilot:74 | "**Adding New Tasks:** Register a handler function in `dw/tasks/task.py` with the `@register_command("name")` decorator" | map | Adding a task / `dw/tasks/task.py` (`register_command`) / a task is a registered command whose implementation's signature is its argument schema / `tests/test_task_discovery.py::TestDescribeTask::test_signature_becomes_the_schema`. True and not stated in a doc beyond docs/TASKS.md:1146 (which covers `assessment=True` only). | +| 14 | copilot:75 | "**Adding Pipeline Types:** Update `workflow_schema.json`" | delete (stale) | dw/workflow_schema.json:584-587 `component_type` is a free-form string ("'module.typename' module defaults to diffusers"), so no schema change is needed; docs/QUANTIZATION.md:211 (dynamic import). | +| 15 | copilot:77-80 | "All file paths in workflows are relative ... Built-in workflows use `"builtin:filename.json"`" | delete (redundant) | docs/WORKFLOW_GUIDE.md:159, 251-252. | +| 16 | copilot:81 | "Model offloading patterns: `"sequential"`, `"model"`" | delete (redundant) | docs/WORKFLOW_GUIDE.md:1115 (Memory Offloading). | +| 17 | copilot:82 | "Quantization configs follow BitsAndBytesConfig pattern" | delete (redundant) | docs/QUANTIZATION.md:3, 41. | +| 18 | copilot:83 | "Image results default to JPEG, videos to MP4 unless overridden" | delete (stale) | dw/result.py:306-307: no `content_type` means the result is not saved ("Skipping save - disabled or no content type specified"). There is no default. | +| 19 | copilot:85-90 | "## Key Integration Points / **HuggingFace Diffusers:** Direct pipeline instantiation" | delete (redundant) | docs/WORKFLOW_GUIDE.md:62-84, 1347-1384 (LoRAs, IP-Adapter); docs/WORKFLOW_GUIDE.md:1465 (device detection). | +| 20 | copilot:92-99 | "## Security / **Critical security modules**: `dw/security.py`" | delete (redundant) | CLAUDE.md:251-259 (kept); docs/SECURITY.md. | +| 21 | copilot:101-106 | "All entry points (run.py, validate.py, serve.py) use security validation. When adding features:" | delete (redundant) | CLAUDE.md:253-258 (kept). | +| 22 | copilot:108-112 | "## Common Gotchas / Schema validation happens before variable substitution" (+ cartesian, component keys) | delete (redundant) | docs/WORKFLOW_GUIDE.md:274-275 (schema before substitution), :487-500 (several `previous_result` references multiply), :1384 (Sharing Components). | +| 23 | copilot:113 | "Built-in workflows inherit parent variable scope but need explicit argument mapping" | delete (redundant) | docs/WORKFLOW_GUIDE.md:141-172 (Workflow Steps, `arguments` mapping). Also the root's Critical Gotchas (other half's table). | +| 24 | copilot:114-115 | "Variable names must match `^[a-zA-Z_][a-zA-Z0-9_-]*$` ... File paths with `../` are blocked" | delete (redundant) | CLAUDE.md:255, 259 (kept). | diff --git a/docs/stabilization/phase-4.md b/docs/stabilization/phase-4.md index 73e5e110..d0580018 100644 --- a/docs/stabilization/phase-4.md +++ b/docs/stabilization/phase-4.md @@ -377,7 +377,7 @@ Work on branch `stabilization/phase-4c` in the worktree, from `develop` at `edb6 - `ui/CLAUDE.md` 154 (design system, assets); - `dw_mcp/CLAUDE.md` 95; - `dw/server/CLAUDE.md` 34. -- `.github/copilot-instructions.md`, 114 lines. It is a parallel description of the architecture, partly stale (it still describes a 5-minute execution timeout and memory warnings), and the metric does not count it. +- `.github/copilot-instructions.md`, 114 lines. It is a parallel description of the architecture, partly stale (it still describes a 5-minute execution timeout and SHA-256 change detection), and the metric does not count it. - What reads a CLAUDE.md mechanically: - `tests/test_plugin_skills.py::test_a_skill_is_enumerated_where_the_plugin_describes_itself`, which needs every plugin skill's name in backticks in the root file; - `arch_metrics.py` and `arch_report.py`, for counting. From dcfc93324fed25d6f99e6abeb7cf644b28623c1c Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 14:09:35 -0500 Subject: [PATCH 23/41] docs(stabilization): 4c hot zone adds the docstring destinations Co-Authored-By: Claude Opus 5.5 --- docs/stabilization/hot-zone.txt | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/docs/stabilization/hot-zone.txt b/docs/stabilization/hot-zone.txt index 3d04f63e..bd58d439 100644 --- a/docs/stabilization/hot-zone.txt +++ b/docs/stabilization/hot-zone.txt @@ -15,3 +15,17 @@ docs/AGENT_LOOP.md tests/test_architecture_map.py scripts/arch_metrics.py docs/stabilization/ +# 4c Task 1: modules receiving moved rules in their docstrings +dw/subfolders.py +dw/reference_names.py +dw/pipeline_processors/placement.py +dw/previous_results.py +dw/runs.py +dw/step_cache.py +dw/variable_constraints.py +dw/shots.py +dw/__init__.py +dw/settings.py +dw/server/job_history.py +dw/realize.py +dw_mcp/__init__.py From 0a1876cbcc79d6eea23471488e57bb2ea100200b Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 14:09:35 -0500 Subject: [PATCH 24/41] docs(stabilization): 4c hot zone adds the docstring destinations Co-Authored-By: Claude Opus 5.5 --- docs/stabilization/hot-zone.txt | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/docs/stabilization/hot-zone.txt b/docs/stabilization/hot-zone.txt index 3d04f63e..bd58d439 100644 --- a/docs/stabilization/hot-zone.txt +++ b/docs/stabilization/hot-zone.txt @@ -15,3 +15,17 @@ docs/AGENT_LOOP.md tests/test_architecture_map.py scripts/arch_metrics.py docs/stabilization/ +# 4c Task 1: modules receiving moved rules in their docstrings +dw/subfolders.py +dw/reference_names.py +dw/pipeline_processors/placement.py +dw/previous_results.py +dw/runs.py +dw/step_cache.py +dw/variable_constraints.py +dw/shots.py +dw/__init__.py +dw/settings.py +dw/server/job_history.py +dw/realize.py +dw_mcp/__init__.py From 5f4e6d3992119c741ac523806bb82477d5bd0331 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 14:19:55 -0500 Subject: [PATCH 25/41] docs: the seam map, docs/ARCHITECTURE.md, and its path test One row per concept: owner module, the rule in one sentence (read off the code), and the test or check that enforces it. Rows are the triage's 26 map rows plus the stabilization's owners and dw/variables.py. tests/test_architecture_map.py checks that every backticked repo path in the map exists (a glob must match) and that every tests/x.py::Class::test it names is defined in that file. Co-Authored-By: Claude Opus 5.5 --- docs/ARCHITECTURE.md | 134 +++++++++++++++++++++++++++++++++ tests/test_architecture_map.py | 94 +++++++++++++++++++++++ 2 files changed, 228 insertions(+) create mode 100644 docs/ARCHITECTURE.md create mode 100644 tests/test_architecture_map.py diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md new file mode 100644 index 00000000..ceb380df --- /dev/null +++ b/docs/ARCHITECTURE.md @@ -0,0 +1,134 @@ +# Architecture: the seam map + +Where each concept lives. One row per concept: the module that owns it, the rule that holds +across the seam in one sentence, and what enforces the rule. The detail is in the owning +module's docstring, so read that before editing. The map only says which one to open. + +- **Owner**: repo-relative paths. A function name follows a colon (`dw/runs.py`: `run_versions`). +- **Enforced by**: a test (a file, or `file::Class::test_name`), a ratchet in `scripts/arch_metrics.py` + (its key in `docs/stabilization/baseline.json`), the CodeQL pack, or the validation registry. A + `—` means nothing mechanical holds the rule, so a change there is checked only by review. +- `tests/test_architecture_map.py` checks that every path named here exists and that every test + named here is defined in its file. + +## Engine core + +| Concept | Owner | Rule | Enforced by | +| --- | --- | --- | --- | +| Workflow | `dw/workflow.py`: `Workflow`, `workflow_from_file` | A workflow is loaded and checked here, then run by `Workflow.run`, which hands each phase to `dw/workflow_run.py`. | `tests/test_workflow.py` | +| One execution, in phases | `dw/workflow_run.py` | `Workflow.run` calls `prepare_run`, `open_run`, `begin_steps` and then `run_step` per step, and the cache probe `cache_hits` shares `prepare_definition` and `cache_lookup` with the run, so the probe answers for the run that would happen. | `tests/test_workflow_step_cache.py::TestCacheHits::test_after_a_run_the_probe_names_what_the_next_run_reuses` | +| Step | `dw/step.py`: `Step`, `dw/workflow.py`: `Workflow.create_step_action` | A step is exactly one of a pipeline, a task or a sub-workflow (the schema's `oneOf`), `create_step_action` builds its action, and `dw/workflow_run.py` runs it. | `tests/test_step.py` | +| Pipeline loading | `dw/pipeline_processors/pipeline.py`: `Pipeline`, `dw/pipeline_processors/components.py`: `load_component`, `configure_components` | Components load from `from_pretrained_arguments`, and a quantization config is built in `dw/pipeline_processors/config_objects.py` from a `config_type` that the `_type` key convention loads by name, so a new backend needs no code. | `tests/test_config_objects.py` | +| Placement and offload | `dw/pipeline_processors/placement.py`: `place_component` | Every device a workflow names passes through `resolve_device()` (`dw/__init__.py`) before placement reads its backend, so the MPS accommodations also fire for a translated device. | `tests/test_device_portability.py::TestPlacement::test_a_translated_device_gets_its_backends_offload_downgrade` | +| On-demand residency | `dw/pipeline_processors/placement.py`: `apply_on_demand_placement` | The wrappers keep the wrapped signature (`functools.wraps`; H3's denoiser reads `signature(transformer.forward)`), they cannot be combined with `group_offload` on one component, and either one stops the wholesale `pipeline.to(device)`. | `tests/test_modular_pipeline.py::TestOnDemandResidency::test_the_wrapped_signature_survives`, `tests/test_modular_pipeline.py::TestOnDemandResidency::test_group_offload_and_on_demand_together_are_rejected` | +| LoRAs | `dw/pipeline_processors/adapters.py`: `active_loras`, `load_loras` | A `loras` entry whose `model_name` is null is switched off, because a template's `loras` list is fixed JSON and a variable can null a value but cannot remove an entry. | `tests/test_lora_disable.py::TestLoadLoras::test_an_all_null_entry_loads_nothing` | +| Step progress | `dw/pipeline_processors/progress.py`: `reported_progress_bars` | A pipeline with no step callback (every `ModularPipeline`) reports its denoise steps through its progress bars instead. | `tests/test_modular_progress.py` | +| Adding a task | `dw/tasks/task.py`: `register_command` | A task is a function registered with `@register_command`, and the signature of its implementation is its argument schema. | `tests/test_task_discovery.py::TestDescribeTask::test_signature_becomes_the_schema` | +| Task argument domains | `dw/task_domains.py` | A numeric domain that a signature cannot express is declared in its table, which is checked at validation and again at run time (`check_arguments`). | `tests/test_task_domains.py` | +| Variables | `dw/variables.py`: `set_variables`, `replace_variables`, `resolve_variable_values`, `argument_errors` | `argument_errors` folds a caller's arguments in exactly as `set_variables` does at the start of a run, so a bad name or value is a 400 before anything is queued. | `tests/test_variables.py::test_argument_errors_reports_a_dict_passed_for_a_string_variable`, `tests/test_variables.py::TestResolveVariableValues::test_an_undeclared_name_is_the_usual_error` | +| List variables from the CLI | `dw/run.py`, `dw/variables.py`: `get_value` | A `name=value` string given for a list variable is split on commas, so a list of objects (such as a template's `shots`) can only be passed as JSON over the API or MCP. | `tests/test_variables.py::test_set_variables_list_default_splits_on_comma` | +| `for_each` expansion | `dw/for_each.py`: `expand_for_each` | Expansion runs after substitution and before the reference check, and it names each member `@`, so the step loop, the cache and the manifest only ever see ordinary steps. | `tests/test_for_each.py::TestNaming::test_member_name_joins_with_at` | +| Previous results | `dw/previous_results.py`: `get_iterations`, `previous_result_reference_errors` | References multiply into a cartesian product, and a literal reference to a step that is not earlier is a validation error at its JSON path. | `tests/test_previous_results.py::TestGetIterations::test_multiple_references_create_cartesian_product`, `tests/test_previous_results.py::TestStaticReferenceChecking::test_a_step_cannot_reference_itself_or_a_later_one` | +| Type conversion | `dw/arguments.py`: `realize_args`, `dw/type_helpers.py` | A key ending `_type` or `_dtype`, or named `dtype`, loads a Python object at realize time, and a `{}`-wrapped value stays a string (see `docs/WORKFLOW_GUIDE.md`, "Types and escaping"). | `tests/test_type_helpers.py::TestGetType::test_get_type_from_diffusers`, `tests/test_arguments.py::TestRealizeArgs::test_realize_escaped_type_reference` | + +## References and libraries + +| Concept | Owner | Rule | Enforced by | +| --- | --- | --- | --- | +| Reference prefixes | `dw/references.py` | Every prefix (`variable:`, `asset:`, `output:` and the rest) is spelled only here, and every module that tests for, strips or builds one calls this module. | `scripts/arch_metrics.py` ratchets `prefix_literals` and `prefix_handling`; `tests/test_references.py::test_references_imports_nothing_from_dw` | +| Explicit references resolve first | `dw/arguments.py`: `_realize_explicit_reference`, `dw/assets.py`, `dw/runs.py`: `resolve_output_reference`, `dw/prompts.py` | `asset:`, `output:`, `constant:` and `prompt:` resolve in `realize_args` before any key-name convention, and an `asset:` or `output:` path is confined to its library or output root by a `dw/security.py` validator. | `tests/test_assets.py::TestReferences::test_a_symlink_out_of_the_library_is_refused`, `tests/test_security_symlinks.py::TestOutputs::test_an_output_reference_does_not_follow_a_linked_run_directory` | +| Reference name shape | `dw/reference_names.py`: `reference_name_errors` | The shape of every `asset:`, `prompt:` and `output:` name is checked before the queue, but whether the file exists is checked against a workspace later. | `tests/test_reference_names.py` | +| Library search paths | `dw/library.py`: `LibraryPath`, `library_path` | Reads go through the roots front to back, so an earlier name shadows a later one, and writes go only to the front root, so saving something opened from a read-only root writes a copy. | `tests/test_library_path.py`, `tests/test_server_library_path.py::TestOneListingEnvelope::test_deleting_a_read_only_entry_gives_one_message` | +| Packaged builtins and sub-workflows | `dw/library.py`: `builtin_root`, `resolve_sub_workflow` | `builtin:` names the packaged `dw/workflows/` (not the top-level `workflows/` examples), and a sub-workflow path resolves beside its parent first, then by catalog name along the search path, and is confined to the root it resolves in. | `tests/test_sub_workflow_resolver.py` | +| Workspace | `dw/workspace.py`: `resolve_workspace`, `set_workspace` | The order is `--workspace`, then `DW_WORKSPACE`, then the `workspace` setting, then a working directory that looks like a workspace, then `~/diffusers-workspace`, and resolving creates nothing. | `tests/test_workspace.py::TestResolution::test_a_flag_wins_over_everything`, `tests/test_workspace.py::TestResolution::test_a_bare_working_directory_falls_back_to_the_home_workspace` | +| Realized workflow | `dw/realize.py`: `realize_workflow`, `dw/runs.py`: `write_realized_workflow` | Each run directory holds `workflow.json`, which pins every input that can change between runs: arguments, seed, stored prompt text and `output:.../latest/...`. | `tests/test_realize.py::TestVariablesAndSeed::test_arguments_become_the_variable_defaults` | + +## Runs and outputs + +| Concept | Owner | Rule | Enforced by | +| --- | --- | --- | --- | +| Run directories | `dw/runs.py`: `open_run`, `workflow_identity`, `new_run_id` | Each execution writes into `///` with a `manifest.json`, unless the flat layout (`output_layout`) is chosen. | `tests/test_runs.py` | +| Run versions | `dw/runs.py`: `open_run`, `run_versions`, `record_run_versions` | A run takes its number once, when it opens, as one more than the highest number recorded by any sibling, so deleting a middle run leaves a gap rather than renumbering. | `tests/test_runs.py::TestRunVersions::test_a_deleted_middle_run_leaves_a_gap_rather_than_renumbering`, `tests/test_runs.py::TestRunVersions::test_runs_started_in_the_same_second_still_number_upward` | +| Run-version surfaces | `dw/runs.py` (owner), `dw/server/routes/gallery.py`, `dw/server/outputs.py`, `dw_mcp/tools_catalog.py`: `list_gallery`, `ui/src/lib/pages/GalleryPage.svelte` | The number is computed only in `dw/runs.py`, and everywhere else it is read back, as a gallery field, as `output:/v/`, or as a job's `run_version`. | `tests/test_runs.py::TestRunVersions::test_an_output_reference_can_name_a_run_by_its_version` | +| Result subfolders | `dw/subfolders.py`: `subfolder_errors`, `dw/security.py`: `SUBFOLDER_PATTERN` | A step's `result.subfolder` is checked for shape after expansion and at run time, and it is confined to the run directory by `validate_output_path`. | `tests/test_subfolders.py` | +| Template subfolder roles | `workflows/templates/`, `dw/workflows/` | Every saving step of a template is marked `final` or `intermediate` with at least one `final`, and packaged builtins stay unmarked because assigning a role is the parent's job. | `tests/test_template_subfolders.py::test_every_saving_step_of_a_template_names_its_role`, `tests/test_template_subfolders.py::test_the_packaged_builtins_stay_unmarked` | +| Shots in a joined video | `dw/shots.py` | Every `AudioVideo` constructor carries, rescales, re-measures or drops `shots`, and the sample spans are measured from the joined waveform, not derived from frames. | `tests/test_shots.py` | +| Audio level of a deliverable | `dw/audio_qc.py`: `warn_without_headroom`, `warn_if_written_above_full_scale` | Each check warns and changes nothing, and a muxed video keeps the pre-encode headroom prediction to fall back on only when the post-write probe cannot measure the file. | `tests/test_result.py` | +| Template audio-level placement | `workflows/templates/minimax/music-video.json`, `workflows/templates/minimax/music.json` | `normalize_audio` (-3 dBFS) acts only on the track that goes into the final mux or file, so the slices that condition the shots are left untouched. | — | + +## Validation, plan and cost + +| Concept | Owner | Rule | Enforced by | +| --- | --- | --- | --- | +| Validation registry | `dw/validation.py`: `ERROR_CHECKS`, `WARNING_CHECKS`, `run_checks` | Adding a check means adding one `Check` to a registry, which runs it in order, and a check that raises becomes one `internal` finding while the rest still run. | `tests/test_validation.py::TestExceptionPolicy::test_a_raising_check_is_one_internal_error_and_the_rest_still_run` | +| Step-value checks | `dw/step_value_checks.py` | The bodies of `fps_errors`, `null_media_errors` and `select_errors` live here, and their only caller is the error registry in `dw/validation.py`. | the validation registry | +| Admission | `dw/server/admission.py`: `admit` | Validate, submit, rerun and enhance load and check a request once, and `JobManager.submit` queues what was admitted without checking it again. | `tests/test_admission.py::test_a_submit_expands_once` | +| Variable constraints | `dw/variable_constraints.py` | A model's rule about a value is declared in the workflow's `variable_constraints`, and it is checked at validation and at run time (`apply_constraints`) and reported in the catalog. | `tests/test_variable_constraints.py` | +| Plan | `dw/plan.py` | The plan comes from the same resolvers the run uses, and a bound `acknowledged_cost` is refused with 409 when the fingerprint or the required downloads changed. | `tests/test_server.py::TestBoundRerun::test_a_rerun_bound_to_a_stale_plan_is_refused` | +| Per-repo gating | `dw/plan.py`: `_collect_sources`, `downloads_required` | Each gated repo, including a `loras` entry's `model_name`, is probed and reported on its own, because access to one gated repo does not grant another. | `tests/test_plan.py::TestDownloadsRequired::test_a_gated_repo_this_token_lacks_access_to_is_blocked` | +| Observed cost | `dw/server/observed_cost.py`, `dw/plan.py`: `_tempered` | `observed` comes only from this box's finished jobs and never writes `cost`, and `plan.estimate` quotes the cold observed median ahead of the curated figure, blending toward that figure below three runs. | `tests/test_observed_cost.py::TestColdIsNotWarm::test_the_two_are_reported_separately_each_with_its_runs`, `tests/test_plan.py::TestLowConfidenceObservedEstimate::test_a_single_run_blends_toward_the_curated_figure` | +| Observed-cost residue | `dw/server/observed_cost.py`: `declared_drivers`, `dw/server/routes/library.py`: `get_workflow` | A cost driver that names no declared variable is dropped, and the raw workflow GET is served verbatim (with no `observed`) because the editor saves what it reads. | `tests/test_observed_cost.py::TestComparability::test_a_driver_naming_no_variable_is_dropped`, `tests/test_observed_cost.py::TestTheCatalogsDriversAreReal::test_every_declared_driver_is_a_variable_of_its_workflow` (the verbatim GET: —) | +| VRAM projection | `dw/vram_estimate.py` | A template's `vram_estimate` is projected for each pipeline step after `for_each` expansion, and only the largest step over the ceiling is reported. | `tests/test_vram_estimate.py` | +| H3 Ref2VA VRAM numbers | `workflows/templates/minimax/` (`vram_estimate`) | Every Ref2VA template declares `base_gb` 16.0, `bytes_per_voxel` 28.71 and `gb_per_reference` 1.0, a classification from field runs rather than a fitted curve. | `tests/test_h3_vram_ceiling.py::test_every_ref2va_minimax_template_declares_gb_per_reference` | +| Inherited VRAM ceiling | `dw/vram_inheritance.py`, `dw/server/deps.py`: `ceiling_index` | A workflow with no `vram_estimate` is matched by pipeline identity against an index of single-identity templates, cached on the listing's mtimes, and is warned, never refused. | `tests/test_vram_inheritance.py::test_every_template_declaring_one_identity_declares_the_same_numbers` | +| H3 adapter partition | `dw/adapter_compatibility.py` | An FL2VA LoRA on a reference (`ref2va`) step is refused, and a LoRA whose file name says neither `ref2v` nor `fl2v` is warned. | `tests/test_h3_adapters.py` | +| IC-LoRA reference scale | `workflows/templates/ltx2/` | The three conditioning templates run at `reference_downscale_factor: 1`, and the two upscalers at 2. | `tests/test_ltx2_ic_loras.py::TestEachTemplateMatchesItsCard::test_the_reference_is_encoded_at_the_output_resolution`, `tests/test_ltx2_ic_loras.py::TestTheGenerativeUpscaleMatchesTheSameCard::test_it_loads_the_upscaler_at_factor_two` | +| IC-LoRA numbers | `workflows/templates/ltx2/` | Every number in the IC-LoRA templates is taken from the vendor card. | `tests/test_ltx2_ic_loras.py::TestEachTemplateMatchesItsCard::test_the_strength_is_the_cards_default`, `tests/test_ltx2_ic_loras.py::TestEachTemplateMatchesItsCard::test_the_defaults_are_the_trained_bucket` | +| IC-LoRA prompt genre | `prompts/ltx2/` (tag `ic-lora`) | An IC-LoRA stored prompt is checked against its trained caption form, not against the 150-220-word T2V paragraph rule. | `tests/test_ltx_prompt_library.py::test_an_ic_lora_prompt_is_in_its_trained_form` | + +## Execution and caching + +| Concept | Owner | Rule | Enforced by | +| --- | --- | --- | --- | +| Step cache | `dw/step_cache.py`, `dw/workflow_run.py`: `cache_lookup` | A seeded rerun with unchanged inputs is served from the cache (`reused: true`, nothing written), a workflow with no `seed` skips the cache, and `rerun(new_seed=true)` is how to get a different result. | `tests/test_workflow_step_cache.py::test_cache_hit_marks_its_manifest_entry_and_event_reused`, `tests/test_workflow_step_cache.py::TestCacheHits::test_an_unseeded_workflow_has_no_hits`, `tests/test_rerun_new_seed.py::test_a_rerun_with_a_new_seed_draws_one_into_that_variable` | +| Worker protocol | `dw/worker_protocol.py` | Every command and reply is a frozen dataclass that travels as a wire dict, and this module imports neither `dw/worker.py` nor `dw/worker_manager.py`. | `tests/test_worker_messages.py::test_from_wire_inverts_to_wire`, `tests/test_worker_messages.py::test_an_unknown_reply_type_is_kept_whole_rather_than_raised` | +| Persistent worker | `dw/worker.py`, `dw/worker_manager.py`, `dw/serve.py` | Jobs run in one spawned worker process that keeps models loaded between runs, so a change to engine code needs a server restart. | `tests/test_worker_manager.py` | +| Failed-run reporting | `dw/worker.py`, `dw/worker_protocol.py`: `Failed`, `Cancelled` | A failed or cancelled run's reply still carries the manifest of the steps that ran. | `tests/test_worker_execute.py::test_failure_carries_the_manifest_of_the_steps_that_ran`, `tests/test_worker_execute.py::test_cancellation_carries_the_manifest_too` | +| Run context and events | `dw/events.py`: `RunContext`, `emit_warning` | A run's context travels through contextvars, and a run-time warning reaches the job through `emit_warning`, not just the log. | `tests/test_events.py` | + +## Media and DSP + +| Concept | Owner | Rule | Enforced by | +| --- | --- | --- | --- | +| Opening media | `dw/media.py` | Every `av.open` in the engine is in this module, which uses PyAV only and imports no step types. | `tests/test_media_layering.py::test_av_open_appears_only_in_media`, `tests/test_media_layering.py::test_media_imports_no_step_types` | +| Signal processing | `dw/dsp.py` | Pure numpy, scipy and pyloudnorm: it measures and transforms a waveform, decides nothing, and imports nothing from `dw`. | `tests/test_media_layering.py::test_dsp_imports_nothing_from_dw` | +| Assessment probes | `dw/tasks/assess.py`, `dw/assessment_rules.py`, `dw/server/assess.py` | A probe measures a finished file and lists `findings` against the rules table, and nothing in the engine acts on a finding. | `tests/test_assessment_rules.py` | + +## Security and architecture guardrails + +| Concept | Owner | Rule | Enforced by | +| --- | --- | --- | --- | +| Path, URL and argument validators | `dw/security.py` | Filesystem access goes through `validate_path` / `validate_workflow_path` / `validate_output_path` (with a base), URLs through `validate_url`, and subprocess arguments through `sanitize_command_args`. | `tests/test_security.py::test_path_validation`, CodeQL (`.github/codeql/dw-security/`) | +| Locations from a workflow | `dw/locations.py` | A media location in a workflow's arguments is checked by one policy (a local path is confined, a URL is filtered for SSRF), because the workflow JSON is untrusted. | `tests/test_security_ssrf.py` | +| Trust gate | `dw/trust.py` | An untrusted workflow (the default) may name only allowlisted, constructible classes and no remote code, and a dotted type it may not use is a validation error before anything is imported. | `tests/test_security_trust_gate.py::TestValidationRefusesBeforeImport::test_a_dotted_type_is_a_validation_error` | +| CodeQL path-injection model | `.github/codeql/dw-security/`, `.github/workflows/codeql.yml`, `dw/security.py` | The local pack models the validators in `dw/security.py` as sanitizers (`validate_path` only when it is given a base), so a validator that moves is re-modelled in the same commit. | the CodeQL query dw/path-injection | +| Archives never follow links | `dw/server/outputs.py`: `zip_download`, `dw/security.py` | A gallery or asset listing drops a symlink that leaves its root, and an archive skips one. | `tests/test_security_symlinks.py::TestOutputs::test_the_archive_route_does_not_follow_the_link`, `tests/test_security_symlinks.py::TestOutputs::test_the_gallery_listing_does_not_enumerate_the_link` | +| Engine/server import direction | `dw/`, `dw/server/`, `dw/workspace.py`: `EXPORTS_SUBDIR` | No engine module imports `dw.server` except the entry point `dw/serve.py`, and a constant both sides need lives engine-side. | `scripts/arch_metrics.py` ratchet `import_cycles` | +| Module and function size | `scripts/arch_metrics.py` | A module stays at or under 1,100 lines (with a warning above 1,000), and a function stays at or under 150 lines. | `scripts/arch_metrics.py` ratchets `modules_over_size_ceiling` and `functions_over_150_lines` | + +## Server + +| Concept | Owner | Rule | Enforced by | +| --- | --- | --- | --- | +| App factory | `dw/server/app.py`: `create_app` | `create_app` builds the JobManager and `app.state`, installs the middleware and registers the routers, and all state lives in the JobManager. | `tests/test_server.py` | +| Routers | `dw/server/routes/*.py`, `dw/server/routes/__init__.py`: `ROUTERS` | There is one router per resource, registered in `ROUTERS` order because a greedy `{name:path}` route must come after its more specific siblings. | `tests/test_server_downloads.py::test_download_workflow_sets_content_disposition_attachment` | +| Job queue | `dw/server/jobs.py`: `JobManager` | One runner thread runs jobs FIFO on the single worker, and a job carries its own `output_dir`, `asset_dir` and `workflow_dir`, so it stays in its workspace. | `tests/test_server_jobs.py` | +| Job-to-run link | `dw/server/job_record.py`, `dw/server/jobs.py`: `JobManager.realized` | A job records `run_id`, `run_dir` and `run_version`, and `JobManager.realized` reads that run's `workflow.json`, confined to the output root. | `tests/test_server_jobs.py::test_realized_reads_the_file_the_run_wrote`, `tests/test_server_jobs.py::test_realized_refuses_a_run_dir_that_escapes_the_output_root` | +| Job history | `dw/server/job_history.py` | Finished jobs persist in `jobs.sqlite` (WAL mode), so the Jobs view survives a restart. | `tests/test_server.py::test_job_history_survives_restart_and_reruns`, `tests/test_job_history_wal.py::test_the_jobs_database_uses_wal_mode` | +| HTTP security | `dw/server/http_security.py` | Four middlewares read the bind and token values from `app.state` at request time, and only a route marked `query_token_ok` accepts `?token=`. | `tests/test_security_auth.py::TestTheTokenGate::test_every_api_spelling_needs_the_token` | + +## MCP + +| Concept | Owner | Rule | Enforced by | +| --- | --- | --- | --- | +| `dw_mcp` stays torch-free | `dw_mcp/` | `dw_mcp` is a top-level package that talks to `dw.serve` over HTTP and imports no `dw` module, because `dw/__init__.py` pulls in torch. | `tests/test_mcp_server.py::TestStartupWeight::test_the_server_starts_without_importing_the_engine` | +| Tool surface | `dw_mcp/server.py`, `dw_mcp/tools_*.py` | Only these modules import the MCP SDK, each tool body is a one-line call into a handler, and the registration order is the listing order an agent reads. | `tests/test_mcp_server.py::test_the_wiring_table_covers_every_registered_tool`, `tests/test_mcp_server.py::test_the_stated_tool_count_is_the_registered_one` | +| Surface text budget | `dw_mcp/server.py`, `dw_mcp/tools_*.py` | The instructions and each tool description stay at or under 2,048 characters (Claude Code truncates past that), and the whole surface stays within `SURFACE_BUDGET`. | `tests/test_mcp_server.py::test_no_text_the_agent_reads_is_cut_off_by_the_client`, `tests/test_mcp_server.py::test_the_tool_surface_fits_the_budget` | +| Spending needs consent | `dw_mcp/diagnose.py` | `run_workflow` and `rerun_job` refuse until `acknowledged_cost` is set, and submitting returns at once while progress is polled from the event log. | `tests/test_mcp_diagnose.py::test_run_refuses_without_an_acknowledged_cost`, `tests/test_mcp_diagnose.py::test_rerun_refuses_without_an_acknowledged_cost` | +| API errors | `dw_mcp/client.py` | An API failure becomes a message a person can act on here, and nowhere else. | `tests/test_mcp_client.py::test_a_400_surfaces_the_servers_detail_verbatim` | + +## UI + +| Concept | Owner | Rule | Enforced by | +| --- | --- | --- | --- | +| The UI reads engine fields | `ui/src/lib/plan.ts`: `describePlan`, `ui/src/lib/results.ts`: `sectionBySubfolder`, `ui/src/lib/pages/` | The UI reads `plan`, `version` and `subfolder` as fields the server sends and derives nothing of its own, and it never sends `acknowledged_cost`. | `ui/src/lib/plan.test.ts`, `ui/src/lib/results.test.ts` | diff --git a/tests/test_architecture_map.py b/tests/test_architecture_map.py new file mode 100644 index 00000000..d3c0ae15 --- /dev/null +++ b/tests/test_architecture_map.py @@ -0,0 +1,94 @@ +"""docs/ARCHITECTURE.md names only things that exist. + +The seam map sends an agent from a concept to the module that owns it and to +the test that enforces the rule. A map naming a deleted module or a renamed +test is worse than none, and nothing else would notice the drift: every +backticked repo path in it must exist (a glob must match), and every +`tests/x.py::test_name` must name a function defined in that file. +""" + +import re +from pathlib import Path + +REPO = Path(__file__).resolve().parents[1] +MAP = REPO / "docs" / "ARCHITECTURE.md" + +PREFIXES = ( + "dw/", + "dw_mcp/", + "ui/", + "scripts/", + "tests/", + "docs/", + "workflows/", + "prompts/", + ".github/", +) + +_TOKEN = re.compile(r"`([^`\s]+)`") + + +def map_paths(text): + """Every backticked token in `text` that starts with a repo prefix, as + (path, names). A `:name` suffix (a function in a module) is split off and + not checked; a `::Class::test_name` suffix gives the names, each of which + the file must define.""" + found = [] + for token in _TOKEN.findall(text): + if not token.startswith(PREFIXES): + continue + path, _, suffix = token.partition(":") + names = suffix[1:].split("::") if suffix.startswith(":") else [] + found.append((path, [name for name in names if name])) + return found + + +def map_faults(text, root=REPO): + """What in a map text names nothing: a missing path, an empty glob, or a + test function its file does not define.""" + faults = [] + for path, names in map_paths(text): + if "*" in path: + if not any(root.glob(path)): + faults.append(f"{path}: the glob matches nothing") + continue + target = root / path + if not target.exists(): + faults.append(f"{path}: no such file or directory") + continue + if path.startswith("tests/") and names and target.is_file(): + source = target.read_text() + *classes, function = names + for name in classes: + if not re.search(rf"^class {re.escape(name)}\b", source, re.M): + faults.append(f"{path}::{name}: no such test class") + if not re.search(rf"def {re.escape(function)}\(", source): + faults.append(f"{path}::{function}: no such test function") + return faults + + +def test_the_checker_reports_a_missing_path_and_a_missing_test(tmp_path): + (tmp_path / "dw").mkdir() + (tmp_path / "dw" / "real.py").write_text("") + (tmp_path / "tests").mkdir() + (tmp_path / "tests" / "test_real.py").write_text( + "class TestX:\n def test_here(self):\n pass\n" + ) + text = ( + "| concept | `dw/real.py`: owner | `dw/gone.py` |\n" + "| x | `tests/test_real.py::TestX::test_here` | " + "`tests/test_real.py::test_absent` | `tests/test_real.py::TestY::test_here` |\n" + "| y | `dw/*.py` | `dw_mcp/tools_*.py` | `pathlib` | `dw/real.py:fn` |\n" + ) + assert map_faults(text, tmp_path) == [ + "dw/gone.py: no such file or directory", + "tests/test_real.py::test_absent: no such test function", + "tests/test_real.py::TestY: no such test class", + "dw_mcp/tools_*.py: the glob matches nothing", + ] + + +def test_every_path_the_architecture_map_names_exists(): + text = MAP.read_text() + assert map_paths(text), "the map names no repo path" + assert map_faults(text) == [] From 265a47b98b56706c5a8f50a593569b2df8d2bdbe Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 14:28:47 -0500 Subject: [PATCH 26/41] docs: seam map review fixes; the map test checks owner names and ratchets The checker now reads `path`: `name`, `name` and requires each name to be defined or assigned in that file (or to be a key of a .json file, which is how a ratchet is named against docs/stabilization/baseline.json). A `file::Class::test` must be a test defined inside that class. Map: compound rules split into one sentence per row; partial enforcement narrowed or marked; whole-file citations replaced by specific tests; the Step and persistent-worker rows marked as unenforced; the builtin owner is resolve_sub_workflow_reference. Co-Authored-By: Claude Opus 5.5 --- docs/ARCHITECTURE.md | 172 ++++++++++++++++++--------------- tests/test_architecture_map.py | 142 +++++++++++++++++++++------ 2 files changed, 204 insertions(+), 110 deletions(-) diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index ceb380df..6b528104 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -1,134 +1,150 @@ # Architecture: the seam map -Where each concept lives. One row per concept: the module that owns it, the rule that holds -across the seam in one sentence, and what enforces the rule. The detail is in the owning -module's docstring, so read that before editing. The map only says which one to open. - -- **Owner**: repo-relative paths. A function name follows a colon (`dw/runs.py`: `run_versions`). -- **Enforced by**: a test (a file, or `file::Class::test_name`), a ratchet in `scripts/arch_metrics.py` - (its key in `docs/stabilization/baseline.json`), the CodeQL pack, or the validation registry. A - `—` means nothing mechanical holds the rule, so a change there is checked only by review. -- `tests/test_architecture_map.py` checks that every path named here exists and that every test - named here is defined in its file. +This map says where each concept lives. Each row names a concept, the module that owns it, the +rule that holds across the seam (in one sentence), and what enforces that rule. The detail is in +the owning module's docstring. Read the docstring before editing; the map only tells you which +one to open. + +- **Owner**: repo-relative paths. Function, class or constant names follow the path after a colon + (`dw/runs.py`: `run_versions`). +- **Enforced by**: one of four kinds. + - A test, written as a file, or as `file::Class::test_name`. + - A ratchet, named by its key in `docs/stabilization/baseline.json`. + - The CodeQL pack. + - The validation registry. + + A `—` means nothing mechanical holds the rule, so a change there is caught only in review. +- `tests/test_architecture_map.py` checks three things: every path named here exists, every name + after a path is defined in that file, and every test named here is defined in its class. ## Engine core | Concept | Owner | Rule | Enforced by | | --- | --- | --- | --- | -| Workflow | `dw/workflow.py`: `Workflow`, `workflow_from_file` | A workflow is loaded and checked here, then run by `Workflow.run`, which hands each phase to `dw/workflow_run.py`. | `tests/test_workflow.py` | -| One execution, in phases | `dw/workflow_run.py` | `Workflow.run` calls `prepare_run`, `open_run`, `begin_steps` and then `run_step` per step, and the cache probe `cache_hits` shares `prepare_definition` and `cache_lookup` with the run, so the probe answers for the run that would happen. | `tests/test_workflow_step_cache.py::TestCacheHits::test_after_a_run_the_probe_names_what_the_next_run_reuses` | -| Step | `dw/step.py`: `Step`, `dw/workflow.py`: `Workflow.create_step_action` | A step is exactly one of a pipeline, a task or a sub-workflow (the schema's `oneOf`), `create_step_action` builds its action, and `dw/workflow_run.py` runs it. | `tests/test_step.py` | -| Pipeline loading | `dw/pipeline_processors/pipeline.py`: `Pipeline`, `dw/pipeline_processors/components.py`: `load_component`, `configure_components` | Components load from `from_pretrained_arguments`, and a quantization config is built in `dw/pipeline_processors/config_objects.py` from a `config_type` that the `_type` key convention loads by name, so a new backend needs no code. | `tests/test_config_objects.py` | -| Placement and offload | `dw/pipeline_processors/placement.py`: `place_component` | Every device a workflow names passes through `resolve_device()` (`dw/__init__.py`) before placement reads its backend, so the MPS accommodations also fire for a translated device. | `tests/test_device_portability.py::TestPlacement::test_a_translated_device_gets_its_backends_offload_downgrade` | -| On-demand residency | `dw/pipeline_processors/placement.py`: `apply_on_demand_placement` | The wrappers keep the wrapped signature (`functools.wraps`; H3's denoiser reads `signature(transformer.forward)`), they cannot be combined with `group_offload` on one component, and either one stops the wholesale `pipeline.to(device)`. | `tests/test_modular_pipeline.py::TestOnDemandResidency::test_the_wrapped_signature_survives`, `tests/test_modular_pipeline.py::TestOnDemandResidency::test_group_offload_and_on_demand_together_are_rejected` | -| LoRAs | `dw/pipeline_processors/adapters.py`: `active_loras`, `load_loras` | A `loras` entry whose `model_name` is null is switched off, because a template's `loras` list is fixed JSON and a variable can null a value but cannot remove an entry. | `tests/test_lora_disable.py::TestLoadLoras::test_an_all_null_entry_loads_nothing` | -| Step progress | `dw/pipeline_processors/progress.py`: `reported_progress_bars` | A pipeline with no step callback (every `ModularPipeline`) reports its denoise steps through its progress bars instead. | `tests/test_modular_progress.py` | +| Workflow | `dw/workflow.py`: `Workflow`, `workflow_from_file` | A workflow is loaded and schema-checked here, and `Workflow.run` hands each phase of a run to `dw/workflow_run.py`. | `tests/test_workflow.py::test_workflow_validation_invalid` | +| One execution, in phases | `dw/workflow_run.py`: `prepare_definition`, `cache_lookup`, `cache_hits` | The cache probe shares `prepare_definition` and `cache_lookup` with the run, so the probe answers for the run that would happen. | `tests/test_workflow_step_cache.py::TestCacheHits::test_after_a_run_the_probe_names_what_the_next_run_reuses` | +| Step | `dw/step.py`: `Step`, `dw/workflow.py`: `Workflow.create_step_action` | A step is a pipeline, a task or a sub-workflow, `create_step_action` builds its action, and `dw/workflow_run.py` runs it. | — | +| Pipeline loading | `dw/pipeline_processors/pipeline.py`: `Pipeline`, `dw/pipeline_processors/components.py`: `load_component` | A quantization config is built from a `config_type` that the `_type` key convention loads by name, so a new backend needs no code. | `tests/test_config_objects.py::TestQuantizationConfiguration::test_the_config_type_is_constructed_with_its_arguments` | +| Device translation | `dw/pipeline_processors/placement.py`: `place_component` | A device is translated by `resolve_device()` before anything reads its backend, so the MPS accommodations also fire for a translated device. | `tests/test_device_portability.py::TestPlacement::test_a_translated_device_gets_its_backends_offload_downgrade` | +| On-demand wrappers | `dw/pipeline_processors/placement.py`: `apply_on_demand_placement` | The on-demand wrappers keep the wrapped signature, because H3's denoiser reads `signature(transformer.forward)`. | `tests/test_modular_pipeline.py::TestOnDemandResidency::test_the_wrapped_signature_survives` | +| On-demand and offload | `dw/pipeline_processors/placement.py`: `place_component`, `has_component_group_offload` | On-demand residency and `group_offload` cannot both be set on one component, and either one stops the whole pipeline from being moved to the device. | `tests/test_modular_pipeline.py::TestOnDemandResidency::test_group_offload_and_on_demand_together_are_rejected`, `tests/test_modular_pipeline.py::TestOffloadDevice::test_component_group_offload_is_not_moved_to_device` | +| LoRAs | `dw/pipeline_processors/adapters.py`: `active_loras`, `load_loras` | A `loras` entry whose `model_name` is null is switched off. | `tests/test_lora_disable.py::TestLoadLoras::test_an_all_null_entry_loads_nothing` | +| Step progress | `dw/pipeline_processors/progress.py`: `reported_progress_bars` | A pipeline with no step callback reports its denoise steps through its progress bars. | `tests/test_modular_progress.py::test_every_advance_of_the_bar_is_reported` | | Adding a task | `dw/tasks/task.py`: `register_command` | A task is a function registered with `@register_command`, and the signature of its implementation is its argument schema. | `tests/test_task_discovery.py::TestDescribeTask::test_signature_becomes_the_schema` | -| Task argument domains | `dw/task_domains.py` | A numeric domain that a signature cannot express is declared in its table, which is checked at validation and again at run time (`check_arguments`). | `tests/test_task_domains.py` | -| Variables | `dw/variables.py`: `set_variables`, `replace_variables`, `resolve_variable_values`, `argument_errors` | `argument_errors` folds a caller's arguments in exactly as `set_variables` does at the start of a run, so a bad name or value is a 400 before anything is queued. | `tests/test_variables.py::test_argument_errors_reports_a_dict_passed_for_a_string_variable`, `tests/test_variables.py::TestResolveVariableValues::test_an_undeclared_name_is_the_usual_error` | -| List variables from the CLI | `dw/run.py`, `dw/variables.py`: `get_value` | A `name=value` string given for a list variable is split on commas, so a list of objects (such as a template's `shots`) can only be passed as JSON over the API or MCP. | `tests/test_variables.py::test_set_variables_list_default_splits_on_comma` | -| `for_each` expansion | `dw/for_each.py`: `expand_for_each` | Expansion runs after substitution and before the reference check, and it names each member `@`, so the step loop, the cache and the manifest only ever see ordinary steps. | `tests/test_for_each.py::TestNaming::test_member_name_joins_with_at` | -| Previous results | `dw/previous_results.py`: `get_iterations`, `previous_result_reference_errors` | References multiply into a cartesian product, and a literal reference to a step that is not earlier is a validation error at its JSON path. | `tests/test_previous_results.py::TestGetIterations::test_multiple_references_create_cartesian_product`, `tests/test_previous_results.py::TestStaticReferenceChecking::test_a_step_cannot_reference_itself_or_a_later_one` | -| Type conversion | `dw/arguments.py`: `realize_args`, `dw/type_helpers.py` | A key ending `_type` or `_dtype`, or named `dtype`, loads a Python object at realize time, and a `{}`-wrapped value stays a string (see `docs/WORKFLOW_GUIDE.md`, "Types and escaping"). | `tests/test_type_helpers.py::TestGetType::test_get_type_from_diffusers`, `tests/test_arguments.py::TestRealizeArgs::test_realize_escaped_type_reference` | +| Task argument domains | `dw/task_domains.py`: `check_arguments` | A numeric domain that a signature cannot express is declared in this table, and it is checked at validation and again at run time. | `tests/test_task_domains.py::TestTheStaticPass::test_a_negative_frame_count_is_refused_at_its_path`, `tests/test_task_domains.py::TestAtRunTime::test_slice_audio_refuses_a_negative_count` | +| Variables | `dw/variables.py`: `set_variables`, `argument_errors`, `resolve_variable_values` | `argument_errors` folds a caller's arguments in exactly as `set_variables` does when a run starts, so a bad name or value is refused before anything is queued. | `tests/test_variables.py::test_argument_errors_reports_a_dict_passed_for_a_string_variable` | +| List variables from the CLI | `dw/run.py`, `dw/variables.py`: `get_value` | A `name=value` string given for a list variable is split on commas, so a list of objects can only be passed as JSON over the API or MCP. | `tests/test_variables.py::test_set_variables_list_default_splits_on_comma` | +| `for_each` expansion | `dw/for_each.py`: `expand_for_each` | Each entry becomes an ordinary step named `@`, after substitution and before the reference check. | `tests/test_for_each.py::TestNaming::test_member_name_joins_with_at` | +| Previous results | `dw/previous_results.py`: `get_iterations`, `previous_result_reference_errors` | Several `previous_result:` references in one step multiply into a cartesian product. | `tests/test_previous_results.py::TestGetIterations::test_multiple_references_create_cartesian_product` | +| Previous-result check | `dw/previous_results.py`: `previous_result_reference_errors` | A literal reference to a step that is not earlier in the workflow is a validation error. | `tests/test_previous_results.py::TestStaticReferenceChecking::test_a_step_cannot_reference_itself_or_a_later_one` | +| Type conversion | `dw/arguments.py`: `realize_args`, `dw/type_helpers.py`: `get_type` | A key ending `_type` or `_dtype` loads a Python object at realize time. The other spellings, and `{}` escaping, are in `docs/WORKFLOW_GUIDE.md`. | `tests/test_type_helpers.py::TestGetType::test_get_type_from_diffusers`, `tests/test_arguments.py::TestRealizeArgs::test_realize_escaped_type_reference` | ## References and libraries | Concept | Owner | Rule | Enforced by | | --- | --- | --- | --- | -| Reference prefixes | `dw/references.py` | Every prefix (`variable:`, `asset:`, `output:` and the rest) is spelled only here, and every module that tests for, strips or builds one calls this module. | `scripts/arch_metrics.py` ratchets `prefix_literals` and `prefix_handling`; `tests/test_references.py::test_references_imports_nothing_from_dw` | -| Explicit references resolve first | `dw/arguments.py`: `_realize_explicit_reference`, `dw/assets.py`, `dw/runs.py`: `resolve_output_reference`, `dw/prompts.py` | `asset:`, `output:`, `constant:` and `prompt:` resolve in `realize_args` before any key-name convention, and an `asset:` or `output:` path is confined to its library or output root by a `dw/security.py` validator. | `tests/test_assets.py::TestReferences::test_a_symlink_out_of_the_library_is_refused`, `tests/test_security_symlinks.py::TestOutputs::test_an_output_reference_does_not_follow_a_linked_run_directory` | -| Reference name shape | `dw/reference_names.py`: `reference_name_errors` | The shape of every `asset:`, `prompt:` and `output:` name is checked before the queue, but whether the file exists is checked against a workspace later. | `tests/test_reference_names.py` | -| Library search paths | `dw/library.py`: `LibraryPath`, `library_path` | Reads go through the roots front to back, so an earlier name shadows a later one, and writes go only to the front root, so saving something opened from a read-only root writes a copy. | `tests/test_library_path.py`, `tests/test_server_library_path.py::TestOneListingEnvelope::test_deleting_a_read_only_entry_gives_one_message` | -| Packaged builtins and sub-workflows | `dw/library.py`: `builtin_root`, `resolve_sub_workflow` | `builtin:` names the packaged `dw/workflows/` (not the top-level `workflows/` examples), and a sub-workflow path resolves beside its parent first, then by catalog name along the search path, and is confined to the root it resolves in. | `tests/test_sub_workflow_resolver.py` | -| Workspace | `dw/workspace.py`: `resolve_workspace`, `set_workspace` | The order is `--workspace`, then `DW_WORKSPACE`, then the `workspace` setting, then a working directory that looks like a workspace, then `~/diffusers-workspace`, and resolving creates nothing. | `tests/test_workspace.py::TestResolution::test_a_flag_wins_over_everything`, `tests/test_workspace.py::TestResolution::test_a_bare_working_directory_falls_back_to_the_home_workspace` | -| Realized workflow | `dw/realize.py`: `realize_workflow`, `dw/runs.py`: `write_realized_workflow` | Each run directory holds `workflow.json`, which pins every input that can change between runs: arguments, seed, stored prompt text and `output:.../latest/...`. | `tests/test_realize.py::TestVariablesAndSeed::test_arguments_become_the_variable_defaults` | +| Reference prefixes | `dw/references.py` | Every reference prefix is spelled only in this module, and every other module goes through it. | `docs/stabilization/baseline.json`: `prefix_literals`, `prefix_handling`; `tests/test_references.py::test_references_imports_nothing_from_dw` | +| Explicit references resolve first | `dw/arguments.py`: `_realize_explicit_reference` | `asset:`, `output:`, `constant:` and `prompt:` resolve in `realize_args` before any key-name convention. | — | +| Reference confinement | `dw/assets.py`: `resolve_asset_reference`, `dw/runs.py`: `resolve_output_reference` | `dw/security.py` validators confine an `asset:` path to its library and an `output:` path to the output root. | `tests/test_assets.py::TestReferences::test_a_symlink_out_of_the_library_is_refused`, `tests/test_security_symlinks.py::TestOutputs::test_an_output_reference_does_not_follow_a_linked_run_directory` | +| Reference name shape | `dw/reference_names.py`: `reference_name_errors` | The shape of every `asset:`, `prompt:` and `output:` name is checked before the queue, and existence is checked later. | `tests/test_reference_names.py::TestTheValidationPass::test_a_malformed_reference_is_refused_before_the_queue` | +| Library reads | `dw/library.py`: `LibraryPath` | Reads go through the roots front to back, so an earlier name shadows a later one. | `tests/test_library_path.py::TestResolution::test_the_front_of_the_path_shadows_the_rest` | +| Library writes | `dw/library.py`: `LibraryPath` | Writes go only to the front root, so saving something opened from a read-only root writes a copy. | `tests/test_library_path.py::TestConstruction::test_the_writable_root_comes_first`, `tests/test_library_sources.py::TestServer::test_saving_an_example_prompt_writes_a_copy` | +| Packaged builtins | `dw/library.py`: `builtin_root`, `resolve_sub_workflow_reference` | `builtin:` names the packaged `dw/workflows/`, not the top-level `workflows/` examples. | `tests/test_sub_workflow_resolver.py::TestTheResolver::test_a_builtin_resolves_in_the_packaged_root` | +| Sub-workflow paths | `dw/library.py`: `resolve_sub_workflow_reference`, `resolve_sub_workflow` | A sub-workflow path is confined to the root it resolves in. The search order is in the docstring of `resolve_sub_workflow`. | `tests/test_sub_workflow_resolver.py::TestTheResolver::test_a_path_escaping_its_root_is_refused` | +| Workspace | `dw/workspace.py`: `resolve_workspace`, `set_workspace` | `--workspace` beats `DW_WORKSPACE`, which beats the setting, which beats a working directory that looks like a workspace, which beats `~/diffusers-workspace`. | `tests/test_workspace.py::TestResolution::test_a_flag_wins_over_everything`, `tests/test_workspace.py::TestResolution::test_a_bare_working_directory_falls_back_to_the_home_workspace` | +| Realized workflow | `dw/realize.py`: `realize_workflow`, `dw/runs.py`: `write_realized_workflow` | Each run directory holds `workflow.json`, with the run's arguments, seed and stored prompt text pinned. | `tests/test_realize.py::TestVariablesAndSeed::test_arguments_become_the_variable_defaults`, `tests/test_runs.py::TestRealizedWorkflow::test_the_run_directory_holds_a_realized_copy` | ## Runs and outputs | Concept | Owner | Rule | Enforced by | | --- | --- | --- | --- | -| Run directories | `dw/runs.py`: `open_run`, `workflow_identity`, `new_run_id` | Each execution writes into `///` with a `manifest.json`, unless the flat layout (`output_layout`) is chosen. | `tests/test_runs.py` | -| Run versions | `dw/runs.py`: `open_run`, `run_versions`, `record_run_versions` | A run takes its number once, when it opens, as one more than the highest number recorded by any sibling, so deleting a middle run leaves a gap rather than renumbering. | `tests/test_runs.py::TestRunVersions::test_a_deleted_middle_run_leaves_a_gap_rather_than_renumbering`, `tests/test_runs.py::TestRunVersions::test_runs_started_in_the_same_second_still_number_upward` | -| Run-version surfaces | `dw/runs.py` (owner), `dw/server/routes/gallery.py`, `dw/server/outputs.py`, `dw_mcp/tools_catalog.py`: `list_gallery`, `ui/src/lib/pages/GalleryPage.svelte` | The number is computed only in `dw/runs.py`, and everywhere else it is read back, as a gallery field, as `output:/v/`, or as a job's `run_version`. | `tests/test_runs.py::TestRunVersions::test_an_output_reference_can_name_a_run_by_its_version` | -| Result subfolders | `dw/subfolders.py`: `subfolder_errors`, `dw/security.py`: `SUBFOLDER_PATTERN` | A step's `result.subfolder` is checked for shape after expansion and at run time, and it is confined to the run directory by `validate_output_path`. | `tests/test_subfolders.py` | -| Template subfolder roles | `workflows/templates/`, `dw/workflows/` | Every saving step of a template is marked `final` or `intermediate` with at least one `final`, and packaged builtins stay unmarked because assigning a role is the parent's job. | `tests/test_template_subfolders.py::test_every_saving_step_of_a_template_names_its_role`, `tests/test_template_subfolders.py::test_the_packaged_builtins_stay_unmarked` | -| Shots in a joined video | `dw/shots.py` | Every `AudioVideo` constructor carries, rescales, re-measures or drops `shots`, and the sample spans are measured from the joined waveform, not derived from frames. | `tests/test_shots.py` | -| Audio level of a deliverable | `dw/audio_qc.py`: `warn_without_headroom`, `warn_if_written_above_full_scale` | Each check warns and changes nothing, and a muxed video keeps the pre-encode headroom prediction to fall back on only when the post-write probe cannot measure the file. | `tests/test_result.py` | -| Template audio-level placement | `workflows/templates/minimax/music-video.json`, `workflows/templates/minimax/music.json` | `normalize_audio` (-3 dBFS) acts only on the track that goes into the final mux or file, so the slices that condition the shots are left untouched. | — | +| Run directories | `dw/runs.py`: `open_run`, `workflow_identity`, `new_run_id` | Each execution writes into `///` with a `manifest.json`, unless the flat layout is chosen. | `tests/test_runs.py::TestRunDirectories::test_the_flat_layout_writes_where_it_always_did` | +| Run versions | `dw/runs.py`: `open_run`, `run_versions`, `record_run_versions` | A run takes its number once, when it opens, so deleting a middle run leaves a gap rather than renumbering. | `tests/test_runs.py::TestRunVersions::test_a_deleted_middle_run_leaves_a_gap_rather_than_renumbering`, `tests/test_runs.py::TestRunVersions::test_runs_started_in_the_same_second_still_number_upward` | +| Run-version surfaces | `dw/runs.py`: `run_versions`, `dw/server/routes/gallery.py`, `dw/server/outputs.py`, `dw_mcp/tools_catalog.py`: `list_gallery`, `ui/src/lib/pages/GalleryPage.svelte` | Only `dw/runs.py` computes the number, and every other module reads it back. | `tests/test_runs.py::TestRunVersions::test_an_output_reference_can_name_a_run_by_its_version` | +| Result subfolders | `dw/subfolders.py`: `subfolder_errors`, `dw/security.py`: `SUBFOLDER_PATTERN` | A step's `result.subfolder` is checked for shape at validation. | `tests/test_subfolders.py::TestValidationErrorsIntegration::test_validation_errors_reports_a_bad_subfolder` | +| Template subfolder roles | `workflows/templates/`, `dw/workflows/` | Every saving step of a template is marked `final` or `intermediate`, with at least one `final`, and packaged builtins stay unmarked. | `tests/test_template_subfolders.py::test_every_saving_step_of_a_template_names_its_role`, `tests/test_template_subfolders.py::test_the_packaged_builtins_stay_unmarked` | +| Shots in a joined video | `dw/shots.py` | Every `AudioVideo` constructor decides what to do with `shots`, and the sample spans are measured from the joined waveform. | `tests/test_shots.py::test_every_audio_video_constructor_site_is_decided`, `tests/test_shots.py::TestConcatVideosShots::test_an_overrun_track_is_measured_not_derived` | +| Audio level of a deliverable | `dw/audio_qc.py`: `warn_without_headroom`, `warn_if_written_above_full_scale` | A muxed video keeps the pre-encode headroom prediction and falls back to it only when the post-write probe cannot measure the file. | `tests/test_result.py::TestNoHeadroom::test_a_clean_video_mux_drops_the_stale_prediction`, `tests/test_result.py::TestNoHeadroom::test_an_unprobeable_video_mux_falls_back_to_the_prediction`, `tests/test_result.py::TestTheWrittenLevel::test_a_file_that_decodes_above_full_scale_warns` | +| Template audio-level placement | `workflows/templates/minimax/music-video.json`, `workflows/templates/minimax/music.json` | `normalize_audio` (-3 dBFS) acts only on the track that goes into the final file, so the slices that condition the shots are untouched. | — | ## Validation, plan and cost | Concept | Owner | Rule | Enforced by | | --- | --- | --- | --- | -| Validation registry | `dw/validation.py`: `ERROR_CHECKS`, `WARNING_CHECKS`, `run_checks` | Adding a check means adding one `Check` to a registry, which runs it in order, and a check that raises becomes one `internal` finding while the rest still run. | `tests/test_validation.py::TestExceptionPolicy::test_a_raising_check_is_one_internal_error_and_the_rest_still_run` | -| Step-value checks | `dw/step_value_checks.py` | The bodies of `fps_errors`, `null_media_errors` and `select_errors` live here, and their only caller is the error registry in `dw/validation.py`. | the validation registry | -| Admission | `dw/server/admission.py`: `admit` | Validate, submit, rerun and enhance load and check a request once, and `JobManager.submit` queues what was admitted without checking it again. | `tests/test_admission.py::test_a_submit_expands_once` | -| Variable constraints | `dw/variable_constraints.py` | A model's rule about a value is declared in the workflow's `variable_constraints`, and it is checked at validation and at run time (`apply_constraints`) and reported in the catalog. | `tests/test_variable_constraints.py` | -| Plan | `dw/plan.py` | The plan comes from the same resolvers the run uses, and a bound `acknowledged_cost` is refused with 409 when the fingerprint or the required downloads changed. | `tests/test_server.py::TestBoundRerun::test_a_rerun_bound_to_a_stale_plan_is_refused` | -| Per-repo gating | `dw/plan.py`: `_collect_sources`, `downloads_required` | Each gated repo, including a `loras` entry's `model_name`, is probed and reported on its own, because access to one gated repo does not grant another. | `tests/test_plan.py::TestDownloadsRequired::test_a_gated_repo_this_token_lacks_access_to_is_blocked` | -| Observed cost | `dw/server/observed_cost.py`, `dw/plan.py`: `_tempered` | `observed` comes only from this box's finished jobs and never writes `cost`, and `plan.estimate` quotes the cold observed median ahead of the curated figure, blending toward that figure below three runs. | `tests/test_observed_cost.py::TestColdIsNotWarm::test_the_two_are_reported_separately_each_with_its_runs`, `tests/test_plan.py::TestLowConfidenceObservedEstimate::test_a_single_run_blends_toward_the_curated_figure` | -| Observed-cost residue | `dw/server/observed_cost.py`: `declared_drivers`, `dw/server/routes/library.py`: `get_workflow` | A cost driver that names no declared variable is dropped, and the raw workflow GET is served verbatim (with no `observed`) because the editor saves what it reads. | `tests/test_observed_cost.py::TestComparability::test_a_driver_naming_no_variable_is_dropped`, `tests/test_observed_cost.py::TestTheCatalogsDriversAreReal::test_every_declared_driver_is_a_variable_of_its_workflow` (the verbatim GET: —) | -| VRAM projection | `dw/vram_estimate.py` | A template's `vram_estimate` is projected for each pipeline step after `for_each` expansion, and only the largest step over the ceiling is reported. | `tests/test_vram_estimate.py` | -| H3 Ref2VA VRAM numbers | `workflows/templates/minimax/` (`vram_estimate`) | Every Ref2VA template declares `base_gb` 16.0, `bytes_per_voxel` 28.71 and `gb_per_reference` 1.0, a classification from field runs rather than a fitted curve. | `tests/test_h3_vram_ceiling.py::test_every_ref2va_minimax_template_declares_gb_per_reference` | -| Inherited VRAM ceiling | `dw/vram_inheritance.py`, `dw/server/deps.py`: `ceiling_index` | A workflow with no `vram_estimate` is matched by pipeline identity against an index of single-identity templates, cached on the listing's mtimes, and is warned, never refused. | `tests/test_vram_inheritance.py::test_every_template_declaring_one_identity_declares_the_same_numbers` | -| H3 adapter partition | `dw/adapter_compatibility.py` | An FL2VA LoRA on a reference (`ref2va`) step is refused, and a LoRA whose file name says neither `ref2v` nor `fl2v` is warned. | `tests/test_h3_adapters.py` | +| Validation registry | `dw/validation.py`: `ERROR_CHECKS`, `WARNING_CHECKS`, `run_checks` | A check is one `Check` in a registry, and a check that raises becomes one `internal` finding while the rest still run. | `tests/test_validation.py::TestExceptionPolicy::test_a_raising_check_is_one_internal_error_and_the_rest_still_run` | +| Step-value checks | `dw/step_value_checks.py`: `fps_errors`, `null_media_errors`, `select_errors` | The only caller of these checks is the error registry. | `dw/validation.py`: `ERROR_CHECKS` | +| Admission | `dw/server/admission.py`: `admit` | Validate, submit, rerun and enhance load and check a request once, and `JobManager.submit` does not check it again. | `tests/test_admission.py::test_a_submit_expands_once` | +| Variable constraints | `dw/variable_constraints.py`: `apply_constraints` | A model's rule about a value is declared in the workflow's `variable_constraints`, and it is checked at validation and again before anything loads. | `tests/test_variable_constraints.py::TestTheGrid::test_without_snap_an_off_grid_value_is_refused_not_rounded`, `tests/test_variable_constraints.py::TestTheRunTimePass::test_an_illegal_value_is_refused_before_anything_loads` | +| Bound acknowledgement | `dw/plan.py`: `fingerprint`, `dw/server/admission.py`: `check_bound_acknowledgement` | A bound `acknowledged_cost` is refused with 409 when the plan's fingerprint or its required downloads changed. | `tests/test_server.py::TestBoundRerun::test_a_rerun_bound_to_a_stale_plan_is_refused`, `tests/test_server.py::TestBoundAcknowledgement::test_a_matching_fingerprint_queues` | +| Per-repo gating | `dw/plan.py`: `build_plan` | Each gated repo, including a `loras` entry's repo, is probed and reported on its own. | `tests/test_plan.py::TestDownloadsRequired::test_a_gated_repo_this_token_lacks_access_to_is_blocked` | +| Observed cost | `dw/server/observed_cost.py`: `declared_drivers` | `observed` comes only from this box's finished jobs and never writes `cost`. Cold and warm runs are reported separately. | `tests/test_observed_cost.py::TestColdIsNotWarm::test_the_two_are_reported_separately_each_with_its_runs` | +| The estimate | `dw/plan.py`: `estimate`, `_tempered` | `plan.estimate` quotes the observed figure ahead of the curated one, and blends it toward the curated figure below three runs. | `tests/test_plan.py::TestObservedEstimate::test_history_beats_a_curated_figure`, `tests/test_plan.py::TestLowConfidenceObservedEstimate::test_a_single_run_blends_toward_the_curated_figure` | +| Cost drivers | `dw/server/observed_cost.py`: `declared_drivers` | A cost driver that names no declared variable is dropped. | `tests/test_observed_cost.py::TestComparability::test_a_driver_naming_no_variable_is_dropped`, `tests/test_observed_cost.py::TestTheCatalogsDriversAreReal::test_every_declared_driver_is_a_variable_of_its_workflow` | +| Raw workflow GET | `dw/server/routes/library.py`: `get_workflow` | The raw workflow GET is served verbatim, with no `observed`, because the editor saves what it reads. | — | +| VRAM projection | `dw/vram_estimate.py` | A template's `vram_estimate` is projected per step after `for_each` expansion, and only the largest step over the ceiling is reported. | `tests/test_vram_estimate.py::test_only_the_largest_over_ceiling_step_is_reported_once_per_cost_entry` | +| H3 Ref2VA VRAM numbers | `workflows/templates/minimax/` | Every Ref2VA template declares `base_gb` 16.0, `bytes_per_voxel` 28.71 and `gb_per_reference` 1.0. | `tests/test_h3_vram_ceiling.py::test_every_ref2va_minimax_template_declares_gb_per_reference` | +| Inherited VRAM ceiling | `dw/vram_inheritance.py`, `dw/server/deps.py`: `ceiling_index` | A workflow with no `vram_estimate` is matched against the catalog by pipeline identity, and is warned, never refused. | `tests/test_vram_inheritance.py::test_every_template_declaring_one_identity_declares_the_same_numbers` | +| H3 adapter partition | `dw/adapter_compatibility.py` | An FL2VA LoRA on a reference step is refused, and a LoRA whose file name says neither `ref2v` nor `fl2v` is warned. | `tests/test_h3_adapters.py::TestWhatIsRefused::test_a_keyframe_adapter_on_the_reference_path`, `tests/test_h3_adapters.py::TestWhatIsRefused::test_an_unrecognised_name_is_a_warning_not_a_refusal` | | IC-LoRA reference scale | `workflows/templates/ltx2/` | The three conditioning templates run at `reference_downscale_factor: 1`, and the two upscalers at 2. | `tests/test_ltx2_ic_loras.py::TestEachTemplateMatchesItsCard::test_the_reference_is_encoded_at_the_output_resolution`, `tests/test_ltx2_ic_loras.py::TestTheGenerativeUpscaleMatchesTheSameCard::test_it_loads_the_upscaler_at_factor_two` | -| IC-LoRA numbers | `workflows/templates/ltx2/` | Every number in the IC-LoRA templates is taken from the vendor card. | `tests/test_ltx2_ic_loras.py::TestEachTemplateMatchesItsCard::test_the_strength_is_the_cards_default`, `tests/test_ltx2_ic_loras.py::TestEachTemplateMatchesItsCard::test_the_defaults_are_the_trained_bucket` | -| IC-LoRA prompt genre | `prompts/ltx2/` (tag `ic-lora`) | An IC-LoRA stored prompt is checked against its trained caption form, not against the 150-220-word T2V paragraph rule. | `tests/test_ltx_prompt_library.py::test_an_ic_lora_prompt_is_in_its_trained_form` | +| IC-LoRA numbers | `workflows/templates/ltx2/` | Every number in the IC-LoRA templates comes from the vendor's model card. | `tests/test_ltx2_ic_loras.py::TestEachTemplateMatchesItsCard::test_the_strength_is_the_cards_default`, `tests/test_ltx2_ic_loras.py::TestEachTemplateMatchesItsCard::test_the_defaults_are_the_trained_bucket` | +| IC-LoRA prompt genre | `prompts/ltx2/` | A stored prompt tagged `ic-lora` is checked against its trained caption form, not against the T2V paragraph rule. | `tests/test_ltx_prompt_library.py::test_an_ic_lora_prompt_is_in_its_trained_form` | ## Execution and caching | Concept | Owner | Rule | Enforced by | | --- | --- | --- | --- | -| Step cache | `dw/step_cache.py`, `dw/workflow_run.py`: `cache_lookup` | A seeded rerun with unchanged inputs is served from the cache (`reused: true`, nothing written), a workflow with no `seed` skips the cache, and `rerun(new_seed=true)` is how to get a different result. | `tests/test_workflow_step_cache.py::test_cache_hit_marks_its_manifest_entry_and_event_reused`, `tests/test_workflow_step_cache.py::TestCacheHits::test_an_unseeded_workflow_has_no_hits`, `tests/test_rerun_new_seed.py::test_a_rerun_with_a_new_seed_draws_one_into_that_variable` | -| Worker protocol | `dw/worker_protocol.py` | Every command and reply is a frozen dataclass that travels as a wire dict, and this module imports neither `dw/worker.py` nor `dw/worker_manager.py`. | `tests/test_worker_messages.py::test_from_wire_inverts_to_wire`, `tests/test_worker_messages.py::test_an_unknown_reply_type_is_kept_whole_rather_than_raised` | -| Persistent worker | `dw/worker.py`, `dw/worker_manager.py`, `dw/serve.py` | Jobs run in one spawned worker process that keeps models loaded between runs, so a change to engine code needs a server restart. | `tests/test_worker_manager.py` | +| Step cache | `dw/step_cache.py`, `dw/workflow_run.py`: `cache_lookup` | A seeded rerun with unchanged inputs is served from the cache, marked `reused: true`, and writes nothing. | `tests/test_workflow_step_cache.py::test_cache_hit_marks_its_manifest_entry_and_event_reused`, `tests/test_workflow_step_cache.py::test_second_run_with_unchanged_step_reuses_cached_result` | +| No seed, no cache | `dw/workflow_run.py`: `prepare_run` | A workflow that sets no `seed` skips the cache. | `tests/test_workflow_step_cache.py::TestCacheHits::test_an_unseeded_workflow_has_no_hits` | +| A different result | `dw/server/routes/jobs.py`, `dw_mcp/tools_jobs.py`: `rerun_job` | `rerun` with `new_seed` draws a fresh seed into the workflow's seed variable. | `tests/test_rerun_new_seed.py::test_a_rerun_with_a_new_seed_draws_one_into_that_variable` | +| Worker protocol | `dw/worker_protocol.py`: `parse_reply` | Every command and reply is a frozen dataclass that travels as a wire dict. | `tests/test_worker_messages.py::test_from_wire_inverts_to_wire`, `tests/test_worker_messages.py::test_an_unknown_reply_type_is_kept_whole_rather_than_raised` | +| Persistent worker | `dw/worker.py`, `dw/worker_manager.py`, `dw/serve.py` | Jobs run in one spawned worker that keeps models loaded, so a change to engine code needs a server restart. | — | | Failed-run reporting | `dw/worker.py`, `dw/worker_protocol.py`: `Failed`, `Cancelled` | A failed or cancelled run's reply still carries the manifest of the steps that ran. | `tests/test_worker_execute.py::test_failure_carries_the_manifest_of_the_steps_that_ran`, `tests/test_worker_execute.py::test_cancellation_carries_the_manifest_too` | -| Run context and events | `dw/events.py`: `RunContext`, `emit_warning` | A run's context travels through contextvars, and a run-time warning reaches the job through `emit_warning`, not just the log. | `tests/test_events.py` | +| Run-time warnings | `dw/events.py`: `emit_warning` | A warning found at run time is emitted as an event, so it reaches the caller and not just the log. | `tests/test_concat_videos.py::TestWarningsReachTheCaller::test_the_level_spread_warning_is_emitted_as_an_event` | ## Media and DSP | Concept | Owner | Rule | Enforced by | | --- | --- | --- | --- | -| Opening media | `dw/media.py` | Every `av.open` in the engine is in this module, which uses PyAV only and imports no step types. | `tests/test_media_layering.py::test_av_open_appears_only_in_media`, `tests/test_media_layering.py::test_media_imports_no_step_types` | -| Signal processing | `dw/dsp.py` | Pure numpy, scipy and pyloudnorm: it measures and transforms a waveform, decides nothing, and imports nothing from `dw`. | `tests/test_media_layering.py::test_dsp_imports_nothing_from_dw` | -| Assessment probes | `dw/tasks/assess.py`, `dw/assessment_rules.py`, `dw/server/assess.py` | A probe measures a finished file and lists `findings` against the rules table, and nothing in the engine acts on a finding. | `tests/test_assessment_rules.py` | +| Opening media | `dw/media.py` | This is the only module that calls `av.open`. It imports no torch and no step types. | `tests/test_media_layering.py::test_av_open_appears_only_in_media`, `tests/test_media_layering.py::test_media_imports_no_step_types` (no torch: —) | +| Signal processing | `dw/dsp.py` | It measures and transforms waveforms, decides nothing, and imports nothing from `dw`. | `tests/test_media_layering.py::test_dsp_imports_nothing_from_dw` | +| Assessment probes | `dw/tasks/assess.py`, `dw/assessment_rules.py` | Exactly three registered commands are assessment probes. Each measures a finished file, and every rule names a real probe field. | `tests/test_assessment_rules.py::TestRuleProbesAreRealCommands::test_assessment_is_exactly_the_three_probes`, `tests/test_assessment_rules.py::TestRuleProbesAreRealCommands::test_every_rule_names_a_registered_json_command` | ## Security and architecture guardrails | Concept | Owner | Rule | Enforced by | | --- | --- | --- | --- | -| Path, URL and argument validators | `dw/security.py` | Filesystem access goes through `validate_path` / `validate_workflow_path` / `validate_output_path` (with a base), URLs through `validate_url`, and subprocess arguments through `sanitize_command_args`. | `tests/test_security.py::test_path_validation`, CodeQL (`.github/codeql/dw-security/`) | -| Locations from a workflow | `dw/locations.py` | A media location in a workflow's arguments is checked by one policy (a local path is confined, a URL is filtered for SSRF), because the workflow JSON is untrusted. | `tests/test_security_ssrf.py` | -| Trust gate | `dw/trust.py` | An untrusted workflow (the default) may name only allowlisted, constructible classes and no remote code, and a dotted type it may not use is a validation error before anything is imported. | `tests/test_security_trust_gate.py::TestValidationRefusesBeforeImport::test_a_dotted_type_is_a_validation_error` | -| CodeQL path-injection model | `.github/codeql/dw-security/`, `.github/workflows/codeql.yml`, `dw/security.py` | The local pack models the validators in `dw/security.py` as sanitizers (`validate_path` only when it is given a base), so a validator that moves is re-modelled in the same commit. | the CodeQL query dw/path-injection | -| Archives never follow links | `dw/server/outputs.py`: `zip_download`, `dw/security.py` | A gallery or asset listing drops a symlink that leaves its root, and an archive skips one. | `tests/test_security_symlinks.py::TestOutputs::test_the_archive_route_does_not_follow_the_link`, `tests/test_security_symlinks.py::TestOutputs::test_the_gallery_listing_does_not_enumerate_the_link` | -| Engine/server import direction | `dw/`, `dw/server/`, `dw/workspace.py`: `EXPORTS_SUBDIR` | No engine module imports `dw.server` except the entry point `dw/serve.py`, and a constant both sides need lives engine-side. | `scripts/arch_metrics.py` ratchet `import_cycles` | -| Module and function size | `scripts/arch_metrics.py` | A module stays at or under 1,100 lines (with a warning above 1,000), and a function stays at or under 150 lines. | `scripts/arch_metrics.py` ratchets `modules_over_size_ceiling` and `functions_over_150_lines` | +| Path validators | `dw/security.py`: `validate_path`, `validate_workflow_path`, `validate_output_path` | Filesystem access goes through a path validator that is given a base directory. | `tests/test_security.py::test_path_validation`, `.github/codeql/dw-security/DwPathInjection.ql` | +| URL and argument validators | `dw/security.py`: `validate_url`, `sanitize_command_args` | A URL must be http or https, and a subprocess argument must not carry a shell metacharacter. | `tests/test_security.py::TestValidateUrl::test_dangerous_schemes_are_rejected`, `tests/test_security.py::TestSanitizeCommandArgs::test_every_shell_metacharacter_is_rejected` | +| Locations from a workflow | `dw/locations.py` | A local path in a workflow's media arguments is confined to a known root, and a URL must not name a host inside the deployment. | `tests/test_locations.py::TestMediaPathContainment::test_absolute_path_outside_every_root_is_refused`, `tests/test_security_ssrf.py::TestInternalAddressSpellings::test_a_name_that_resolves_inside` | +| Trust gate | `dw/trust.py`: `require_trusted_dotted_name` | In an untrusted workflow (the default), a dotted type outside the allowlist is a validation error before anything is imported. | `tests/test_security_trust_gate.py::TestValidationRefusesBeforeImport::test_a_dotted_type_is_a_validation_error` | +| CodeQL path-injection model | `.github/codeql/dw-security/DwPathSanitizers.qll`, `.github/workflows/codeql.yml` | The local pack models the `dw/security.py` validators as sanitizers, so a validator that moves is re-modelled in the same commit. | `.github/codeql/dw-security/DwPathInjection.ql` | +| Archives never follow links | `dw/server/outputs.py`: `zip_download` | A gallery listing drops a symlink that leaves its root, and an archive skips one. | `tests/test_security_symlinks.py::TestOutputs::test_the_archive_route_does_not_follow_the_link`, `tests/test_security_symlinks.py::TestOutputs::test_the_gallery_listing_does_not_enumerate_the_link` | +| Engine/server import direction | `dw/workspace.py`: `EXPORTS_SUBDIR` | No engine module imports `dw.server` except the entry point `dw/serve.py`, and a constant both sides need lives engine-side. | `docs/stabilization/baseline.json`: `import_cycles` ratchet (indirect: it fails only on an import that closes a cycle) | +| Module and function size | `scripts/arch_metrics.py`: `SIZE_WARNING`, `SIZE_CEILING` | A module stays at or under 1,100 lines, and it warns above 1,000. A function stays at or under 150 lines. | `docs/stabilization/baseline.json`: `modules_over_size_ceiling`, `functions_over_150_lines` | ## Server | Concept | Owner | Rule | Enforced by | | --- | --- | --- | --- | -| App factory | `dw/server/app.py`: `create_app` | `create_app` builds the JobManager and `app.state`, installs the middleware and registers the routers, and all state lives in the JobManager. | `tests/test_server.py` | -| Routers | `dw/server/routes/*.py`, `dw/server/routes/__init__.py`: `ROUTERS` | There is one router per resource, registered in `ROUTERS` order because a greedy `{name:path}` route must come after its more specific siblings. | `tests/test_server_downloads.py::test_download_workflow_sets_content_disposition_attachment` | -| Job queue | `dw/server/jobs.py`: `JobManager` | One runner thread runs jobs FIFO on the single worker, and a job carries its own `output_dir`, `asset_dir` and `workflow_dir`, so it stays in its workspace. | `tests/test_server_jobs.py` | +| App factory | `dw/server/app.py`: `create_app` | `create_app` builds the JobManager and `app.state`, installs the middleware and registers the routes, and all state lives in the JobManager. | — | +| Routers | `dw/server/routes/*.py`, `dw/server/routes/__init__.py`: `ROUTERS`, `include_routers`, `include_file_routes` | Routers register in `ROUTERS` order, because a greedy `{name:path}` route must come after its specific siblings. The files router comes later, after `/mcp`. | `tests/test_server_downloads.py::test_download_workflow_sets_content_disposition_attachment` | +| Job queue | `dw/server/jobs.py`: `JobManager` | One runner thread runs jobs FIFO on the single worker. A job carries its own `output_dir`, `asset_dir` and `workflow_dir`, so it stays in its workspace. | `tests/test_server_workspaces.py::TestRunning::test_rerun_stays_in_the_workspace_it_ran_in` (FIFO: —) | | Job-to-run link | `dw/server/job_record.py`, `dw/server/jobs.py`: `JobManager.realized` | A job records `run_id`, `run_dir` and `run_version`, and `JobManager.realized` reads that run's `workflow.json`, confined to the output root. | `tests/test_server_jobs.py::test_realized_reads_the_file_the_run_wrote`, `tests/test_server_jobs.py::test_realized_refuses_a_run_dir_that_escapes_the_output_root` | -| Job history | `dw/server/job_history.py` | Finished jobs persist in `jobs.sqlite` (WAL mode), so the Jobs view survives a restart. | `tests/test_server.py::test_job_history_survives_restart_and_reruns`, `tests/test_job_history_wal.py::test_the_jobs_database_uses_wal_mode` | -| HTTP security | `dw/server/http_security.py` | Four middlewares read the bind and token values from `app.state` at request time, and only a route marked `query_token_ok` accepts `?token=`. | `tests/test_security_auth.py::TestTheTokenGate::test_every_api_spelling_needs_the_token` | +| Job history | `dw/server/job_history.py`: `JobHistory` | Finished jobs persist in `jobs.sqlite`, in WAL mode, so the Jobs view survives a restart. | `tests/test_server.py::test_job_history_survives_restart_and_reruns`, `tests/test_job_history_wal.py::test_the_jobs_database_uses_wal_mode` | +| HTTP security | `dw/server/http_security.py`: `install_middleware`, `query_token_ok` | Every API route needs the token, and only a route marked `query_token_ok` takes it as `?token=`. | `tests/test_security_auth.py::TestTheTokenGate::test_every_api_spelling_needs_the_token`, `tests/test_security_auth.py::TestTheTokenGate::test_the_query_token_is_refused_where_it_is_not_allowed` | ## MCP | Concept | Owner | Rule | Enforced by | | --- | --- | --- | --- | -| `dw_mcp` stays torch-free | `dw_mcp/` | `dw_mcp` is a top-level package that talks to `dw.serve` over HTTP and imports no `dw` module, because `dw/__init__.py` pulls in torch. | `tests/test_mcp_server.py::TestStartupWeight::test_the_server_starts_without_importing_the_engine` | -| Tool surface | `dw_mcp/server.py`, `dw_mcp/tools_*.py` | Only these modules import the MCP SDK, each tool body is a one-line call into a handler, and the registration order is the listing order an agent reads. | `tests/test_mcp_server.py::test_the_wiring_table_covers_every_registered_tool`, `tests/test_mcp_server.py::test_the_stated_tool_count_is_the_registered_one` | -| Surface text budget | `dw_mcp/server.py`, `dw_mcp/tools_*.py` | The instructions and each tool description stay at or under 2,048 characters (Claude Code truncates past that), and the whole surface stays within `SURFACE_BUDGET`. | `tests/test_mcp_server.py::test_no_text_the_agent_reads_is_cut_off_by_the_client`, `tests/test_mcp_server.py::test_the_tool_surface_fits_the_budget` | -| Spending needs consent | `dw_mcp/diagnose.py` | `run_workflow` and `rerun_job` refuse until `acknowledged_cost` is set, and submitting returns at once while progress is polled from the event log. | `tests/test_mcp_diagnose.py::test_run_refuses_without_an_acknowledged_cost`, `tests/test_mcp_diagnose.py::test_rerun_refuses_without_an_acknowledged_cost` | -| API errors | `dw_mcp/client.py` | An API failure becomes a message a person can act on here, and nowhere else. | `tests/test_mcp_client.py::test_a_400_surfaces_the_servers_detail_verbatim` | +| `dw_mcp` stays torch-free | `dw_mcp/` | `dw_mcp` reaches `dw.serve` over HTTP and imports no `dw` module, because `dw/__init__.py` pulls in torch. | `tests/test_mcp_server.py::TestStartupWeight::test_the_server_starts_without_importing_the_engine` | +| Tool surface | `dw_mcp/server.py`, `dw_mcp/tools_*.py` | Only these modules import the MCP SDK, and each tool body is a one-line call into a handler. | `tests/test_mcp_server.py::test_the_wiring_table_covers_every_registered_tool`, `tests/test_mcp_server.py::test_the_stated_tool_count_is_the_registered_one` | +| Surface text budget | the tool docstrings in `dw_mcp/tools_*.py`, the instructions in `dw_mcp/server.py` | The instructions and each tool description stay at or under 2,048 characters, and the whole surface stays within its token budget. | `tests/test_mcp_server.py`: `SURFACE_BUDGET`, `CLIENT_TEXT_LIMIT`; `tests/test_mcp_server.py::test_no_text_the_agent_reads_is_cut_off_by_the_client`, `tests/test_mcp_server.py::test_the_tool_surface_fits_the_budget` | +| Spending needs consent | `dw_mcp/diagnose.py` | `run_workflow` and `rerun_job` refuse until `acknowledged_cost` is set. | `tests/test_mcp_diagnose.py::test_run_refuses_without_an_acknowledged_cost`, `tests/test_mcp_diagnose.py::test_rerun_refuses_without_an_acknowledged_cost` | +| API errors | `dw_mcp/client.py` | An API failure becomes a message a person can act on, here and nowhere else. | `tests/test_mcp_client.py::test_a_400_surfaces_the_servers_detail_verbatim` | ## UI | Concept | Owner | Rule | Enforced by | | --- | --- | --- | --- | -| The UI reads engine fields | `ui/src/lib/plan.ts`: `describePlan`, `ui/src/lib/results.ts`: `sectionBySubfolder`, `ui/src/lib/pages/` | The UI reads `plan`, `version` and `subfolder` as fields the server sends and derives nothing of its own, and it never sends `acknowledged_cost`. | `ui/src/lib/plan.test.ts`, `ui/src/lib/results.test.ts` | +| The UI reads engine fields | `ui/src/lib/plan.ts`: `describePlan`, `ui/src/lib/results.ts`: `sectionBySubfolder` | The UI reads `plan`, `version` and `subfolder` as fields the server sends and derives nothing of its own, and it never sends `acknowledged_cost`. | `ui/src/lib/plan.test.ts`, `ui/src/lib/results.test.ts` | diff --git a/tests/test_architecture_map.py b/tests/test_architecture_map.py index d3c0ae15..e79e450d 100644 --- a/tests/test_architecture_map.py +++ b/tests/test_architecture_map.py @@ -1,12 +1,16 @@ """docs/ARCHITECTURE.md names only things that exist. The seam map sends an agent from a concept to the module that owns it and to -the test that enforces the rule. A map naming a deleted module or a renamed -test is worse than none, and nothing else would notice the drift: every -backticked repo path in it must exist (a glob must match), and every -`tests/x.py::test_name` must name a function defined in that file. +the test that enforces the rule. A map naming a deleted module, a renamed +function or a renamed test is worse than none, and nothing else would notice +the drift. So every backticked repo path in it must exist (a glob must +match). Every name written after a path - `path`: `name`, `name` - must be +defined or assigned in that file, or be a key of a `.json` file. Every +`tests/x.py::Class::test_name` must name a test defined in that class. """ +import ast +import json import re from pathlib import Path @@ -26,28 +30,74 @@ ) _TOKEN = re.compile(r"`([^`\s]+)`") +_NAME = r"`[A-Za-z_][\w.]*`" +# `path`: `name`, `name` - the names a path is followed by +_OWNED = re.compile(rf"`([^`\s]+)`:\s*({_NAME}(?:,\s*{_NAME})*)") def map_paths(text): - """Every backticked token in `text` that starts with a repo prefix, as - (path, names). A `:name` suffix (a function in a module) is split off and - not checked; a `::Class::test_name` suffix gives the names, each of which - the file must define.""" + """Every backticked repo path in `text`, as (path, test segments, owned + names). `::Class::test` segments come from the token itself; owned names + are the identifiers written after it as `path`: `name`, `name`.""" + owned = {} + for match in _OWNED.finditer(text): + names = re.findall(r"`([^`]+)`", match.group(2)) + owned.setdefault(match.start(), names) found = [] - for token in _TOKEN.findall(text): + for match in _TOKEN.finditer(text): + token = match.group(1) if not token.startswith(PREFIXES): continue - path, _, suffix = token.partition(":") - names = suffix[1:].split("::") if suffix.startswith(":") else [] - found.append((path, [name for name in names if name])) + path, _, suffix = token.partition("::") + segments = [s for s in suffix.split("::") if s] if suffix else [] + found.append((path, segments, owned.get(match.start(), []))) return found +def _python_names(tree): + """Every name a module defines or assigns, at any depth.""" + names = set() + for node in ast.walk(tree): + if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)): + names.add(node.name) + elif isinstance(node, ast.Assign): + names.update(t.id for t in node.targets if isinstance(t, ast.Name)) + elif isinstance(node, ast.AnnAssign) and isinstance(node.target, ast.Name): + names.add(node.target.id) + return names + + +def _defines_test(tree, segments): + """Whether `Class::test` (or a module-level `test`) is defined there: the + test inside that class, not anywhere in the file.""" + *classes, function = segments + body = tree.body + for name in classes: + match = [n for n in body if isinstance(n, ast.ClassDef) and n.name == name] + if not match: + return False + body = match[0].body + return any( + isinstance(n, (ast.FunctionDef, ast.AsyncFunctionDef)) and n.name == function + for n in body + ) + + +def _defines(target, name): + """Whether a non-python file defines `name`: a key of a JSON object, or a + function, class or binding in a script.""" + source = target.read_text() + if target.suffix == ".json": + return name in json.loads(source) + pattern = rf"\b(?:function|class|const|let|var|def)\s+{re.escape(name)}\b" + return re.search(pattern, source) is not None + + def map_faults(text, root=REPO): - """What in a map text names nothing: a missing path, an empty glob, or a - test function its file does not define.""" + """What in a map text names nothing: a missing path, an empty glob, a + name its file does not define, or a test its file does not hold.""" faults = [] - for path, names in map_paths(text): + for path, segments, names in map_paths(text): if "*" in path: if not any(root.glob(path)): faults.append(f"{path}: the glob matches nothing") @@ -56,38 +106,66 @@ def map_faults(text, root=REPO): if not target.exists(): faults.append(f"{path}: no such file or directory") continue - if path.startswith("tests/") and names and target.is_file(): - source = target.read_text() - *classes, function = names - for name in classes: - if not re.search(rf"^class {re.escape(name)}\b", source, re.M): - faults.append(f"{path}::{name}: no such test class") - if not re.search(rf"def {re.escape(function)}\(", source): - faults.append(f"{path}::{function}: no such test function") + if not target.is_file() or not (segments or names): + continue + tree = ast.parse(target.read_text()) if target.suffix == ".py" else None + if segments and not (tree and _defines_test(tree, segments)): + faults.append(f"{path}::{'::'.join(segments)}: no such test") + defined = _python_names(tree) if tree else None + for name in names: + parts = name.split(".") + if defined is not None: + ok = all(part in defined for part in parts) + else: + ok = all(_defines(target, part) for part in parts) + if not ok: + faults.append(f"{path}: {name}: not defined there") return faults -def test_the_checker_reports_a_missing_path_and_a_missing_test(tmp_path): +def test_the_checker_reports_what_names_nothing(tmp_path): (tmp_path / "dw").mkdir() - (tmp_path / "dw" / "real.py").write_text("") + (tmp_path / "dw" / "real.py").write_text( + "LIMIT = 3\n\nclass Thing:\n def method(self):\n pass\n" + ) + (tmp_path / "base.json").write_text('{"import_cycles": 0}') (tmp_path / "tests").mkdir() (tmp_path / "tests" / "test_real.py").write_text( - "class TestX:\n def test_here(self):\n pass\n" + "class TestX:\n def test_here(self):\n pass\n\n" + "def test_top():\n pass\n" ) text = ( - "| concept | `dw/real.py`: owner | `dw/gone.py` |\n" - "| x | `tests/test_real.py::TestX::test_here` | " - "`tests/test_real.py::test_absent` | `tests/test_real.py::TestY::test_here` |\n" - "| y | `dw/*.py` | `dw_mcp/tools_*.py` | `pathlib` | `dw/real.py:fn` |\n" + "| a | `dw/real.py`: `LIMIT`, `Thing.method` | `dw/gone.py` |\n" + "| b | `dw/real.py`: `absent` | `tests/test_real.py::TestX::test_here` |\n" + "| c | `tests/test_real.py::test_top` | `tests/test_real.py::test_absent` |\n" + "| d | `tests/test_real.py::TestY::test_here` |" + " `tests/test_real.py::test_here` |\n" + "| e | `dw/*.py` | `dw_mcp/tools_*.py` | `pathlib` |\n" ) assert map_faults(text, tmp_path) == [ "dw/gone.py: no such file or directory", - "tests/test_real.py::test_absent: no such test function", - "tests/test_real.py::TestY: no such test class", + "dw/real.py: absent: not defined there", + "tests/test_real.py::test_absent: no such test", + "tests/test_real.py::TestY::test_here: no such test", + "tests/test_real.py::test_here: no such test", "dw_mcp/tools_*.py: the glob matches nothing", ] +def test_every_ratchet_the_map_names_is_a_baseline_key(): + baseline = REPO / "docs" / "stabilization" / "baseline.json" + keys = set(json.loads(baseline.read_text())) + text = MAP.read_text() + named = { + name + for path, _, names in map_paths(text) + if path == "docs/stabilization/baseline.json" + for name in names + } + assert named, "the map names no ratchet" + assert named <= keys + + def test_every_path_the_architecture_map_names_exists(): text = MAP.read_text() assert map_paths(text), "the map names no repo path" From 5c099742fe571eec8cced02d36230b25a39b5198 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 14:33:26 -0500 Subject: [PATCH 27/41] docs(stabilization): move the docstring-verdict CLAUDE.md rules into their modules Fourteen rules from the 4c triage table now live in the module or function docstring that owns them. Docstrings only; the surface snapshot is unchanged. Co-Authored-By: Claude Opus 5.5 --- dw/__init__.py | 9 +++++++++ dw/pipeline_processors/placement.py | 7 +++++++ dw/previous_results.py | 5 ++++- dw/realize.py | 5 +++++ dw/reference_names.py | 7 +++++++ dw/runs.py | 7 +++++++ dw/server/job_history.py | 8 +++++++- dw/settings.py | 14 ++++++++++++++ dw/shots.py | 6 ++++++ dw/step_cache.py | 4 +++- dw/subfolders.py | 6 ++++++ dw/variable_constraints.py | 6 ++++++ dw_mcp/__init__.py | 4 ++++ 13 files changed, 85 insertions(+), 3 deletions(-) diff --git a/dw/__init__.py b/dw/__init__.py index 9f62d1f2..3a76faf5 100644 --- a/dw/__init__.py +++ b/dw/__init__.py @@ -1,3 +1,12 @@ +"""diffusers-workflow: device detection and the startup that configures it. + +The default torch device is deliberately never set. Diffusers loads weights +into system memory and then places them, onto the device or under offload +hooks; a default device of 'cuda' would build every module directly in VRAM +and run a large pipeline out of memory before any offload hook exists. Device +placement is explicit throughout dw. +""" + from .settings import resolve_path, load_settings from .log_setup import setup_logging import functools diff --git a/dw/pipeline_processors/placement.py b/dw/pipeline_processors/placement.py index 1dec13da..39fb081c 100644 --- a/dw/pipeline_processors/placement.py +++ b/dw/pipeline_processors/placement.py @@ -284,6 +284,13 @@ def place_component( accelerator and the adapter's own tensors are never among them, which runs the step on uninitialized weights and produces NaN. + `device` is translated by `resolve_device` first, before anything reads the + backend, so the MPS accommodations below (the sequential-to-model downgrade + here, and attention slicing and the compile skip elsewhere) fire for a + device translated to MPS as they do for one written as `mps`. + `exclude_from_cpu_offload` is sequential-only: it does not survive the + downgrade to model offload, which warns that it was dropped. + Args: component: The loaded pipeline or component component_name: What is being placed, for the log diff --git a/dw/previous_results.py b/dw/previous_results.py index f7e52077..0819889a 100644 --- a/dw/previous_results.py +++ b/dw/previous_results.py @@ -304,7 +304,10 @@ def _not_found(previous_results, previous_result_name): Names what is there as well as what was asked for: the gap between them is the fix, and on a long run it is the only thing standing between a - typo and another 40 minutes of GPU. + typo and another 40 minutes of GPU. A step whose result + `release_unreferenced_results` has already dropped is still named, under + "Earlier steps that ran" (from `completed_steps`), so the list is never + empty on a run that had finished steps. """ message = ( f"Previous result '{previous_result_name}' not found. " diff --git a/dw/realize.py b/dw/realize.py index eaa8f5c6..e191e963 100644 --- a/dw/realize.py +++ b/dw/realize.py @@ -13,6 +13,11 @@ the variables block, which records the values the run folded - a variable defaulted to a 'constant:' holds the value it realized to there. +The realized file keeps `for_each` as written - a step carrying one is not +expanded into its members - so it is the file an author would edit and +re-run. The members (`@`) are named in the run's manifest, not +here. + Two rules hold this module together. It never mutates its input: the caller hands it the definition the run is about to work from. And it never fails a run: a reference that will not resolve is left exactly as written, so the diff --git a/dw/reference_names.py b/dw/reference_names.py index 08492f16..c3d856ff 100644 --- a/dw/reference_names.py +++ b/dw/reference_names.py @@ -15,6 +15,13 @@ own references (a template's `prompt:ltx2/hummingbird_garden`) are not the caller's to answer for; the caller's `arguments` are separately resolved against the workspace by the validate route. + +An `output:` or `asset:` name accepts `@`, because the engine writes it: a +`for_each` member is `@` and its files carry that in their base +name, so every file the server names can be named back to it. `@` is not a +separator and not `..`, and containment is still `validate_path`'s; a name +(and each segment of it) may not *start* with one. `_name_fault`, in +`dw/security.py`, is what names the offending character and its position. """ from . import references diff --git a/dw/runs.py b/dw/runs.py index 5f0ad9cc..5a2d9e76 100644 --- a/dw/runs.py +++ b/dw/runs.py @@ -528,6 +528,13 @@ def run_versions(identity_dir): unrecorded run anywhere later continues from the highest number before it. Ordering is by run id, which is chronological. + A new run takes `max(recorded) + 1` over *every* sibling manifest + (`open_run`), not one past the newest: run ids are chronological only to + the second, so within one second the digest decides the sort. Two limits + are deliberate. Deleting the *newest* run frees its number for reuse, + since the high-water mark lived in the manifest that went with it; and + the flat layout has no runs, so a version is None there. + Read only. A ranked number is only as stable as its neighbours until `record_run_versions` writes it down. """ diff --git a/dw/server/job_history.py b/dw/server/job_history.py index c7609623..da7a52e1 100644 --- a/dw/server/job_history.py +++ b/dw/server/job_history.py @@ -1,4 +1,10 @@ -"""Finished jobs, persisted so the Jobs view survives server restarts.""" +"""Finished jobs, persisted so the Jobs view survives server restarts. + +The `jobs` table has a `workspace` column; a database that predates it gets +the column added and every existing row backfilled to `default`, since +history that cannot say which workspace a job ran in stops making sense once +there are two. +""" import json import logging diff --git a/dw/settings.py b/dw/settings.py index 70c0c1d5..290304be 100644 --- a/dw/settings.py +++ b/dw/settings.py @@ -1,3 +1,17 @@ +"""The standing settings, read from `~/.diffusers_helper/settings.json`. + +Keys: `device`, `workspace`, `output_layout`, `public_url`, `enable_tf32`, +`cudnn_benchmark`, `cudnn_deterministic`, `log_level`, `log_filename` and +`log_to_console`. A missing or unreadable file means the defaults on `Settings`. + +A setting is the standing choice and each has something that overrides it for +one run: `DW_DEVICE` for `device`; `--workspace` then `DW_WORKSPACE` for +`workspace`; `--output-layout` / `DW_OUTPUT_LAYOUT` for `output_layout`; +`DW_PUBLIC_URL` for `public_url`; a `log_level` passed to `startup()`. The +three PyTorch keys have no override and are read when `startup()` configures +CUDA. +""" + import json import os from pathlib import Path diff --git a/dw/shots.py b/dw/shots.py index 89b8633b..5ea4f54b 100644 --- a/dw/shots.py +++ b/dw/shots.py @@ -31,6 +31,12 @@ rescales it (`interpolate_frames`), re-measures the sample side for a new track (`pair_audio`), or builds a video with no shots at all. `tests/test_shots.py` fails on a constructor site nobody decided for. + +`Result.save` keeps each file's shots as plain data in `saved_shots` (path -> +shots), so a step cache hit's stripped copy still reports them. The manifest +entry and `step_end` carry them renamed `shot@` from the step's `videos` +references (`step_shots`). The mp4 itself carries nothing: the record lives in +the manifest and on the artifact, not in the file. """ import copy diff --git a/dw/step_cache.py b/dw/step_cache.py index 6d461756..67692998 100644 --- a/dw/step_cache.py +++ b/dw/step_cache.py @@ -9,7 +9,9 @@ dicts/lists/scalars after variable substitution, not hashable). A step is safe to skip only if: - 1. its own resolved definition (step_data) matches last run's, AND + 1. its own resolved definition (step_data) matches last run's, AND - step_data + is the whole step, `result` block included, so changing a step's + `subfolder` misses the cache 2. its seed matches last run's - step_data does NOT carry the seed (Workflow.run resolves it separately, and draws a fresh random one per run when the workflow sets none), so seed must be compared diff --git a/dw/subfolders.py b/dw/subfolders.py index be21f39c..263d80fd 100644 --- a/dw/subfolders.py +++ b/dw/subfolders.py @@ -12,6 +12,12 @@ definition. Containment - that the joined path really is inside the run directory - is the engine's, at the moment it joins (see Workflow.step_output_dir). + +A subfolder is `output:`-addressable only up to OUTPUT_REFERENCE_PATTERN's +ceiling of seven segments (workflow identity, run id, subfolder and file name +all count). SUBFOLDER_PATTERN has no depth bound of its own, so a deep +subfolder under a nested identity validates and runs but cannot be named by a +later `output:` reference. """ from . import references diff --git a/dw/variable_constraints.py b/dw/variable_constraints.py index 732cdb0d..929d2493 100644 --- a/dw/variable_constraints.py +++ b/dw/variable_constraints.py @@ -22,6 +22,12 @@ `"constraint:"` so a template states `17n + 5` once rather than twice in one file. +`snap` is chosen per model, by what its pipeline does with an off-grid value. +H3 rounds up, so its templates say `snap: "up"`. LTX-2.5's templates declare +the `8 * n + 1` grid with no `snap`, because those pipelines floor an off-grid +count rather than raising: rounding up here would be a second silent change +to the length, so the value is refused instead. + A constraint key is a plain variable name, matched wherever a value by that name sits: a top-level variable, or a field of an entry of a `for_each` list where some step hands that field to a pipeline (#145). `dialogue-short` has diff --git a/dw_mcp/__init__.py b/dw_mcp/__init__.py index 22144130..6de8f8b1 100644 --- a/dw_mcp/__init__.py +++ b/dw_mcp/__init__.py @@ -3,4 +3,8 @@ A stdio MCP server that is an HTTP client of a running `dw.serve`. It owns no job state and no GPU worker - every tool is a call against the REST API that the web UI already uses. + +The surface covers the REST API except three things: the SSE event stream +(`get_job_events` reads its polling twin, `/event-log`), the two bulk zips of +the gallery and of the assets, and the SPA's static mount. """ From cddb1e459a03e7e5cf0a00f50d7864516b7c2c6e Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 14:35:12 -0500 Subject: [PATCH 28/41] docs(docstrings): subfolder depth counts path segments; _not_found's list claim narrowed Co-Authored-By: Claude Opus 5.5 --- dw/previous_results.py | 4 ++-- dw/subfolders.py | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/dw/previous_results.py b/dw/previous_results.py index 0819889a..d882b460 100644 --- a/dw/previous_results.py +++ b/dw/previous_results.py @@ -306,8 +306,8 @@ def _not_found(previous_results, previous_result_name): is the fix, and on a long run it is the only thing standing between a typo and another 40 minutes of GPU. A step whose result `release_unreferenced_results` has already dropped is still named, under - "Earlier steps that ran" (from `completed_steps`), so the list is never - empty on a run that had finished steps. + "Earlier steps that ran" (from `completed_steps`), so a typo is still + checked against every step that ran. """ message = ( f"Previous result '{previous_result_name}' not found. " diff --git a/dw/subfolders.py b/dw/subfolders.py index 263d80fd..d8c48bfc 100644 --- a/dw/subfolders.py +++ b/dw/subfolders.py @@ -14,8 +14,8 @@ Workflow.step_output_dir). A subfolder is `output:`-addressable only up to OUTPUT_REFERENCE_PATTERN's -ceiling of seven segments (workflow identity, run id, subfolder and file name -all count). SUBFOLDER_PATTERN has no depth bound of its own, so a deep +ceiling of seven path segments in all: a nested identity or subfolder counts +one per `/`, plus the run id and the file name. SUBFOLDER_PATTERN has no depth bound of its own, so a deep subfolder under a nested identity validates and runs but cannot be named by a later `output:` reference. """ From 1781680d8b0ff542a60a9f068caedab51acaede1 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 14:39:27 -0500 Subject: [PATCH 29/41] docs(claude-md): cut the root CLAUDE.md to a map (674 -> 108 lines) The root keeps only the triage's keep rows: the overview, the common commands plus the unit-test line, a "where things are" block that sends an agent to docs/ARCHITECTURE.md before grepping (with the worker-restart and builtin-vs-workflows facts), the plugin paragraph with every skill named, the security rules with the two CodeQL edit-time lines, and three "before you edit" gotchas, each ending at its owner. Everything else is in a doc, a docstring or a seam-map row, per docs/stabilization/phase-4-surveys/claude-md-triage.md. Co-Authored-By: Claude Opus 5.5 --- CLAUDE.md | 682 +++++------------------------------------------------- 1 file changed, 58 insertions(+), 624 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index eae3d8ed..78ec40a8 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -15,10 +15,8 @@ bash ./install.sh && source ./activate # HTTP server + web UI (http://127.0.0.1:8765, API docs at /docs) python -m dw.serve -# Run a workflow - dw.run is a thin client of dw.serve, above; it queues the -# job over HTTP and never runs one itself. templates/text-to-image.json uses -# a small, ungated model and a literal prompt, so it needs no Hugging Face -# login and downloads only a few GB +# Run a workflow - dw.run only queues the job on a running dw.serve. This +# template uses a small, ungated model, so it needs no Hugging Face login python -m dw.run workflows/templates/text-to-image.json python -m dw.run workflows/templates/text-to-image.json prompt="a cat" num_images_per_prompt=4 @@ -28,21 +26,49 @@ python -m dw.run workflows/templates/text-to-image.json prompt="a cat" num_image # Validate a workflow against schema python -m dw.validate workflows/models/z-image.json +# Unit tests - parallel across every core (pytest-xdist, `-n auto` in +# pytest.ini); naming test files runs them in one process +pytest tests/ -v +pytest tests/test_security.py -v + # System test - downloads SD 1.5 (a few GB) and generates one image python -m dw.test ``` -## Architecture - -### Server & Web UI - -`dw/serve.py` runs a FastAPI app over a persistent worker process, -queueing jobs FIFO and persisting history to `~/.diffusers_helper/jobs.sqlite`. -See docs/SERVER.md, `dw/server/CLAUDE.md` and `ui/CLAUDE.md`. - -### MCP Server - -The stdio MCP server lives in `dw_mcp/` — see `dw_mcp/CLAUDE.md` and docs/MCP.md. +## Where things are + +`docs/ARCHITECTURE.md` is the map: concept, owning module, the rule that holds across +the seam, and the test or check that enforces it. Open it before grepping - it names the +module, and that module's docstring holds the detail. Its sections are engine core, +references and libraries, runs and outputs, validation/plan/cost, execution and caching, +media and DSP, security, server, MCP and UI. + +- `dw.serve` runs every job in one persistent spawned worker (`dw/worker.py`, managed by + `dw/worker_manager.py`) that keeps models cached between runs, so a change to engine + code needs a server restart before a job sees it. +- The packaged `dw/workflows/` is what a `builtin:` step names (resolved in + `dw/library.py`); it is not the top-level `workflows/` folder of runnable examples. +- The workflow schema is `dw/workflow_schema.json` - read it for the full structure. +- The HTTP server is `dw/server/` (see `dw/server/CLAUDE.md`), the web UI is `ui/` (see + `ui/CLAUDE.md`), and the stdio MCP server is `dw_mcp/` (see `dw_mcp/CLAUDE.md`). + +The guides in `docs/`, by topic: +- `WORKFLOW_GUIDE.md` - writing workflow JSON. It owns the reference conventions + (`variable:`, `previous_result:`, `constant:`, `asset:`, `output:`, `prompt:`, `item:`, + `gather:`): its *Authoring a workflow from an agent* section, *References* first, and + the *Type System* section. +- `WORKSPACES.md` - workspaces, the library search paths, run directories and manifests. +- `SERVER.md` - the HTTP API and web UI; `REMOTE.md` - using a server on another machine. +- `MCP.md` - the MCP tool surface; `AGENT_LOOP.md` - the automated implementer/tester loop. +- `WORKER_GUIDE.md` - the persistent worker and its queue protocol. +- `TASKS.md` - the utility task commands (`dw/tasks/`). +- `ACCELERATION.md` - caching, compile, attention backends, device settings, MPS; + `QUANTIZATION.md` - per-component quantization backends; `RECIPES_24GB.md` - tested + combinations of both per model family. +- `LORAS.md`, `IP_ADAPTER.md`, `PROMPT_WEIGHTING.md` - those features, one each. +- `SECURITY.md` - the validators and trust gate in depth; `SECURITY_QUICKREF.md` - the + same at a glance. +- `TESTING.md` - the test suite; `DEPENDENCIES.md` - installation; `RELEASING.md` - releases. ### Claude Code plugin @@ -52,201 +78,8 @@ chooses a template for a request's shape and states the family's hard rules, plu cross-cutting composition skills (`script-to-video`, `series-episodes`) - shapes above the families that orchestrate the decision trees and cast consistency across multiple generations. Every skill the directory holds is named in `plugins/dw/README.md` and here, -pinned by the same test. Model knowledge lives there and in the catalog, never in engine -code; every number a skill states is pinned to a diffusers symbol by -`tests/test_plugin_skills.py`. `plugin.json`'s version is the engine's, bumped by -`scripts/release.sh`. Adding or re-auditing a family is `.claude/skills/model-family-onboarding/`. - -### Worker - -A **persistent worker subprocess** (`dw/worker.py`), managed by `dw/worker_manager.py`, keeps GPU models cached between runs for `JobManager`. Communication is via `multiprocessing.Queue`, with `multiprocessing.set_start_method("spawn")` for CUDA/MPS compatibility. - -### Workspaces on the server - -`dw.serve` can hold several workspaces under one root: the root's own -`workflows/assets/outputs` are the `default` workspace, a named one is a -subdirectory beside them (`named_workspace`, `create_workspace` in -`dw/workspace.py`), and the server's prompt library (`--prompt-dir`) is shared by -all of them - there is one, because `prompt:` is shared by reference. Routes take an -optional `workspace`; omitting it means the default, so pre-workspace calls are -unchanged. A job carries its own `output_dir`, `asset_dir` and `workflow_dir` -(`JobManager.submit`), so it stays in its workspace whatever the manager serves -next; the worker activates the asset root per job (`activate_asset_dir`), which -is the one root that could not stay process-wide. `jobs.sqlite` has a -`workspace` column, backfilled to `default`. `common/assets` at the root is -the one asset library every workspace shares - assets are otherwise per -workspace, which is wrong for a recurring cast a later workspace still has to -reach. It sits on every workspace's asset search path behind that workspace's -own library (so a workspace name shadows a shared one), is tagged `origin: -common` by `GET /api/assets`, and is written to only when a call says so -(`?shared=true` on uploads, `"shared": true` on keep, `shared=True` over MCP). -Reserved names: `workflows`, `prompts`, `assets`, `outputs`, `exports`, -`common`. The web UI is organised by workspace, with a sidebar listing every -workspace on the server (`ui/src/lib/Sidebar.svelte`) and the selected one -named in the hash (`#/ws//...`). - -The web UI has a page for it: `ui/src/lib/pages/AssetsPage.svelte` -reads `GET /api/assets` and shows the library the way the gallery shows -outputs, tagged by `origin` so a shadowed or read-only entry is visible -before a 403 explains it. - -### Library search paths - -`dw/library.py` owns the three content libraries' search paths (workflows, prompts, -assets): `LibraryRoot` is one root (`origin`, `root`, `writable`) and `LibraryPath` -(`library_path(kind, ...)`) the ordered path - the workspace's own root first, then -`common/assets` (assets only), then each `--examples-dir` (and the `prompts/` and -`assets/` beside it, `example_libraries` in `dw/workspace.py`), all read-only. -`find`/`entries` span every root front-to-back, so an earlier name shadows a later one -and the hidden copies come back as `shadowed`; a symlink out of its root is a miss; -saves go through `writable_root`, so saving something opened from a read-only root -writes a copy, and deleting a read-only entry is the one 403 (`ReadOnlyLibraryError`). -`dw.serve` pins the read-only tails into `DW_PROMPT_PATH` / `DW_ASSET_PATH` so the -spawned worker resolves as the API does. A job carries the root it is confined to -(`JobManager.submit(workflow_dir=...)`), so an examples workflow runs confined to the -examples directory. All three listings share one envelope: `libraries: [{origin, root, -writable}]`, per-entry `origin`/`writable`, `shadowed: [{name, origin, shadowed_by}]`. -Packaged `dw/workflows/` is off the path - it is what `builtin:` steps name, -resolved in `dw/workflow.py`. - -### Workspaces - -`dw/workspace.py` resolves the one directory a run's content belongs to - -`workflows/`, `prompts/`, `assets/`, `outputs/`. Order: `--workspace` > -`DW_WORKSPACE` > the `workspace` setting > the working directory when it holds -any of `workflows/`, `prompts/` or `outputs/` > `~/diffusers-workspace`. A -checkout satisfies rule four, so every default lands where it did before -workspaces existed. Resolution creates nothing; an entry point about to write -calls `ensure()` (or creates the one folder it needs). `set_workspace` pins the -root *and* how it was chosen into the environment, so a spawned worker does not -read an inferred workspace back as one the user named - `get_prompt_dir` yields -to its older discovery (`./prompts`, then the walk up from the workflow file) -for an inferred workspace but not for an explicit one. `--workflow-dir`, -`--output-dir` and `--prompt-dir` each still override one folder. See docs/WORKSPACES.md; -the later stages (workflow search path, run directories, `asset:`/`output:` -references) are documented above in *Library search paths* and *Type System*. - -### Type System - -`arguments.py` + `type_helpers.py` handle dynamic type conversion during workflow loading: -- Keys ending in `_type` or `_dtype`, or named `dtype`, are auto-converted: `"FluxPipeline"` → loaded from `diffusers`, `"torch.bfloat16"` → `torch.bfloat16` -- Values wrapped in `{}` are escaped (stay as strings): `"{nf4}"` → `"nf4"` -- Dotted names use full module path: `"sdnq.SDNQConfig"` → `importlib.import_module("sdnq").SDNQConfig` -- Values prefixed with `constant:` read a value declared in python rather than copying it - into JSON: `"constant:diffusers.pipelines.ltx2.utils.DISTILLED_SIGMA_VALUES"`. Resolved - in `realize_args`, validated by `validate_constant_name()`; anything callable is refused -- Values prefixed with `asset:` resolve to the path of a file in the asset library: - `"asset:iris.png"` or `"asset:gyre/frames/web.mp4"`. Resolved in `realize_args` before - every other convention (`dw/assets.py`), rooted at the library rather than the workflow - file, confined to it, and then loaded by whatever would have loaded a path written - there. The library is `DW_ASSET_DIR` / `--asset-dir`, else the workspace's `assets/` - when a workspace was named, else `./assets` if it exists, else found by walking up - from the workflow file's directory -- Values prefixed with `output:` resolve to the path of a file an earlier run wrote: - `"output:ltx2/Gyre/latest/still.png"`. The name is `//` - under the output root, and `latest` in the run-id position picks the newest run that - holds the file (run ids sort by their UTC timestamp; a failed or fully-cached run holds - only a manifest and is skipped), and `v` there picks the run whose version is N - (below) - exactly that run, with no fallback to an older one. Either is a selector only - where run directories are, and the realized workflow pins both to the run id. Resolved in `realize_args` beside `asset:` (`dw/runs.py`), - against the output root `Workflow.run` activates, and confined to it -- A generated file becomes a stable input with `POST /api/assets/keep` (gallery "Keep as - asset", MCP `keep_output`): it is hard-linked, else copied, from the workspace's outputs - into its assets under a chosen name, so later workflows reference `asset:name` rather - than a run id that pruning would break -- Values prefixed with `prompt:` load a stored prompt's `text` from the prompt library: - `"prompt:name"` or `"prompt:folder/name"`. Resolved in `realize_args` (`dw/prompts.py`), - rooted at the library rather than the workflow file. The library is `DW_PROMPT_DIR` / - `--prompt-dir`, else the workspace's `prompts/` when a workspace was named, else - `./prompts` if it exists, else found by walking up from the workflow file's - directory, else the workspace's `prompts/` -- A step's `result.subfolder` names a subfolder of the run directory for that step's - files - by convention `final` for the deliverable and `intermediate` for the rest; any - relative path (`shots/act-1`); `variable:`/`item:` allowed; no default. Mechanics under - *Result subfolders* in Critical Gotchas -- A step carrying `for_each` (a list, or `variable:` naming one) is expanded by - `expand_for_each` (`dw/for_each.py`) into one ordinary step per entry, named - `@`, immediately after `replace_variables` in - `Workflow.run` and, with the caller's arguments folded, in `validation_errors`. - Inside a member `item:` / `item:field` is the entry (any type, spliced whole); - a later step reads the group with `gather:` (a list; splices inside a - list); two groups over the same list pair by key (`slice` inside `shot@x` is - `slice@x`). `previous_result:` naming a group is a directed error. `@` is - reserved in step names; entry names are validated and unique; 32 entries max; - `release_pipeline`/`release_models` survive on the last member only. The - realized workflow keeps `for_each`; the manifest names the members. An entry - of a list-valued variable may reference another variable - (`"from_file": "variable:character_a_voice"`); `resolve_variable_values` - (`dw/variables.py`) replaces those once, before `realize_args`, refusing a - cycle, and `undeclared_variable_references` walks inside list/dict variable - values too. The catalog derives `lists` (`list_fields`, `dw/for_each.py`): - the fields an entry takes are the `item:` references the steps make, `name` - first; an entry key no step reads is a validation warning - (`entry_field_warnings`). A `cost` entry may carry `per_entry` - (`{variable, minutes, entries}`), measured, never derived. An empty - `for_each` list is an error; `expanded_definition` realizes constants first. -- Every run directory holds `workflow.json` beside its manifest: the *realized* - workflow, with the run's arguments folded into the variable defaults, the seed - it used, stored prompt text inlined and `output:.../latest/...` pinned to the - run it resolved to. Written by `realize_workflow` (`dw/realize.py`) at run - start, best effort. Over MCP, `get_job_workflow` reads it back and - `save_workflow` names it; `export_job` bundles the run - -The same conventions, written for an agent composing a workflow over MCP, are -the `Authoring a workflow from an agent` section of docs/WORKFLOW_GUIDE.md; -change both when one changes. - -### LTX-2.5 IC-LoRAs - -The catalog's IC-LoRA templates are two upscalers, -`templates/ltx2/generative-upscale` and `upscale-clip`, and three -conditioning templates, all through `LTX2InContextPipeline` + -`LTX2ReferenceCondition`; the three run at `reference_downscale_factor: 1` (the -upscalers' is 2). `upscale-clip` runs the upscaler over the caller's own -`source_video` and then `pair_audio`s that file's soundtrack back onto the -upscale, so the `final/` deliverable carries the original track (a silent -source fails there, after the upscale is saved in `intermediate/`). -`reference-sheet` drives Ingredients — the family's only identity route, and -one of the templates here whose reference is a file the workflow did not make; the sheet -is a still, so a `loop_frames` step (`dw/tasks/video_utils.py`, the video -analogue of `loop_audio`) laps it into the static video the LoRA reads -through its 121-frame bucket. `restore-deblur` and `restore-decompression` -each invert one defect and no other. Every number in the three is the vendor -card's and is pinned by `tests/test_ltx2_ic_loras.py`; the trained caption -form is a *different* genre from a T2V shot caption, so those stored prompts -are tagged `ic-lora` and `tests/test_ltx_prompt_library.py` checks them -against their own convention rather than the 150-220-word paragraph rule. -The weights are `gated: auto` on Hugging Face — per repo, so a box that pulls -one can still 403 on another. A `loras` entry counts toward -`plan.downloads_required` (`_collect_sources`, `dw/plan.py`): it names its repo -under `model_name` directly rather than through `from_pretrained_arguments`, and -a walk that skipped it would report `[]` on a box missing only the IC-LoRA, -which then downloads mid-run. - -### Quantization Support - -Quantization configs are defined per-component in workflow JSON and instantiated in `dw/pipeline_processors/config_objects.py`. Supported frameworks: BitsAndBytes, TorchAO, GGUF, SDNQ, optimum-quanto. The `config_type` field is a free-form string — new quantization backends work automatically via dynamic import. - -SDNQ pre-quantized models use a different pattern: `pre_load_modules` imports sdnq (registers with diffusers), then the entire pipeline loads from the pre-quantized repo. Optional `sdnq_optimize` applies quantized matmul post-load (CUDA/XPU only). - -### Cross-Platform Device Support - -`dw/__init__.py` handles device detection (CUDA > MPS > CPU) and platform-specific optimizations: -- **CUDA**: TF32 matmul, cuDNN benchmark, deterministic mode (configurable via settings) -- **MPS**: `PYTORCH_MPS_HIGH_WATERMARK_RATIO=0.0` (use all unified memory), autocast warnings suppressed; attention slicing automatic (faster for SD 1.5's head dims, 2.4x slower for SDXL's - `disable_attention_slicing`) -- **CPU**: Warning displayed - -Detection is overridden by the `DW_DEVICE` environment variable (single run) or the `device` setting (standing), either of which can name a specific accelerator such as `cuda:1`. Device placement is explicit throughout — no default torch device is set, since that would build models directly in VRAM and defeat offloading. Compare backends with `get_device_type()` rather than `== "cuda"`, which a device like `cuda:1` would fail. - -A step can override the device it runs on: `device` in a pipeline `configuration` (also the default for that pipeline's components), in a component `configuration`, or in a task's `arguments`. - -Every device a workflow names passes through `resolve_device()`, which translates a backend this machine does not have into the one it does and warns — a `cuda` workflow runs on a Mac and an `mps` one runs on a CUDA box. Only the backend is translated: an index survives when the backend matches (`cuda:1` on a single-GPU CUDA box stays a genuine error) and is dropped when it does not. `cpu` is never rewritten, since pinning a step to the CPU is how a GPU-specific problem gets ruled out. Translation happens before anything reads the backend, so the MPS accommodations (the sequential-offload downgrade, attention slicing, the compile skip) fire for a translated device too. - -The same translation reaches the settings that carry a device or a CUDA-only feature, so a template written on the CUDA box runs unchanged on a Mac and a caller (an MCP agent included) never has to know the backend: SDNQ `quantization_device`/`return_device` go through `resolve_device()`, and `use_quantized_matmul(_conv)` is turned off on MPS, where it falls back to `torch._int_mm`, ~500x slower (`portable_quantization_arguments`, `config_objects.py`). Group-offload `use_stream`/`record_stream` are dropped when the onload device is not CUDA/XPU. `vram_estimate` checks against the serving device's own `cost` entries, else its `device_capacity_gb()` (Metal's recommended working set on a Mac), else every entry. `device_memory_stats()` reports real unified-memory figures on MPS. Each adaptation logs a warning, and torch's MPS CPU-fallback warning is let through the blanket `UserWarning` filter. `HF_ENABLE_PARALLEL_LOADING` defaults to off on macOS (`_parallel_loading_default`): diffusers' per-shard loader threads copying onto MPS at once segfaulted the LTX-2.5 SDNQ transformer load; it is chosen by platform because diffusers reads it at import, before dw can ask torch for a device. - -A `components` entry can additionally set `residency: "on_demand"`, which rests the component on the CPU and wraps its `forward`/`encode`/`decode` to move it to the device around each call (`apply_on_demand_placement` in `dw/pipeline_processors/placement.py`). The wrappers use `functools.wraps` because callers introspect the signature — MiniMax H3's denoiser picks its arguments from `signature(transformer.forward)`. It is mutually exclusive with `group_offload` on the same component, and like `group_offload` it suppresses the wholesale `pipeline.to(device)` at load. - -Settings in `~/.diffusers_helper/settings.json` (`dw/settings.py`): `device`, `workspace`, `output_layout`, `public_url`, `enable_tf32`, `cudnn_benchmark`, `cudnn_deterministic`, `log_level`, `log_filename`. +pinned by `tests/test_plugin_skills.py`; that README says how the rest is pinned and +versioned. ## Security Rules @@ -258,417 +91,18 @@ All entry points use `dw/security.py`'s validators (paths, URLs, subprocess argu - **Never** use `eval()`, `exec()`, or `shell=True` - Path traversal (`../`) is blocked -CodeQL knows about these validators, which is why the scan is quiet: the local -query pack in `.github/codeql/dw-security/` models them as sanitizers for -`py/path-injection`, because the built-in query recognizes a -normalize-then-check only as a local barrier guard and so cannot see one that -lives in another module and returns the safe value. This is what makes code -scanning useful here rather than 26 identical false positives - but it only -holds while new filesystem access goes through a validator. Reaching the disk -some other way is a real alert, so treat one as a finding rather than as more -of the old noise. `validate_path(path, base)` is modeled as a barrier only when -`base` is not `None`; with `None` it is normalization only, and the path stays -reportable. Scanning is advanced setup (`.github/workflows/codeql.yml`) for the -same reason - default setup cannot load a pack. - -## Critical Gotchas - -- **Schema validation runs before variable substitution** — variable defaults must match expected JSON types (use `25` not `"25"` for numbers) -- **`previous_result:` references are checked statically too** — once the schema - passes, `previous_result_reference_errors` (`dw/previous_results.py`) reports any - literal `previous_result:` or `from_previous_result` naming no *earlier* step, with - the JSON path it sits at. References otherwise resolve lazily per step, so without - the check a step renamed in one place and not another would fail only when the run - reached it, after every step before it had generated. The definition is substituted before the check, - so a reference spelled by a *declared* variable is checked by its value; one spelled - by an undeclared variable is itself a validation error (below) -- **`for_each` expands before the reference check** — `validation_errors` substitutes - (the caller's `arguments` when they are all good, else the defaults) and expands - first, so `gather:` and `item:` errors carry the path of the template step - (`steps[0].for_each[1].name`). Expansion records each expanded step's *source* index - (`expand_for_each(definition, source_indices)`), so a reference error always carries a - path in the file the author wrote, and one inside a member names the member in its - message. An undeclared `variable:` is a validation error at the path it sits at, not a - warning and not a complaint about the `for_each` list that did substitute: once a - `variables` block exists, `replace_variables` refuses an undeclared reference, so it is - a run that cannot start -- **The two MiniMax cut templates take one `shots` list** — - `templates/minimax/dialogue-short` and `music-video` have no per-shot - variables; a scripted caller passes `shots` (entries - `{name, prompt, references, num_frames}` and `{name, prompt, start_frame}`). - The members are `shot@` in the manifest and the gallery. The CLI only - takes `name=value` strings, and a string handed to a list variable is - comma-split - so `shots` can only be supplied over the API/MCP (a JSON - body); `python -m dw.run` runs the templates' default list -- **A reference name is checked for its shape before the queue, and `@` is - part of it** — a `for_each` member is `@` and the files it - writes carry that `@` in their base name, so `OUTPUT_REFERENCE_PATTERN` - and `ASSET_REFERENCE_PATTERN` accept it: every file the server names can be - named back to it. `@` is safe in a path — not a separator, not `..`, and - containment is still `validate_path`'s — but a name may not *start* with - one. `reference_name_errors` (`dw/reference_names.py`) checks the shape of - every `asset:`/`prompt:`/`output:` reference in the definition in - `validation_errors`, so a bad name is refused before the queue, and - `_name_fault` (`dw/security.py`) names the offending character and - position. Shape only — *existence* depends on the workspace and on what - pruning has taken: the validate route resolves the caller's `arguments` - against the workspace, and the definition's own references resolve at run - time -- **Cartesian product explosion** — multiple `previous_result` references multiply: 4 images × 3 masks = 12 iterations -- **Component sharing requires exact key matching** between `shared_components` and `reused_components` -- **Built-in workflows** need explicit argument mapping: `"prompt": "variable:prompt"` -- **MPS differences from CUDA**: no bitsandbytes, no flash_attn, no triton, no torch.compile (`torch.autocast("mps")` works on torch 2.14, but dw does not use it). Model offloading has less benefit on unified memory, and `"offload": "sequential"` is downgraded to `"model"` with a warning there (`place_component`, `dw/pipeline_processors/placement.py`) — per-submodule streaming hands back no residency when the CPU and the accelerator share one pool. `exclude_from_cpu_offload` is sequential-only and does not survive the downgrade. -- **`{}`-escaped strings** in JSON arguments: `"{nf4}"` stays as string `"nf4"`, without braces it would try to load as a type -- **A stored prompt's `text` may not begin with a reference prefix** (`variable:`, `previous_result:`, `constant:`, `asset:`, `output:`, `prompt:`) — the engine rejects it to prevent double resolution or iteration expansion -- **Audio+video muxing**: pipelines that generate audio alongside video (LTX-2) have the two muxed into one `video/mp4` file with PyAV in `result.py` -- **A caller's `arguments` are checked before anything is queued** - - `argument_errors` (`dw/variables.py`) folds them into the declared variables - exactly as `set_variables` does at the top of a run, so an undeclared name or - a value that will not coerce is a 400 from `POST /api/jobs` rather than a - job that fails on its first step, and `POST /api/validate` takes the same - `arguments` (plus an `asset:`/`prompt:`/`output:` existence check against - the workspace) so the free pre-flight covers the part the caller wrote. - A workflow that declares no variables takes no arguments at all, since - `Workflow.run` only substitutes when a `variables` block exists - any passed - would be dropped in silence. A valid `POST /api/validate` answer also carries `plan` - (`dw/plan.py`): the fingerprint of the work, step and list counts, - `downloads_required` and a cost `estimate` with its `basis` - the number an - agent quotes, with `basis` saying whether it was measured for this list - (`catalog`/`per_entry`) or extrapolated over one the caller resized - (`derived`); `plan: null` when it could not be built, never a changed - verdict. `acknowledged_cost` on `POST /api/jobs` / `rerun` takes `true` - (recorded) or the plan's `{fingerprint, minutes, downloads}` (checked - 409 - with the current plan when the fingerprint or the required downloads - changed; `minutes` never compared), and the job records `acknowledged: - none | boolean | bound`. `cached_steps` is the worker's answer to a - `probe_cache` command (`workflow_run.cache_hits`, which shares - `prepare_definition` / `cache_lookup` (`dw/workflow_run.py`) with `run` so the two cannot drift). - The web UI reads the fields only: the editor lists the plan under a valid - verdict (`describePlan`, `ui/src/lib/plan.ts`), and a job queued `bound` - says so on the job page and in the jobs list; the UI itself sends no - acknowledgement -- **A failed run still reports what it wrote** — the worker carries its partial - manifest on the error and cancelled messages as well as on success, and the - "Previous result not found" error names the steps that ran even after - `release_unreferenced_results` (`dw/workflow_run.py`) has dropped their results -- **Run directories**: each execution writes `///` - with a `manifest.json` beside its files (`dw/runs.py`, `Workflow.effective_output_dir`). - Identity is the workflow's path under a `workflows/` tree, else its file name, else its - `id`; the run id is `-<8 hex of the spec>`, with a `-N` counter if taken. - A sub-workflow inherits the parent's run directory and writes no manifest of its own. - `--output-layout flat` / `DW_OUTPUT_LAYOUT` / the `output_layout` setting writes - straight into the output directory with no run directories. The gallery groups a workflow's runs under one folder by stripping the run - id (`strip_run_id`). The realized workflow is written into the same directory as - `workflow.json` (`dw/realize.py`, `write_realized_workflow`), and the manifest's - `workflow` block carries `realized`, `prompts` (the stored prompts inlined) and - `sub_workflows` (path -> SHA-256). A job records the run it was - (`run_id`/`run_dir` on `Job` and in `jobs.sqlite`), which is how - `JobManager.realized` finds the file. `exports` is a reserved workspace name: - `POST /api/jobs/{id}/export` gathers one finished job into - `/exports//` and `GET /exports/.zip` streams it. -- **A run has a number, and it is not derived from the listing** - a file's name - is per *step*, so four runs of one workflow write four files with the same - name; the run id tells them apart but is not something anyone says out - loud, so the number is how an agent names one of them to a person. Every run - takes an ordinal, `open_run` (`dw/runs.py`) at the moment - `Workflow.run` opens the run directory, recorded as `version` in - `manifest.json` and read back by `run_versions`. Assigned once and never - recomputed, which is the point: deleting a middle run leaves a gap rather - than sliding every later number down, so "version 5" still means the same - run tomorrow. Assignment is `max(recorded) + 1` over *every* sibling - manifest, not one past the newest - run ids are chronological only to the - second, and within one second the spec digest decides the sort, which is - exactly what three quick reruns hit. The number is on disk from the moment - the run opens - a `status: "running"` manifest is written before the first - step and rewritten in full at the end - so a hard kill does not lose it and - a second process opening a run of the same workflow sees it. A run with no - recorded number (made before the field, or killed before even that first - manifest) is ranked: the unrecorded runs older than every recorded one take - the numbers beneath the lowest, later ones continue from the highest before - them. A ranked number would move when an older sibling is deleted, so - `record_run_versions` writes it into the manifest on the two write paths - - a run opening and a run directory being deleted; the listing never writes, - and a run with no manifest at all is left ranked. A gap in the numbers is - not only a deletion: a failed run or a fully cached rerun takes a number - and may have no media for the gallery to show under it. `GET - /api/gallery` and the metadata route carry `version` and `run_id` - (`run_versions` read once per identity per listing, not per file), and - `?folder=&version=` lists one run's files. The number is also a name: - `output:/v4/`. The `run_start` event carries it, the job - records it (`run_version`, a `jobs.sqlite` column) and the export README and - zip download name (`-v4-.zip`) carry it too. MCP - `list_gallery` teaches the vocabulary and takes `folder`/`version`, and the - web UI reads the field only - a `v4` chip on the gallery card, the jobs list - and the job page, the run id in the gallery's detail pane. Nothing on disk is - renamed, so `output:` references, the step cache and `keep_output` are - untouched. Two limits taken deliberately: deleting the *newest* run frees - its number for reuse (the high-water mark lived in the manifest that went - with it), and the flat layout has no runs, so `version` is null there. -- **Result subfolders**: a step's `result.subfolder` (`dw/subfolders.py`) puts its files - in a subfolder of the run directory - `/final/x.mp4` - by convention `final` or - `intermediate`; the engine treats no name specially and there is no default. - `Workflow.step_output_dir` computes the directory once and hands it to both - `Result.save` and the pipeline wrapper, so a chain's `save_segments` spill follows it. - Shape is `SUBFOLDER_PATTERN` (the `output:` segment rule, so a subfolder is - `output:`-addressable up to `OUTPUT_REFERENCE_PATTERN`'s seven-segment ceiling), - checked by `subfolder_errors` in `validation_errors` after - `for_each` expansion and again at run time; containment is `validate_output_path` - against the run directory. Manifest entries and `step_end` carry `subfolder`. - `split_run_path` finds the run id anywhere in a path, so `strip_run_id` still groups a - workflow's runs. Gallery entries carry it too; `GET /api/gallery?subfolder=` and MCP - `list_gallery(subfolder=)` filter on it. The web UI reads the field only: - the gallery page offers a subfolder pick once any entry has one, and the - job page sections results under `final/` / `intermediate/` headings (or - whatever the step named) (`sectionBySubfolder`, `ui/src/lib/results.ts`), - unchanged for a run that chose none. `file_base_name` may not contain a - separator - it is a name, not a path - and it *replaces* the derived - `-.` base rather than prefixing it, so - two steps in one subfolder that set the same one collide onto - `output_file_path`'s `-2` counter. - Every `workflows/templates/**` file saves at least one step and - marks each saving step `final`/`intermediate`, at least one `final` (`tests/test_template_subfolders.py` pins the rule; - `dw/workflows/` builtins stay unmarked - a role is the parent's to assign). A - template's outputs land in `/final/` and `/intermediate/`, so gallery names - read `