From 2b2b82ce0b0d64b4e335925df9bdf7f9c8f8319c Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 19:31:43 -0500 Subject: [PATCH 001/140] chore: open 0.8.0-alpha.1 on develop --- plugins/dw/.claude-plugin/plugin.json | 2 +- pyproject.toml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/plugins/dw/.claude-plugin/plugin.json b/plugins/dw/.claude-plugin/plugin.json index 53f4787b..c4915d5b 100644 --- a/plugins/dw/.claude-plugin/plugin.json +++ b/plugins/dw/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "dw", "description": "Compose MiniMax H3 video, MiniMax Music 3 and LTX-2.5 workflows over a dw MCP server, and cut a multi-episode series from them: which template fits which shape, the hard rules, cost, and how to judge the output. Prompt format comes from the vendors' own guides.", - "version": "0.7.0", + "version": "0.8.0-alpha.1", "author": { "name": "Don Kackman" }, diff --git a/pyproject.toml b/pyproject.toml index 9d088ba7..d1173ddc 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -8,7 +8,7 @@ name = "diffusers-workflow" # runtime (static in TOML because importing dw at build time would drag # torch into the build environment). Release tags must match it - see # docs/RELEASING.md -version = "0.7.0" +version = "0.8.0-alpha.1" description = "Declarative workflow engine and web UI for Hugging Face Diffusers" readme = "README.md" requires-python = ">=3.10" From d546d87c44e64e68baf3087453168e6d757c9da2 Mon Sep 17 00:00:00 2001 From: Don Kackman Date: Thu, 1 Oct 2026 19:53:44 -0500 Subject: [PATCH 002/140] docs(stabilization): dw_mcp assessment and UI scope The gates measured dw_mcp and ui/src but never read them. Records both surveys, and that every gate's UI SLOC counted only .ts (pygount reads .svelte as 0 lines). Co-Authored-By: Claude Opus 5.5 --- docs/stabilization/ROADMAP.md | 8 +++ docs/stabilization/mcp-assessment.md | 63 ++++++++++++++++++ docs/stabilization/ui-scope.md | 98 ++++++++++++++++++++++++++++ 3 files changed, 169 insertions(+) create mode 100644 docs/stabilization/mcp-assessment.md create mode 100644 docs/stabilization/ui-scope.md diff --git a/docs/stabilization/ROADMAP.md b/docs/stabilization/ROADMAP.md index 268357e8..06b505ec 100644 --- a/docs/stabilization/ROADMAP.md +++ b/docs/stabilization/ROADMAP.md @@ -821,6 +821,14 @@ Regenerated by `scripts/arch_report.py` at Phase 1 Task 2. The complexity figure | 3096 | 172 | 18 | dw/server/app.py | | 3024 | 36 | 84 | dw/worker.py | +## After gate 4: the API's client and the UI + +The gates measured `dw_mcp` and `ui/src` but never read them. Two follow-up +surveys do: [mcp-assessment.md](mcp-assessment.md) and +[ui-scope.md](ui-scope.md). Every gate's UI SLOC row above counts only `.ts`: +pygount has no lexer for `.svelte` and reports those files as 0 lines, so +`ui/src` is about 15,500 raw lines, not the ~2,270 code lines shown. + ## Working rules for the duration These held from gate 0 to gate 4. `FREEZE` was deleted at gate 4 (`5623dd03`); the hot zone, the ratchet and the new-module rule outlive it in the harness's stage C. diff --git a/docs/stabilization/mcp-assessment.md b/docs/stabilization/mcp-assessment.md new file mode 100644 index 00000000..9a21ff01 --- /dev/null +++ b/docs/stabilization/mcp-assessment.md @@ -0,0 +1,63 @@ +# dw_mcp assessment (2026-10-01, develop 2b2b82ce) + +The engine stabilization (gates 0-4) measured `dw_mcp` but never read it +module by module; the 2026-09-28 assessment's whole verdict was "correctly a +thin HTTP client; only the shared-session defect (B6)". This is that read: +all 18 modules (4,597 lines), against the owners the stabilization put in +`dw/`. + +## Verdict + +Mostly true. Nothing in `dw_mcp` parses reference prefixes, walks a +definition, does cost or VRAM math, computes run versions or handles sample +rates: those all come from the server. But `dw_mcp` cannot import `dw` +(it stays torch-free), so every rule it does know is a second copy, and it +knows about twenty. All of them agree with their owners today and only one +is pinned by a test; `tests/test_mcp_*.py` run against `httpx.MockTransport`, +so they test `dw_mcp`'s idea of the server, not the server. The one real +bug is the stabilization's pattern in miniature: a fix landed in one of two +twin functions. + +## Findings + +| # | Severity | Finding | Evidence | Fix | +| --- | --- | --- | --- | --- | +| M1 | bug, medium | #389's per-call `workspace` reached `media._remote_root` but not its twin `assets._remote_roots`. A mounted session pinned to `ep4` calling `upload_asset(file_path="/assets/x.wav", workspace="default")` is confined to ep4's roots and refused. Fails closed. | `dw_mcp/assets.py:52`, `:353`; `dw_mcp/media.py:494` | Pass `workspace` through; merge the two root/confine pairs (M8) | +| M2 | structure, medium | `dw` imports `dw_mcp`: `dw/run.py` is a client of `dw.serve` through `dw_mcp.client`, and keeps a third `TERMINAL_STATUSES`. `dw/media.py:37` says the packages do not import each other. No seam-map row. | `dw/run.py:17`, `:36` | Either a map row naming the edge, or move the HTTP client to a torch-free module both import | +| M3 | duplication, medium | Image and frame handling copied from the server: `_crop_box` clones `dw/media_frames.resolve_crop_box`; the one-selector check and `_fit` repeat `dw/server/routes/media.py`; `MIN_DIMENSION`, `MAX_RETURNED_BYTES` twin server constants. `get_output_image` downloads and decodes the whole file client-side; `_fit_tiles_within_budget` re-encodes tiles the server already made. | `dw_mcp/media.py:96`, `:136`, `:274`, `:364` | A server image route with `max_dimension`/`crop`/`max_bytes` and a byte budget on `/frames`; `media.py` becomes call-and-reshape, and the UI can use the same routes | +| M4 | duplication, medium | `get_gallery_metadata` writes audio-QC thresholds (-0.5 / 0.0 / -40 dBFS) owned by `dw/audio_qc.py`, and the Music 3 "within 0.2 s of the ceiling" rule owned by `plugins/dw/skills/minimax-music3/SKILL.md` - model knowledge outside plugins and the catalog. | `dw_mcp/catalog.py:304-361` | The metadata route emits its findings and next step, as assess already does | +| M5 | duplication, low | Twin constants, none pinned except `MAX_DECODE_PIXELS`: upload extensions and size cap (`routes/assets.py`), `LOOPBACK_HOSTS` (`server/netinfo.py`), `TERMINAL_STATUSES` (`job_record.py`), `DEFAULT_WORKSPACE` (`workspace.py`), `ASSESSMENT_PROBES` (`server/assess.py`; redundant, the server returns the same 400). Three owner comments name `dw/server/app.py` or `dw/security.py`, which no longer own them. | `dw_mcp/assets.py:18-31`, `client.py:17`, `:130`, `diagnose.py:15-18`, `media.py:426` | One test (tests may import `dw`) asserting each twin equals its owner; delete `ASSESSMENT_PROBES`; fix the comments | +| M6 | duplication, low | Engine word lists in tool text: shapes and traits, reserved prompt-text prefixes (`references.RESERVED_TEXT`), reserved workspace names, estimate bases, `move_job` directions, job states. A test checks some words appear, not that the lists match. | `dw_mcp/server.py:65`, `tools_catalog.py:26`, `:251`, `tools_authoring.py:36`, `:76`, `:192`, `tools_jobs.py:201` | Pin the rendered descriptions against the owners' tuples | +| M7 | misplaced logic, low-medium | Work the server should own: `save_workflow` patch mode is GET + merge + PUT, not atomic, so a UI save in between is lost; `delete_output(job_id=)` derives the run dir; the acknowledgement body strips null repos because admission types `downloads: List[str]`; the 409 re-acknowledge body is rebuilt client-side; `workspaces.server_info` re-derives directories `/api/server` already scopes. | `dw_mcp/authoring.py:96`, `media.py:477`, `diagnose.py:57`, `client.py:426`, `workspaces.py:195` | A `PATCH` workflow route; `DELETE /api/jobs/{id}/run`; the server tolerates nulls and sends the body in its 409; delete `server_info`'s re-derivation | +| M8 | structure, low | Duplicates inside `dw_mcp`: the root/confine pairs in `media.py` and `assets.py` (source of M1); three field-projection helpers; the base64-size formula three times; two loopback sets that differ; two copies of the name/path/inline alias parsing; the acknowledgement gate in six places while the map's *Spending needs consent* row names only `diagnose.py`. `diagnose.py` is mostly queue operations. | `dw_mcp/media.py:523`, `assets.py:118`, `:164`, `authoring.py:39`, `diagnose.py:104` | Consolidate each pair; widen the map row to every gated tool | +| M9 | defects, low | `get_output_frames` with `hear` budgets frames and audio separately, so a reply can reach about 2x the cap; `_upload_inline` decodes before checking the size cap; `_probe` matches "401" in text, not `status_code`; a bad `DW_MCP_MAX_WAIT_SECONDS` fails the import; `get_gallery_metadata` reads `job["id"]` unguarded. | `dw_mcp/media.py:317`, `:341`, `assets.py:414`, `__main__.py:112`, `diagnose.py:27`, `catalog.py:312` | Each a one-line fix with a test | + +Complexity over 10 (ruff C901): `media.get_output_frames` 16, +`client._format_detail` 13, `assets._remote_roots` 11, +`tools_media.get_output_frames` 11. None is over the ratchet's limit. + +## What is already sound + +- `dw_mcp` stays torch-free and only `server.py` and `tools_*.py` import the + MCP SDK, both pinned (`docs/ARCHITECTURE.md`, *MCP*). +- B6, the shared session pin on the mounted surface, is documented as + deliberate and warns on change; a per-call `workspace` reaches every tool + but M1's. +- Drift that is caught today: `MAX_DECODE_PIXELS` + (`tests/test_security_decoder_bombs.py`), the tool listing over the real + mount (`tests/test_server_mcp.py`), symlink confinement through the real + app (`tests/test_security_symlinks.py`). + +## Proposed pass + +Three stages, each small. The ratchet already covers `dw_mcp`, so none needs +new tooling. + +1. **Fixes and pins.** M1, M9; the twin-constant test and word-list pin + (M5, M6); stale comments; the M2 map row; widen the consent row (M8). + No server change. +2. **Move logic to the server.** M3, M4, M7: new or widened routes, then + `dw_mcp` calls them. The UI is the second consumer, so this stage is + shared with the UI pass. +3. **Consolidate inside `dw_mcp`.** M8's pairs, once stage 2 has removed the + code some of them guard. diff --git a/docs/stabilization/ui-scope.md b/docs/stabilization/ui-scope.md new file mode 100644 index 00000000..a315b32b --- /dev/null +++ b/docs/stabilization/ui-scope.md @@ -0,0 +1,98 @@ +# UI stabilization scope (2026-10-01, develop 2b2b82ce) + +The engine stabilization (gates 0-4) never read `ui/` and has no ratchet over +it. This is the survey that scopes a pass: what is there, where it already +disagrees with the engine, and what tooling a ratchet needs. + +## The gate reports undercounted the UI + +Every gate's *SLOC by layer* row says the UI is about 2,270 code lines. +`scripts/arch_report.py` lists `.svelte` in `SLOC_SUFFIXES`, but pygount has +no lexer for it and returns `SourceState.unknown` with 0 lines, so only the +`.ts` files were counted. `ui/src` without tests is **15,508 raw lines**: +11,623 in `.svelte`, the rest `.ts` and `app.css`, over three times +`dw_mcp`'s 4,597. The measurement needs fixing before any UI baseline is taken. + +## What is there + +- **Tooling is good and green.** ESLint (typescript-eslint, eslint-plugin-svelte, + recommended presets only), svelte-check (0 errors, 0 warnings), vitest (36 + files, 340 tests), prettier, Playwright (7 specs). CI's `ui` job runs + prettier, lint, check, test and build; `e2e` runs only on PRs into master. + `tsconfig.app.json` is `strict`. +- **One API client.** Every `fetch` and `EventSource` is in `ui/src/lib/api.ts`. + Response types are hand-written in `types.ts` (379 lines), and nothing checks + them against the server: no route declares a `response_model`, so the + OpenAPI document has no response schemas to generate from. +- **Layering holds.** `App` imports pages, pages import components, everything + imports `lib/*.ts`; no `.ts` imports a `.svelte` but `main.ts`. One import + cycle: `lib/api.ts` <-> `lib/workspace.svelte.ts`. +- **Size.** 14 non-test files over 400 lines: `EditorPage.svelte` 1,075, + `PromptEditorPage.svelte` 954, `JobPage.svelte` 855, `editor/StepEditor.svelte` + 802, `api.ts` 671, then `AssetsPage`, `GalleryPage`, `ModelsPage`, + `WorkflowsPage`, `App`, `app.css`, `FlowView`, `OverviewPage`, `ServerPage`. + In the big pages about 30% is `\n" + ) + counts = _load().sloc(tmp_path) + assert counts["UI (ui/src, tests excluded)"] == 7 +``` + +The seven: ``, `

...`, ``. + +- [ ] **Step 2: Run it to see it fail** + +Run: `venv/bin/python -m pytest tests/test_arch_report.py -q -k svelte` +Expected: FAIL, `0 == 7`. + +- [ ] **Step 3: Count `.svelte` by hand** + +In `scripts/arch_report.py`, add above `sloc`: + +```python +# A line that is only a comment. A multi-line /* */ comment's inner lines +# count as code: matching them by a leading '*' would also match a CSS '*' +# selector, and an undercount hides growth where an overcount does not. +SVELTE_COMMENT = re.compile(r"^(|//.*|/\*.*\*/)$") + + +def svelte_code_lines(path): + """Code lines in a .svelte file, which pygount reads as 0: every line + that is neither blank nor a comment on its own. Script, markup and style + all count, since all three are what a reader holds.""" + return sum( + 1 + for line in path.read_text(encoding="utf-8").splitlines() + if line.strip() and not SVELTE_COMMENT.match(line.strip()) + ) +``` + +and in `sloc`, replace the `counts[layer] += ...` line with: + +```python + if path.suffix == ".svelte": + counts[layer] += svelte_code_lines(path) + else: + counts[layer] += SourceAnalysis.from_file( + str(path), layer + ).code_count +``` + +Add `import re` if the module does not already import it. + +- [ ] **Step 4: Run the test and the file** + +Run: `venv/bin/python -m pytest tests/test_arch_report.py -q` +Expected: PASS. + +- [ ] **Step 5: Record the corrected number** + +Run: `venv/bin/python scripts/arch_report.py 2>/dev/null | grep -A4 "SLOC by layer"` (or the script's documented invocation for the current tree). In `docs/stabilization/ROADMAP.md`'s *After gate 4* section, add one sentence giving the corrected UI code-line count from this run, measured the same way the API row is. + +- [ ] **Step 6: Commit** + +```bash +git add scripts/arch_report.py tests/test_arch_report.py docs/stabilization/ROADMAP.md +git commit -m "fix(stabilization): the gate report counts .svelte code lines" +``` + +--- + +### Task 3: One speller of reference prefixes; `isReference` knows them all (U2) + +`isReference` knows 4 of the engine's prefixes. A value like `item:flag` on +a parameter annotated `bool` gets a checkbox, and toggling it replaces the +reference with `true`/`false`. + +**Files:** +- Create: `ui/src/lib/references.ts` +- Modify: `ui/src/lib/editor.ts` (`isReference` moves out; re-exported) +- Modify: `ui/src/lib/runstate.ts` (imports `MEMBER_SEPARATOR`) +- Test: `ui/src/lib/editor.test.ts` +- Create: `tests/test_ui_twins.py` + +**Interfaces:** +- Produces: from `ui/src/lib/references.ts`: `ASSET`, `OUTPUT`, `PROMPT`, `VARIABLE`, `PREVIOUS_RESULT`, `CONSTANT`, `ITEM`, `GATHER`, `BUILTIN`, `CONSTRAINT` (each `':'`), `PREFIXES` (all ten), `MEMBER_SEPARATOR` (`'@'`), `FROM_PREVIOUS_RESULT_KEY` (`'from_previous_result'`), `isReference(value: unknown): value is string`. `editor.ts` keeps exporting `isReference` (re-export) so existing imports hold. +- Produces: `tests/test_ui_twins.py` helpers `ts_constants(path) -> dict[str, str]` and `ts_string_array(path, name) -> list[str]`, used by Tasks 5, 7 and 8. + +- [ ] **Step 1: Write the failing UI tests** + +Append to `ui/src/lib/editor.test.ts`: + +```ts +describe('a reference is always edited as text', () => { + const boolParam = param({ annotation: 'bool' }) + const intParam = param({ annotation: 'int' }) + it.each([ + 'item:flag', + 'gather:shots', + 'asset:cast/priya.png', + 'output:run/v2/final.mp4', + ])('%s', (value) => { + expect(isReference(value)).toBe(true) + expect(widgetFor(boolParam, value)).toBe('text') + expect(widgetFor(intParam, value)).toBe('text') + }) +}) +``` + +- [ ] **Step 2: Write the failing twin test** + +Create `tests/test_ui_twins.py`: + +```python +"""The UI's copies of rules the engine owns, pinned to their owners. + +The UI cannot import Python, so a rule it must know is a copy. These tests +read the TypeScript source and compare each copy with its owner: a change +to one side fails until the other follows. A copy that can be read from the +server instead is deleted, not pinned. +""" + +import pathlib +import re + +from dw import references + +REPO = pathlib.Path(__file__).resolve().parent.parent +UI_LIB = REPO / "ui" / "src" / "lib" + + +def ts_constants(path): + """`export const NAME = 'value'` string constants in a TS file.""" + return dict( + re.findall(r"^export const ([A-Z_]+) = '([^']*)'", path.read_text(), re.M) + ) + + +def ts_string_array(path, name): + """The quoted strings of `export const NAME = [ ... ]` in a TS file. + Exported only: a module-private copy is not the one other modules use.""" + found = re.search( + rf"^export const {name}\b[^=]*= \[(.*?)\]", path.read_text(), re.M | re.S + ) + assert found, f"no array {name} in {path}" + return re.findall(r"'([^']*)'", found.group(1)) + + +def test_the_ui_spells_every_reference_prefix_the_engine_does(): + engine = { + name: value + for name, value in vars(references).items() + if name.isupper() + and isinstance(value, str) + and re.fullmatch(r"[a-z_]+:", value) + } + ui = { + name: value + for name, value in ts_constants(UI_LIB / "references.ts").items() + if value.endswith(":") + } + assert ui == engine + + +def test_the_member_separator_and_reference_key_are_the_engines(): + ui = ts_constants(UI_LIB / "references.ts") + assert ui["MEMBER_SEPARATOR"] == references.MEMBER_SEPARATOR + assert ui["FROM_PREVIOUS_RESULT_KEY"] == references.FROM_PREVIOUS_RESULT_KEY +``` + +- [ ] **Step 3: Run both to see them fail** + +Run: `cd ui && npx vitest run src/lib/editor.test.ts` then `venv/bin/python -m pytest tests/test_ui_twins.py -q` +Expected: the four `item:`/`gather:`/`asset:`/`output:` cases FAIL; the twin tests FAIL (no `references.ts`). + +- [ ] **Step 4: Create the owner** + +`ui/src/lib/references.ts`: + +```ts +/** The engine's reference prefixes and the names that go with them. + * dw/references.py owns them; tests/test_ui_twins.py pins this copy. The + * one module in ui/src that spells a prefix: the UI ratchet's + * prefix_literals counts any other. */ + +export const ASSET = 'asset:' +export const OUTPUT = 'output:' +export const PROMPT = 'prompt:' +export const VARIABLE = 'variable:' +export const PREVIOUS_RESULT = 'previous_result:' +export const CONSTANT = 'constant:' +export const ITEM = 'item:' +export const GATHER = 'gather:' +export const BUILTIN = 'builtin:' +export const CONSTRAINT = 'constraint:' + +export const PREFIXES = [ + ASSET, + OUTPUT, + PROMPT, + VARIABLE, + PREVIOUS_RESULT, + CONSTANT, + ITEM, + GATHER, + BUILTIN, + CONSTRAINT, +] as const + +/** Joins a for_each step's name to an entry's: `@`. */ +export const MEMBER_SEPARATOR = '@' + +/** The key a reference object names an earlier step under, unprefixed. */ +export const FROM_PREVIOUS_RESULT_KEY = 'from_previous_result' + +/** A value the engine resolves later, so it is always edited as text: + * a widget that coerces it (a checkbox, a number) would replace it. */ +export function isReference(value: unknown): value is string { + return ( + typeof value === 'string' && PREFIXES.some((p) => value.startsWith(p)) + ) +} +``` + +- [ ] **Step 5: Point the old homes at it** + +In `ui/src/lib/editor.ts`, delete the `isReference` function and its doc comment, and add near the top imports: + +```ts +import { isReference } from './references' +export { isReference } +``` + +In `ui/src/lib/runstate.ts`, replace `const MEMBER_SEPARATOR = '@'` with `import { MEMBER_SEPARATOR } from './references'` (keep the doc comment above it, which explains the separator's use there). + +- [ ] **Step 6: Run the tests, the ratchet and the checks** + +Run: `cd ui && npx vitest run && npm run check && npm run lint && npm run metrics -- --check ../docs/stabilization/ui/baseline.json` then `venv/bin/python -m pytest tests/test_ui_twins.py -q` +Expected: all PASS; `prefix_literals` falls by 4. + +- [ ] **Step 7: Lower the baseline and commit** + +Run: `cd ui && npm run metrics -- --write ../docs/stabilization/ui/baseline.json` + +```bash +git add ui/src/lib/references.ts ui/src/lib/editor.ts ui/src/lib/runstate.ts ui/src/lib/editor.test.ts tests/test_ui_twins.py docs/stabilization/ui/baseline.json +git commit -m "fix(ui): every reference prefix is edited as text; references.ts owns them" +``` + +--- + +### Task 4: The live reference check agrees with the engine (U1) + +`danglingReferenceDetails` (`ui/src/lib/flow.ts`) is a second copy of +`dw/previous_results.py`'s `previous_result_reference_errors`. It flags a +`for_each` member reference the engine accepts, splits a name on its first +`.` where the engine matches the whole name or the name plus a property, and +never looks at a `from_previous_result` key the engine checks. The engine +checks the expanded definition, where a `for_each` step `g` becomes members +`g@`; the editor holds `g` unexpanded, so it accepts any +`g@...` reference to a `for_each` step. Whether this check should become the +server's validate instead is Phase 1's call; here it only stops disagreeing. + +**Files:** +- Modify: `ui/src/lib/flow.ts` (`danglingReferenceDetails`, a new `resolvesTo`) +- Test: `ui/src/lib/flow.test.ts` + +**Interfaces:** +- Consumes: `PREVIOUS_RESULT`, `VARIABLE`, `GATHER`, `PROMPT`, `MEMBER_SEPARATOR`, `FROM_PREVIOUS_RESULT_KEY` from `./references` (Task 3). +- Produces: `danglingReferenceDetails(workflow, promptNames?)` keeps its signature and `DanglingReference` shape. + +- [ ] **Step 1: Write the failing tests** + +Append inside `describe('danglingReferenceDetails', ...)` in `ui/src/lib/flow.test.ts`: + +```ts + const forEachStep = (name: string) => ({ + ...step(name, { x: 'item:prompt' }), + for_each: [{ name: 'a', prompt: 'p' }], + }) + + it('accepts a member reference to a for_each step, with or without a property', () => { + const wf = { + variables: {}, + steps: [ + forEachStep('g'), + step('b', { one: 'previous_result:g@a', two: 'previous_result:g@a.mask' }), + ], + } + expect(danglingReferenceDetails(wf)).toEqual([]) + }) + + it('flags a member reference to a step that is not for_each', () => { + const wf = { + variables: {}, + steps: [step('g', {}), step('b', { y: 'previous_result:g@a' })], + } + expect(danglingReferenceDetails(wf)).toHaveLength(1) + }) + + it('matches a dotted step name whole, and a property of a plain one', () => { + const wf = { + variables: {}, + steps: [ + step('x.y', {}), + step('seg', {}), + step('b', { v: 'previous_result:x.y', m: 'previous_result:seg.mask' }), + ], + } + expect(danglingReferenceDetails(wf)).toEqual([]) + }) + + it('flags a from_previous_result that names no earlier step', () => { + const wf = { + variables: {}, + steps: [step('b', { image: { from_previous_result: 'nope' } })], + } + const details = danglingReferenceDetails(wf) + expect(details).toHaveLength(1) + expect(details[0].message).toContain('nope') + }) +``` + +- [ ] **Step 2: Run them to see three fail** + +Run: `cd ui && npx vitest run src/lib/flow.test.ts` +Expected: the member, dotted-name and `from_previous_result` cases FAIL; "flags a member reference to a step that is not for_each" passes already. + +- [ ] **Step 3: Rewrite the check** + +In `ui/src/lib/flow.ts`, import from `./references` and replace `danglingReferenceDetails` with: + +```ts +import { + FROM_PREVIOUS_RESULT_KEY, + GATHER, + MEMBER_SEPARATOR, + PREVIOUS_RESULT, + PROMPT, + VARIABLE, +} from './references' + +interface EarlierStep { + name: string + forEach: boolean +} + +/** Whether a previous_result reference resolves to an earlier step: the + * name itself or the name plus a property (`seg.mask`), as the engine's + * reference_resolves_to decides - or, for a for_each step, one of the + * `@` members the engine expands it into, which the editor + * cannot list because it holds the step unexpanded. */ +function resolvesTo(reference: string, step: EarlierStep): boolean { + if (reference === step.name || reference.startsWith(step.name + '.')) + return true + return step.forEach && reference.startsWith(step.name + MEMBER_SEPARATOR) +} + +export function danglingReferenceDetails( + workflow: Record, + promptNames?: string[], +): DanglingReference[] { + const problems: DanglingReference[] = [] + const variables = new Set(Object.keys(workflow.variables ?? {})) + const prompts = promptNames === undefined ? null : new Set(promptNames) + const steps: Array> = workflow.steps ?? [] + + steps.forEach((step, index) => { + const earlier: EarlierStep[] = steps + .slice(0, index) + .filter((s) => typeof s.name === 'string' && s.name) + .map((s) => ({ name: s.name, forEach: s.for_each !== undefined })) + const report = (message: string) => + problems.push({ stepIndex: index, message: `Step '${step.name}': ${message}` }) + + scanStringsWithPath(step, [], (value, path) => { + if (path[path.length - 1] === FROM_PREVIOUS_RESULT_KEY) { + if (!earlier.some((s) => resolvesTo(value, s))) + report(`${FROM_PREVIOUS_RESULT_KEY} '${value}' - no earlier step has that name`) + } else if (value.startsWith(VARIABLE)) { + const name = value.slice(VARIABLE.length) + if (!variables.has(name)) + report(`${VARIABLE}${name} - no such variable is declared`) + } else if (value.startsWith(PREVIOUS_RESULT)) { + const reference = value.slice(PREVIOUS_RESULT.length) + if (!earlier.some((s) => resolvesTo(reference, s))) + report(`${PREVIOUS_RESULT}${reference} - no earlier step has that name`) + } else if (value.startsWith(GATHER)) { + const name = value.slice(GATHER.length) + if (!earlier.some((s) => s.name === name)) + report(`${GATHER}${name} - no earlier step has that name`) + } else if (value.startsWith(PROMPT)) { + // Without a listing the server resolves these at run time - only + // a supplied library can say a name is missing + const name = value.slice(PROMPT.length) + if (prompts !== null && !prompts.has(name)) + report(`${PROMPT}${name} - the prompt library has no such prompt`) + } + }) + }) + return problems +} +``` + +If `scanStrings` is now unused, delete it; `npm run lint` says so. + +- [ ] **Step 4: Run the tests, checks and ratchet** + +Run: `cd ui && npx vitest run && npm run check && npm run lint && npm run metrics -- --check ../docs/stabilization/ui/baseline.json` +Expected: all PASS. The existing messages keep their wording, so the existing flow and EditorPage tests still pass; if one asserted the exact old message for a `previous_result:` with a property, update it to the full reference, which is what the engine names. + +- [ ] **Step 5: Lower the baseline if it fell, and commit** + +```bash +cd ui && npm run metrics -- --write ../docs/stabilization/ui/baseline.json && cd .. +git add ui/src/lib/flow.ts ui/src/lib/flow.test.ts docs/stabilization/ui/baseline.json +git commit -m "fix(ui): the live reference check resolves names as the engine does" +``` + +--- + +### Task 5: Workspace names follow the engine's rule (U3) + +The UI allows `[A-Za-z0-9_][A-Za-z0-9_-]*`; the engine's +`WORKSPACE_NAME_PATTERN` is `^[\w][\w.-]*\Z` with Python's Unicode `\w` +(letters, digits, underscore) and a 100-character cap. One case file is read +by a pytest and a vitest, so the two rules cannot drift without a test +failing. + +**Files:** +- Create: `tests/fixtures/workspace_names.json` +- Modify: `ui/src/lib/workspaceActions.ts` (`NAME`, a length cap) +- Test: `ui/src/lib/workspaceActions.test.ts` +- Test: `tests/test_ui_twins.py` + +**Interfaces:** +- Produces: `workspaceNameError(name: string): string | null` keeps its signature; `MAX_WORKSPACE_NAME_LENGTH` (100) exported from `workspaceActions.ts`. + +- [ ] **Step 1: Write the shared cases** + +`tests/fixtures/workspace_names.json` (the 100- and 101-character names written out in full): + +```json +[ + { "name": "ep4", "valid": true }, + { "name": "ep4.v2", "valid": true }, + { "name": "under_score-dash", "valid": true }, + { "name": "café", "valid": true }, + { "name": "_lead", "valid": true }, + { "name": ".hidden", "valid": false }, + { "name": "-lead", "valid": false }, + { "name": "a/b", "valid": false }, + { "name": "a b", "valid": false }, + { "name": "outputs", "valid": false }, + { "name": "", "valid": false }, + { "name": "<100 x characters>", "valid": true }, + { "name": "<101 x characters>", "valid": false } +] +``` + +- [ ] **Step 2: Write both tests** + +Append to `tests/test_ui_twins.py`: + +```python +import json + +import pytest + +from dw.security import InvalidInputError, SecurityError, validate_workspace_name +from dw.workspace import RESERVED_WORKSPACE_NAMES + +WORKSPACE_NAMES = json.loads( + (REPO / "tests" / "fixtures" / "workspace_names.json").read_text() +) + + +@pytest.mark.parametrize("case", WORKSPACE_NAMES, ids=lambda c: c["name"][:12] or "empty") +def test_the_engine_decides_each_shared_workspace_name_case(case): + if case["valid"]: + validate_workspace_name(case["name"], reserved=RESERVED_WORKSPACE_NAMES) + else: + with pytest.raises((InvalidInputError, SecurityError)): + validate_workspace_name(case["name"], reserved=RESERVED_WORKSPACE_NAMES) + + +def test_the_ui_reserves_the_engines_workspace_names(): + assert ts_string_array( + UI_LIB / "workspaceActions.ts", "RESERVED_WORKSPACE_NAMES" + ) == list(RESERVED_WORKSPACE_NAMES) +``` + +Append to `ui/src/lib/workspaceActions.test.ts`: + +```ts +import { readFileSync } from 'node:fs' +import { fileURLToPath } from 'node:url' + +const cases: { name: string; valid: boolean }[] = JSON.parse( + readFileSync( + fileURLToPath( + new URL('../../../tests/fixtures/workspace_names.json', import.meta.url), + ), + 'utf8', + ), +) + +it.each(cases)('decides $name as the engine does', ({ name, valid }) => { + expect(workspaceNameError(name) === null).toBe(valid) +}) +``` + +(`workspaceNameError` is imported at the top of the file with the module's other exports; add it if it is not.) + +- [ ] **Step 3: Run both to see them fail** + +Run: `venv/bin/python -m pytest tests/test_ui_twins.py -q` then `cd ui && npx vitest run src/lib/workspaceActions.test.ts` +Expected: pytest PASSES (the engine is the owner; if a case fails here, the case is wrong). vitest FAILS on `ep4.v2`, `café` and the 101-character name. + +- [ ] **Step 4: Match the rule** + +In `ui/src/lib/workspaceActions.ts`, replace `const NAME = ...` and the pattern check: + +```ts +/** dw/security.py's WORKSPACE_NAME_PATTERN, `^[\w][\w.-]*\Z`: Python's \w + * is Unicode letters and digits plus underscore, so \p{L}\p{N}_ here. + * tests/fixtures/workspace_names.json is read by both sides' tests. */ +const NAME = /^[\p{L}\p{N}_][\p{L}\p{N}_.-]*$/u +export const MAX_WORKSPACE_NAME_LENGTH = 100 +``` + +and in `workspaceNameError`, before the pattern test: + +```ts + if (name.length > MAX_WORKSPACE_NAME_LENGTH) + return `Use at most ${MAX_WORKSPACE_NAME_LENGTH} characters` + if (!NAME.test(name)) + return 'Start with a letter, digit or _; then letters, digits, _, - and .' +``` + +- [ ] **Step 5: Run the tests and checks** + +Run: `cd ui && npx vitest run && npm run check && npm run lint && npm run metrics -- --check ../docs/stabilization/ui/baseline.json` then `venv/bin/python -m pytest tests/test_ui_twins.py -q` +Expected: all PASS. If an existing test asserted the old message text, update it to the new one. + +- [ ] **Step 6: Commit** + +```bash +git add tests/fixtures/workspace_names.json tests/test_ui_twins.py ui/src/lib/workspaceActions.ts ui/src/lib/workspaceActions.test.ts +git commit -m "fix(ui): workspace names follow the engine's pattern and length cap" +``` + +--- + +### Task 6: The job page renders by the server's media kinds (U4) + +`JobPage.svelte` decides image or video from its own extension regexes, which +miss `.bmp`, `.mov`, `.mkv`, `.avi` and every audio file; an audio output is +shown as a bare link. `dw/server/outputs.py`'s `MEDIA_KINDS` is the owner. +The job detail gains one top-level map, `output_kinds`, so no manifest entry +changes shape and no existing manifest assertion moves. + +**Files:** +- Modify: `dw/server/outputs.py` (`output_kinds`) +- Modify: `dw/server/routes/jobs.py` (`get_job`) +- Test: `tests/test_server.py` +- Modify: `ui/src/lib/types.ts` (`JobDetail.output_kinds`) +- Modify: `ui/src/lib/pages/JobPage.svelte` +- Test: `ui/src/lib/pages/JobPage.test.ts` + +**Interfaces:** +- Produces: `output_kinds(manifest: list | None) -> dict[str, str | None]` in `dw/server/outputs.py`: every file a manifest entry lists, mapped to its `MEDIA_KINDS` value, or `None` for a kind the gallery does not show. `GET /api/jobs/{id}` carries it as `output_kinds`. +- Produces: `JobDetail.output_kinds?: Record` in `types.ts`. + +- [ ] **Step 1: Write the failing server test** + +Append to `tests/test_server.py`, after `test_job_files_are_reported_relative_to_the_output_dir`: + +```python +def test_a_job_reports_each_output_files_media_kind(server, tmp_path): + """The job page renders by kind, so the kind comes from MEDIA_KINDS + rather than an extension list in the browser.""" + outputs = tmp_path / "outputs" + files = [str(outputs / name) for name in ("a.bmp", "b.mov", "c.flac", "d.bin")] + + def script(command): + yield { + "type": "success", + "message": "ok", + "run_count": 1, + "manifest": [{"step": "gen", "files": files}], + } + + with server(script) as client: + job = client.post("/api/jobs", json={"workflow": valid_workflow()}).json() + detail = wait_for_status(client, job["id"], TERMINAL_STATES) + assert detail["output_kinds"] == { + "a.bmp": "image", + "b.mov": "video", + "c.flac": "audio", + "d.bin": None, + } + + +def test_output_kinds_tolerates_a_job_with_no_manifest(): + from dw.server.outputs import output_kinds + + assert output_kinds(None) == {} + assert output_kinds([{"step": "s"}, "not an entry"]) == {} +``` + +- [ ] **Step 2: Run them to see them fail** + +Run: `venv/bin/python -m pytest tests/test_server.py -q -k "media_kind or output_kinds"` +Expected: FAIL (`KeyError: 'output_kinds'`, `ImportError`). + +- [ ] **Step 3: Classify on the server** + +In `dw/server/outputs.py`, after `MEDIA_KINDS`: + +```python +def output_kinds(manifest): + """Each file a job manifest lists, mapped to its MEDIA_KINDS entry, or + None for a kind the gallery does not show. A client renders an output + by this rather than keeping its own extension list.""" + kinds = {} + for entry in manifest or []: + if not isinstance(entry, dict): + continue + for name in entry.get("files") or []: + kinds[name] = MEDIA_KINDS.get(os.path.splitext(name)[1].lower()) + return kinds +``` + +(`os` is already imported there; add it if not.) In `dw/server/routes/jobs.py`'s `get_job`, replace the return: + +```python + # a historical job is already a detail dict; a live one renders itself + detail = job if isinstance(job, dict) else manager.describe(job) + return {**detail, "output_kinds": output_kinds(detail.get("manifest"))} +``` + +with `from ..outputs import output_kinds` among the module's imports. + +- [ ] **Step 4: Run the server tests** + +Run: `venv/bin/python -m pytest tests/test_server.py tests/test_server_jobs.py tests/test_mcp_*.py -q` +Expected: PASS. + +- [ ] **Step 5: Write the failing UI test** + +In `ui/src/lib/pages/JobPage.test.ts`, add: + +```ts +it('renders each output by the kind the server reports', async () => { + detail.job = { + ...job([{ step: 'generate', files: ['a.bmp', 'b.mov', 'c.flac'] }]), + output_kinds: { 'a.bmp': 'image', 'b.mov': 'video', 'c.flac': 'audio' }, + } + const { container } = render(JobPage, { jobId: 'j1' }) + await waitFor(() => expect(container.querySelector('audio')).toBeTruthy()) + expect(container.querySelector('img')?.getAttribute('src')).toContain('a.bmp') + expect(container.querySelector('video')?.getAttribute('src')).toContain('b.mov') + expect(container.querySelector('audio')?.getAttribute('src')).toContain('c.flac') +}) +``` + +In the existing tests that render media and read metadata (the one around the `'a.png', 'clip.mp4'` manifest, and any other that expects an `` or a `galleryMetadata` call), add `output_kinds` to the job they build, for example `{ ...job([...]), output_kinds: { 'a.png': 'image', 'clip.mp4': 'video' } }`, because the page no longer guesses from the extension. + +- [ ] **Step 6: Run it to see it fail** + +Run: `cd ui && npx vitest run src/lib/pages/JobPage.test.ts` +Expected: the new test FAILS (no `