@tekyzinc/gsd-t 5.14.10 → 5.16.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,13 +7,30 @@ You are performing a gap analysis between a provided specification and the exist
7
7
  | Invocation | Mode | Output |
8
8
  |---|---|---|
9
9
  | `/gsd-t-gap-analysis <spec>` | **Report** (default) | `.gsd-t/gap-analysis.md` — Steps 0.5-9 below, unchanged |
10
- | `/gsd-t-gap-analysis <spec> --sheet <url>` | **Client deliverable** | The estimating sheet's columns A-D and M-P, red-teamed — Steps 4a-6c |
10
+ | `/gsd-t-gap-analysis <spec> --sheet <url>` | **Client deliverable** | The estimating sheet's columns A-D and M-Q, red-teamed — Steps 3a-6c |
11
+
12
+ ### Column map (client-deliverable mode)
13
+
14
+ | Col | Holds | Written by |
15
+ |---|---|---|
16
+ | A | Domain | this command |
17
+ | B | User Type | this command |
18
+ | C | Functionality — bold name, blank line, italic purpose | this command |
19
+ | D | Low Level Requirements + `OPEN QUESTIONS:` block | this command |
20
+ | E-L | Phase, sizing, days, money | **`/gsd-t-estimate` — never touched here** |
21
+ | M | Build Status (4 colours) | this command |
22
+ | N | What Works / What Doesn't | this command |
23
+ | O | References | this command |
24
+ | P | **Fallbacks in shipped code** | this command |
25
+ | Q | Impacting Open Hardening Tasks | this command |
26
+
27
+ A sheet from an earlier run may hold hardening tasks in P, or a `Scope` column. Move hardening to Q and let fallbacks take P; never leave two columns claiming the same header.
11
28
 
12
29
  **Report mode is the default and is unchanged.** Every existing caller — `/gsd-t-scan`'s next-step offer, `gsd-t-phase.workflow.js` routing — invokes it without `--sheet` and behaves exactly as before.
13
30
 
14
31
  **Client-deliverable mode** is the procedure proven on the HILO AI Scheduling sheet: 141 raw requirements reduced to 33 feature rows carrying 595 checkable claims, judged against the code, then attacked by a red team that found a 70% defect rate in a column a friendly sample had already approved. It writes only the *what and whether* columns; `/gsd-t-estimate` sizes and prices them afterward. Behaviour map: `.gsd-t/pseudocode/PseudoCode-GapAnalysis.md`.
15
32
 
16
- In client-deliverable mode, run Steps 0.5-3 as written (they load context and parse the spec), then branch to Step 4a instead of Step 4.
33
+ In client-deliverable mode, run Step 0.5 and Step 1, then branch to **Step 3a (Harvest)** — do NOT run Step 2 (Parse Requirements). Step 2 assumes a finished specification was handed to you; in this mode **there usually is no such document**, and expecting one is how a run stalls asking for a spec that was never going to exist.
17
34
 
18
35
  ## Step 0.5: Scan Freshness Auto-Refresh
19
36
 
@@ -196,10 +213,40 @@ If agent teams are not available or there are fewer than 3 requirements, run seq
196
213
 
197
214
  ---
198
215
 
199
- # Client-deliverable mode (Steps 4a-6c) — only when `--sheet <url>` was given
216
+ # Client-deliverable mode (Steps 3a-6c) — only when `--sheet <url>` was given
200
217
 
201
218
  Report mode skips this whole block and continues at Step 6.
202
219
 
220
+ ## Step 3a: HARVEST the sources — the requirements do not exist yet, you are deriving them
221
+
222
+ **Do not ask the user for a requirements document. In this mode there usually isn't one.** The job is to harvest everything a tracker project holds and synthesize requirements from it. A run that stops to request a spec has misread the task — this exact stall happened on the FRC Predictive run.
223
+
224
+ Harvest, in this order, and report a count for each:
225
+
226
+ 1. **The project's attachments** — NOT its description. On the proven run the description was empty and all three requirement documents were attached files. Pull `attachments?parent=<project_gid>`, then each attachment's `download_url`, and read them. A Statement of Work PDF was the single richest source: **1,431 lines, whose Exhibit B (~1,000 lines) was the real specification.**
227
+ 2. **Every task** — name, notes, and status.
228
+ 3. **Every subtask.** Parent tasks are usually coarse rollups; the real detail and the file citations live one level down. The proven run pulled **192 subtasks** because *"the rollups are too coarse."* One rollup alone carried 46 findings.
229
+ 4. **Task comments.** One Bugs section — a ticket plus three pull-request comments — drove several rows by itself.
230
+ 5. **Any spec documents in the repo**, if the user named one.
231
+
232
+ **An empty section list or a zero-task view is not an empty project.** Check attachments before concluding anything is missing.
233
+
234
+ **If a genuinely required source cannot be reached** — no credentials, a 403, an attachment that will not download — say exactly which one and stop. That is a blocked run. "The project has no tasks" is not, until attachments have been checked too.
235
+
236
+ **Tracker status is unreliable — never carry it through as truth.** On the proven run *"3 of the 6 tasks still show open, but their comments say implemented in PR #4386 with named commits."* Every status is re-decided against the code in Step 5a; a tracker status only ever becomes a flag that the tracker disagrees with reality.
237
+
238
+ ## Step 3b: REPAIR the graph before you rely on it
239
+
240
+ The judgment in Step 5a is only as good as the index it queries.
241
+
242
+ 1. `gsd-t graph status`.
243
+ 2. **Missing → build it**: `gsd-t graph index` (allow up to 900s on a large repo). An absent index is repairable, never a reason to stop.
244
+ 3. **Stale → re-index** the touched set.
245
+ 4. **Broken → repair, then re-verify anything already judged against it.** The proven run found its own index producing *"unresolved call edges"* and had to re-check the Partial/Implemented calls that hinge on whether code is actually wired in.
246
+ 5. **Cannot build it → HALT.** Do not answer structural questions by grep; grep matches text, and the question is about relationships.
247
+
248
+ Report which of these happened. A silent "the graph was fine" claim is not acceptable — say whether it was built, re-indexed, repaired, or already current.
249
+
203
250
  ## Step 4a: Strip everything that isn't buildable — HUMAN-CONFIRMED
204
251
 
205
252
  **This step removes most of the input, and getting it wrong poisons every step after it.**
@@ -232,27 +279,127 @@ Group the survivors into features a person would name (`Availability, Operating
232
279
 
233
280
  ## Step 4c: Describe each feature
234
281
 
235
- - **Column C** — the feature name in bold, then ONE sentence of purpose beneath it.
282
+ **Columns A, B, C — the reading columns.** A client scans these three and stops; everything right of them is detail. Format them exactly:
283
+
284
+ - **Column A — Domain.** Plain text, one or two words: `Training Pace`, `Configuration`, `Curriculum`.
285
+ - **Column B — User Type.** The roles this serves, comma-separated: `Student, Instructor, Scheduling admin`. Wraps onto two lines; that is fine.
286
+ - **Column C — Functionality.** THREE parts in one cell:
287
+ 1. The feature name, **bold**: `Training Pace (EPW) Service and Exceptions`
288
+ 2. A **blank line**
289
+ 3. One sentence of purpose, in *italic*, saying what it is FOR — not what it does mechanically.
290
+
291
+ ```
292
+ Training Pace (EPW) Service and Exceptions ← bold
293
+
294
+ Would give the whole platform one trustworthy ← italic
295
+ answer to whether a student is keeping up, what
296
+ a proposed change would do to that, and how the
297
+ requirement bends for a part week or an
298
+ approved break.
299
+ ```
300
+
301
+ **Match the verb to the build status.** A purpose sentence in present tense on an unbuilt row asserts the feature exists — the same defect as present-tense requirement bullets:
302
+
303
+ | Status | Voice | Example |
304
+ |---|---|---|
305
+ | Not Implemented | **conditional** | *"**Would give** the whole platform one trustworthy answer to whether a student is keeping up"* |
306
+ | Implemented / Partial | present | *"**Keeps** every scheduling setting a school has chosen in one checked, version-tracked place"* |
307
+
308
+ Purpose, not mechanics. *"Would give the platform one trustworthy answer to whether a student is keeping up"* — not *"computes events-per-week from enrollment data."*
309
+
236
310
  - **Column D** — the requirements as bullets, written as **instructions**: "Work out each student's required events-per-week." **Never** as statements of current fact: "Resolves each student's required events-per-week."
237
311
 
238
312
  Present tense reads as a description of working code. On a row that turns out to be unbuilt, the row contradicts itself — this exact defect appeared in the proven run and had to be rewritten across all 32 rows.
239
313
 
240
314
  Plain words in column D. No jargon a client would have to decode.
241
315
 
316
+ **Every row ends with its OPEN QUESTIONS — the unknowns you could not settle from the sources.** Append to the same cell, after a blank line:
317
+
318
+ ```
319
+ OPEN QUESTIONS:
320
+ ? Which specific settings count as safety settings and therefore cannot be overridden by a lower layer
321
+ ? Whether enrollment-level exceptions are set per student or per cohort
322
+ ```
323
+
324
+ One question per line, prefixed `?`, phrased so a client can answer it without reading code. These are the decisions someone must make before the work can be built or priced — a version-migration policy, which values are non-overridable, whether a school may define its own types.
325
+
326
+ **A row with no `OPEN QUESTIONS` block is asserting there are no unknowns.** That is occasionally true and usually not. On the proven sheet, six of eight sampled rows carried them. Before omitting the block, ask: *could two reasonable people build this differently from what column D says?* If yes, the difference is an open question.
327
+
328
+ **Do not invent questions to fill the block, and do not answer them yourself.** An unknown you resolved by assumption is not an open question — it is an assumption, and it belongs in column D as a stated bullet.
329
+
242
330
  ## Step 5a: Judge each feature against the code
243
331
 
244
- Query the code graph for each feature's surface (`gsd-t graph`), read what it names, and decide:
332
+ Query the code graph for each feature's surface (`gsd-t graph`), read what it names, and decide. **Four statuses, four colours — every one of them coloured, none left plain:**
333
+
334
+ | Status | Colour | Means | What it needs |
335
+ |---|---|---|---|
336
+ | **Implemented** | 🟢 green | Every bullet is built | Verification only |
337
+ | **Partial** | 🟡 yellow | Some bullets are built | **Finishing** |
338
+ | **Incorrect** | 🔵 light blue | Code exists and **contradicts** the requirement | **Fixing** — different work from finishing |
339
+ | **Not Implemented** | 🔴 red | No code exists for it | Building |
340
+
341
+ **Incorrect is not a shade of Partial.** Partial means work is missing; Incorrect means work is present and wrong, and someone must first undo what is there. Pricing the two identically understates the second. If you never assign Incorrect across a whole sheet, check whether you have been folding contradictions into Partial.
245
342
 
246
- - **Implemented** — every bullet is built.
247
- - **Not Implemented** — none of it is.
248
- - **Partial** — some is. **MUST be followed by `Not implemented:` and the specific bullets that are missing.** A bare "Partial" is not an answer anyone can act on.
343
+ **Partial and Incorrect MUST name specifics.** Partial is followed by `Not implemented:` and the exact bullets missing. Incorrect is followed by `Contradicts:` and what the code does instead. A bare "Partial" or "Incorrect" is not an answer anyone can act on.
249
344
 
250
- Write the verdict to **column M**, and to **column N** what works today versus what does not.
345
+ Write the verdict to **column M** with its background colour, and to **column N** what works today versus what does not.
251
346
 
252
347
  ## Step 5b: References and impacting debt
253
348
 
254
349
  - **Column O** — where each claim came from: `Requirements doc (6 reqs) · Asana task · PR #4386`. Short clickable names, each token its own link. Never a wall of file paths.
255
- - **Column P** — open Extreme/Critical findings from `.gsd-t/techdebt.md` that would hit this feature. **Judge by the code each finding cites, not by keyword match.**
350
+ - **Column Q** — open Extreme/Critical findings that would hit this feature. **Judge by the code each finding cites, not by keyword match.** (This sat in P before fallbacks took that slot; a sheet built by an earlier run may still have it in P — move it rather than leaving two columns claiming the same thing.)
351
+
352
+ **Where the findings come from, and why the tracker alone is not enough:** pull the hardening project's open items at **subtask level** — the parent tasks are rollups. Then **match each subtask back to `.gsd-t/techdebt.md`** for its real file citations. On the proven run the tracker notes were too thin to judge from, and 152 of 154 subtasks had to be resolved against the local register to get the file paths that make the mapping decidable. Of 154 open subtasks, 28 genuinely touched the feature set.
353
+
354
+ A feature with no findings is a real answer, not a miss — usually an unbuilt feature with no shipped code for a defect to land on. Say so rather than leaving the cell ambiguous.
355
+
356
+ ## Step 5c: Fallbacks in the shipped code — a second pass over the built features
357
+
358
+ **Only for rows marked Implemented, Partial, or Incorrect.** A Not Implemented row has no shipped code, so it has no fallbacks — leave its cell empty rather than writing "none", which reads as a clean result.
359
+
360
+ A **fallback** is any branch that continues after a failure: a `catch` that carries on, a `|| default`, a silent degrade, a secondary path that hides the first one failing. Each one is a place a real failure can hide, and a place the client's system will do something quietly wrong instead of stopping.
361
+
362
+ Run the detector over the project once:
363
+
364
+ ```bash
365
+ # Project-local copy first, else the global package's copy.
366
+ node bin/gsd-t-fallback-detect.cjs --scan --project . --json
367
+ ```
368
+
369
+ It returns `{ok, filesScanned, found, preExisting, approved, unapproved, findings}`. There is no `gsd-t fallback-detect` subcommand — invoke the file directly.
370
+
371
+ Then attribute each finding to the feature whose code it sits in — **judged by the file the finding cites**, the same rule as the debt mapping in Step 5b, never by keyword.
372
+
373
+ Write to **column P** — the slot right after References, so the reader meets a feature's own defects before the wider hardening list in Q:
374
+
375
+ ```
376
+ ⚠ UNAPPROVED — solver.ts:214
377
+ catch → returns an empty slot list
378
+ ⚠ UNAPPROVED — pace.ts:88
379
+ missing required-lessons-per-week → defaults to 3
380
+ ◆ pre-existing — legacy-import.ts:31
381
+ predates the rule; never justified
382
+ ✓ approved — cache.ts:40
383
+ "stale read beats a hard failure on a cold cache" — DH
384
+ ```
385
+
386
+ **Three states, and they mean different things — do not collapse them into two:**
387
+
388
+ | State | Where it comes from | What to tell the client |
389
+ |---|---|---|
390
+ | **approved** | listed in `.gsd-t/fallbacks.json`, with a written reason | Someone justified this deliberately |
391
+ | **pre-existing** | listed in `.gsd-t/fallbacks-baseline.json` | Predates the rule. **Not approved — merely grandfathered.** Nobody has ever justified it |
392
+ | **unapproved** | in neither file | The finding |
393
+
394
+ An approval carries six required fields — `id`, `location`, `whatFails`, `whyNotHalt`, `whatItDoesInstead`, `approvedBy` — so "approved" means someone wrote down what fails, why stopping was worse, and what happens instead. Quote the `whyNotHalt` reason in the cell; an approval nobody can read is indistinguishable from an unapproved one.
395
+
396
+ **Pre-existing is a finding for a client, even though the gate lets it through.** The gate blocks NEW fallbacks; a grandfathered one is still unexamined code that continues after a failure. Show it.
397
+
398
+ **List all three.** The unapproved and pre-existing ones are the findings. The approved ones prove the pass ran and judged rather than skipped — a column showing only warnings cannot be told apart from one that found nothing.
399
+
400
+ **A malformed `fallbacks.json` is a HALT, never "no approvals."** The detector already refuses to read unreadable as approved; do not paper over that by writing an empty cell.
401
+
402
+ **Where a fallback contradicts the row's own column D, say so.** The requirements already ban several by name — *"never silently add the whole instructor roster as fallbacks when the policy turns fallback off"*, *"refuse to fall back to an out-of-date old field once the move has run."* A fallback in the code that the spec explicitly forbids is the strongest finding on the sheet: mark it `⚠ CONTRADICTS COLUMN D` and quote the bullet it breaks.
256
403
 
257
404
  ## Step 6a: RED TEAM the sheet — NOT a review
258
405
 
@@ -272,10 +419,13 @@ Verdict per checker: `FAIL` (defects listed) or `GRUDGING-PASS` (searched, found
272
419
  | O — references | 76 | **Every one** | A link resolves or it doesn't. |
273
420
  | Dropped list (Step 4a) | all | **Every one** | A wrongly-dropped requirement is invisible downstream — nothing else can catch it. |
274
421
  | M — gap bullets | 108 | **Batch of 20** | Each asserts code is absent; disproving it costs a search. |
275
- | P — debt mappings | 104 | **Batch of 20** | Each needs the finding read AND the feature's code read. |
422
+ | P — fallbacks | varies | **Batch of 20** | Each needs the cited code read to confirm it really continues after a failure. |
423
+ | Q — debt mappings | 104 | **Batch of 20** | Each needs the finding read AND the feature's code read. |
276
424
 
277
425
  **Widening rule:** if **3 or more** of a batch of 20 are wrong, that column goes to every-claim. The batch proved the column is unreliable; sampling further only hides the rest.
278
426
 
427
+ **A red-team finding can overturn the analysis, not just its wording.** On the proven run a checker reversed a substantive verdict: *"My reconciliation claimed aircraft ranking didn't exist. The code does build a real ordered list."* When a checker contradicts a status call, the checker's evidence wins unless you can cite code that refutes it — the checker read the code fresh, the original judgment did not.
428
+
279
429
  **One checker does nothing but hunt cross-column contradictions** — a row whose column C asserts a capability in present tense while column M lists that same capability as the row's defining gap. No single-column checker can see this; it is the defect class that reached the client sheet in the proven run.
280
430
 
281
431
  **All checkers returning clean is a FAILED check, not a clean sheet.** The proven run had one clean column out of six. An all-clean result means the framing was too gentle: re-run with sharper prompts. Do NOT report a clean sheet on the first all-pass.
@@ -289,7 +439,7 @@ Verdict per checker: `FAIL` (defects listed) or `GRUDGING-PASS` (searched, found
289
439
 
290
440
  ## Step 6c: Write the sheet — leave the money alone
291
441
 
292
- Write columns **A-D and M-P** using the service-account path documented in `commands/gsd-t-estimate.md` Step 5 (permanent SA `gsd-t-sheets-writer@ai-estimator-415612.iam.gserviceaccount.com`, self-signed JWT, Sheets v4 REST). A `403` on the read-probe means the sheet isn't shared — prompt the user to share it as Editor and re-probe.
442
+ Write columns **A-D and M-Q** using the service-account path documented in `commands/gsd-t-estimate.md` Step 5 (permanent SA `gsd-t-sheets-writer@ai-estimator-415612.iam.gserviceaccount.com`, self-signed JWT, Sheets v4 REST). A `403` on the read-probe means the sheet isn't shared — prompt the user to share it as Editor and re-probe.
293
443
 
294
444
  **NEVER write columns E-L** (Phase, Web Portal, Backend/API, Days, MFactor Days, Total Days, LOW $, HIGH $). Those are `/gsd-t-estimate`'s to fill, and the sheet's own formulas compute the money from them.
295
445
 
@@ -380,6 +380,12 @@ Use these when user asks for help on a specific command:
380
380
  - **Updates**: new deliverable `share/<Repo>-user-stories.md` (+ optional `.docx` via pandoc) + `share/media/*.png` (rendered diagrams) + `.gsd-t/user-stories/diagrams/*.mmd`
381
381
  - **Use when**: You need to hand a development team discrete, testable user stories in the Tekyz handoff style. Distinct from `/gsd-t-prd` (which writes the INTERNAL `docs/prd.md`) — this is an EXTERNAL client/dev deliverable. Diagrams are authored as Mermaid but embedded as rendered images (needs `@mermaid-js/mermaid-cli`). Format reference: `~/.claude/playbooks/tekyz-user-stories-format.md`
382
382
 
383
+ ### demo-videos
384
+ - **Summary**: Produce narrated screen-recording walkthrough videos of a running application — coverage plan from the live app, Playwright seeding through the real UI, batched text-to-speech with a measured one-voice gate, one continuous recording per walkthrough, audio muxed at recorded timestamps, then silence trimmed
385
+ - **Auto-invoked**: No
386
+ - **Updates**: `docs/demo-videos/PLAN.md`, `docs/demo-videos/HANDOFF.md`, `docs/demo-videos/walkthrough-<name>.mp4`, `e2e/walkthrough/*`, `scripts/walkthrough-*.mjs`, `scripts/demo-data/*`
387
+ - **Use when**: You need shareable demo or training videos of an app that already runs with real data. Narration is the master clock — each step lasts exactly as long as its measured spoken sentence. Needs Playwright, ffmpeg, auto-editor and a TTS key with quota. Templates: `templates/demo-videos/`
388
+
383
389
  ### populate
384
390
  - **Summary**: Auto-populate all living docs from existing codebase analysis
385
391
  - **Auto-invoked**: No
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tekyzinc/gsd-t",
3
- "version": "5.14.10",
3
+ "version": "5.16.10",
4
4
  "description": "GSD-T: Contract-Driven Development for Claude Code — 54 slash commands with headless-by-default workflow spawning, unattended supervisor relay with event stream, graph-powered code analysis, real-time agent dashboard, task telemetry, doc-ripple enforcement, backlog management, impact analysis, test sync, milestone archival, and PRD generation",
5
5
  "author": "Tekyz, Inc.",
6
6
  "license": "MIT",
@@ -0,0 +1,89 @@
1
+ # Demo-video pipeline templates
2
+
3
+ Working scripts behind `/gsd-t-demo-videos`. These are the real thing from a
4
+ completed 12-walkthrough production run, not sketches. Copy them into a project
5
+ and fill the `{tokens}`; do not re-derive the pipeline.
6
+
7
+ ## Install into a project
8
+
9
+ ```bash
10
+ mkdir -p e2e/walkthrough scripts scripts/demo-data docs/demo-videos
11
+ cp <gsd-t>/templates/demo-videos/e2e/* e2e/walkthrough/
12
+ cp <gsd-t>/templates/demo-videos/scripts/walkthrough-*.mjs scripts/
13
+ cp <gsd-t>/templates/demo-videos/scripts/seed-lib.mjs scripts/demo-data/
14
+
15
+ # auto-editor, bundled so it does not depend on PATH
16
+ python3 -m venv .venv-tools/auto-editor
17
+ .venv-tools/auto-editor/bin/pip install auto-editor
18
+
19
+ printf '\n.demo-build/\n.tts-cache-v2/\ndocs/demo-videos/*.mp4\n' >> .gitignore
20
+ ```
21
+
22
+ Add a Playwright project:
23
+
24
+ ```ts
25
+ {
26
+ name: 'walkthrough',
27
+ testDir: './e2e/walkthrough',
28
+ testMatch: /\.spec\.ts$/,
29
+ timeout: 15 * 60_000,
30
+ use: {
31
+ ...devices['Desktop Chrome'],
32
+ baseURL: process.env.DEMO_URL || 'https://demo.example.com',
33
+ viewport: { width: 1440, height: 900 },
34
+ // PIN THE SIZE. Unpinned, Playwright downscales and the result is
35
+ // upscaled back to a blurry finish.
36
+ video: { mode: 'on', size: { width: 1440, height: 900 } },
37
+ trace: 'off',
38
+ },
39
+ }
40
+ ```
41
+
42
+ Then fill: `signin.ts` (persona + tenant id), `preflight.spec.ts` (targets), and
43
+ one `<name>.lines.mjs` + `<name>.spec.ts` pair per walkthrough.
44
+
45
+ ## Run
46
+
47
+ ```bash
48
+ DEMO_RUN=1 npx playwright test --project=walkthrough --grep preflight # 0 check targets
49
+ node scripts/walkthrough-voice-ensure.mjs <name> # 1 narration + gate
50
+ DEMO_RUN=1 npx playwright test --project=walkthrough --grep "<name>" # 2 record
51
+ node scripts/walkthrough-mux.mjs <name> # 3 lay audio on video
52
+ node scripts/walkthrough-trim.mjs <name> # 4 cut the silence
53
+ ```
54
+
55
+ Stage 4 is not optional — it removed 7.4 minutes of dead air across twelve
56
+ videos in the source run.
57
+
58
+ ## Environment
59
+
60
+ | Variable | Purpose |
61
+ |---|---|
62
+ | `DEMO_URL` | the running app to film |
63
+ | `DEMO_GATE_PW` | password gate in front of a demo site, if any |
64
+ | `GEMINI_API_KEY_3` / `_2` / `GEMINI_API_KEY` | TTS, tried in that order |
65
+ | `VOICE_NAME`, `VOICE_SPEED`, `VOICE_BATCH`, `VOICE_LUFS` | narrator settings — all in the cache key |
66
+ | `TTS_MODEL` | pin one model instead of the fallback list |
67
+ | `TRIM_MARGIN` | silence kept either side of speech (default `0.2s`) |
68
+ | `HEADED=1` | watch a seeder run |
69
+
70
+ ## What each file is for
71
+
72
+ | File | Role |
73
+ |---|---|
74
+ | `e2e/runtime.ts` | `step()`, `highlight()`, `click()`, `goTo()`; narration is the clock; throws when a sentence has no visible target |
75
+ | `e2e/manifest.ts` | loads narration; a spec **skips** when its audio is missing rather than failing the run |
76
+ | `e2e/signin.ts` | shared sign-in + the tenant constant — **ask for this, don't derive it** |
77
+ | `e2e/preflight.spec.ts` | walks every target without recording, reports all misses in one pass |
78
+ | `e2e/example.lines.mjs` | narration shape, with the writing rules |
79
+ | `e2e/example.spec.ts` | spec shape, with the overlay/settle rules |
80
+ | `scripts/walkthrough-voice.mjs` | batched TTS (8/request), verified split, two-pass loudnorm |
81
+ | `scripts/walkthrough-voice-check.mjs` | measures LUFS/pitch/rate spread + miscuts; **exit 4** on drift |
82
+ | `scripts/walkthrough-voice-ensure.mjs` | render → measure → re-render → halt at 3 |
83
+ | `scripts/walkthrough-normalise.mjs` | force existing clips to one loudness, no API calls |
84
+ | `scripts/walkthrough-mux.mjs` | places clips at recorded timestamps; no overlaps; warns on overrun |
85
+ | `scripts/walkthrough-trim.mjs` | auto-editor silence removal, in place, re-runnable |
86
+ | `scripts/demo-data/seed-lib.mjs` | seeder sign-in, settle, and the refusal-reporting rule |
87
+
88
+ The reasoning behind every threshold and retry is in the file headers — they are
89
+ the record of what failed, so read them before changing a constant.
@@ -0,0 +1,28 @@
1
+ /**
2
+ * TEMPLATE — the narration for one walkthrough. Nothing but sentences.
3
+ *
4
+ * A BEAT IS ONE IDEA BEING EXPLAINED, NOT ONE SCREEN. A beat may dwell on three
5
+ * things within a screen, or carry across a navigation. Building around screens
6
+ * is what produced fixed-length steps and the drift that followed.
7
+ *
8
+ * Rules, each from a user correction in the source run:
9
+ * - Never name something the viewer cannot see. If the sentence names a
10
+ * thing, the matching step must point at that thing.
11
+ * - Don't invent jargon. Not "groups" — "the left sidebar's top-level menus,
12
+ * which expand to show…".
13
+ * - Explain, don't sell. No "exciting", no "powerful", no enthusiasm.
14
+ * - Give the dependency context: why this screen exists, and what downstream
15
+ * reads from it.
16
+ * - COUNT WHAT IS ON SCREEN before writing about it. "Eight-step wizard"
17
+ * shipped in a video where the UI said "Step 1 of 9".
18
+ *
19
+ * The order here is the order in the spec. Index N of LINES is L[N] there.
20
+ */
21
+ export const LINES = [
22
+ "Before anyone can book a flight, the aircraft has to exist here, at the location it lives at.",
23
+ "The four cards across the top are the fleet's condition right now: how many are available, how many are grounded, how many open squawks there are, and how much maintenance is coming due.",
24
+ "Below that is the fleet itself. Seven aircraft at this location, each with its hourly rate.",
25
+ "That rate is what turns a flight into money later, so it belongs to the aircraft, not to the booking.",
26
+ "Opening an aircraft gives you its full record, and the record has seven tabs — each one a different kind of history for the same airframe.",
27
+ "So: register the aircraft, set its rate, keep its maintenance and its logbook current. Everything downstream reads from this record.",
28
+ ];
@@ -0,0 +1,53 @@
1
+ /**
2
+ * TEMPLATE — one continuous recorded walkthrough.
3
+ *
4
+ * One sentence, one thing on screen, in the same order as the .lines.mjs file.
5
+ * The step's LENGTH comes from the measured audio; nothing here sets a duration
6
+ * except the settle waits after a navigation.
7
+ *
8
+ * highlight(x) outline it, mouse stays put (naming something)
9
+ * click(x) mouse glides to it and clicks (operating something)
10
+ * goTo(url) navigate (jumping elsewhere)
11
+ * none() no target — ONLY for a genuinely abstract line
12
+ *
13
+ * AVOID highlight() IMMEDIATELY FOLLOWED BY goTo(). On camera that reads as
14
+ * "they clicked that thing and it took us here" — and they did not. Click the
15
+ * real navigation, or park the cursor somewhere neutral first.
16
+ */
17
+ import { test } from '@playwright/test';
18
+ import { loadNarration } from './manifest';
19
+ import { LOC, signInAsAdmin } from './signin';
20
+ import { click, highlight, installOverlay, none, startRun, step, writeRunLog } from './runtime';
21
+
22
+ const NAME = 'example';
23
+ const { ready, audio: AUDIO, L } = loadNarration(NAME);
24
+
25
+ test(`walkthrough — ${NAME}`, async ({ page }) => {
26
+ // A spec whose audio is not rendered SKIPS. Playwright imports every spec
27
+ // before applying --grep, so throwing here takes down the whole run.
28
+ test.skip(!ready, 'narration not rendered yet');
29
+ test.setTimeout(20 * 60_000);
30
+
31
+ startRun(AUDIO as never);
32
+ await signInAsAdmin(page);
33
+
34
+ await page.goto(`/location/${LOC}/{route}`, { waitUntil: 'domcontentloaded' });
35
+ await page.waitForTimeout(6000); // 6-8s, not 5 — see the settle note
36
+ await installOverlay(page);
37
+
38
+ await step(page, L[0], none());
39
+ await step(page, L[1], highlight(page.getByText(/{Card Title}/i)));
40
+ await step(page, L[2], highlight(page.getByText(/{List Heading}/i)));
41
+ await step(page, L[3], highlight(page.getByText(/{Rate}/i).first()));
42
+
43
+ await step(page, L[4], click(page.getByRole('button', { name: /{Open Record}/i })));
44
+ // Re-install the overlay after every navigation — the cursor and highlight
45
+ // layers are injected into the page and do not survive one.
46
+ await page.waitForTimeout(6000);
47
+ await installOverlay(page);
48
+
49
+ await step(page, L[5], none());
50
+
51
+ // The step log is the proof of what was on screen when. The mux reads it.
52
+ writeRunLog(NAME);
53
+ });
@@ -0,0 +1,30 @@
1
+ /**
2
+ * Load a walkthrough's narration manifest.
3
+ *
4
+ * Playwright imports EVERY spec in the project before it applies --grep, so a
5
+ * spec whose audio has not been rendered yet used to throw at import time and
6
+ * take the whole run down with it — including the walkthroughs that were ready.
7
+ * Missing audio is a reason to skip that one walkthrough, not to fail the run.
8
+ */
9
+ import { existsSync, readFileSync } from 'node:fs';
10
+ import path from 'node:path';
11
+
12
+ export interface Narration {
13
+ ready: boolean;
14
+ audio: Map<string, { file: string; ms: number }>;
15
+ L: string[];
16
+ }
17
+
18
+ export function loadNarration(name: string): Narration {
19
+ const file = path.join(process.cwd(), '.demo-build', `audio-${name}.json`);
20
+ if (!existsSync(file)) return { ready: false, audio: new Map(), L: [] };
21
+
22
+ const manifest = JSON.parse(readFileSync(file, 'utf8'));
23
+ return {
24
+ ready: true,
25
+ audio: new Map(
26
+ manifest.lines.map((l: any) => [l.text, { file: l.file, ms: l.ms }]),
27
+ ),
28
+ L: manifest.lines.map((l: any) => l.text as string),
29
+ };
30
+ }
@@ -0,0 +1,109 @@
1
+ /**
2
+ * Stage 0 — check every walkthrough's targets WITHOUT recording.
3
+ * TEMPLATE: fill SCREENS and DRILLDOWNS with what your specs point at.
4
+ *
5
+ * Recording is the slow way to discover a bad selector: the run stops at the
6
+ * first sentence whose target is not on screen, so a spec with three bad
7
+ * selectors costs three full recordings to find them all. This walks the same
8
+ * screens and reports EVERY missing target in one pass.
9
+ *
10
+ * DEMO_RUN=1 npx playwright test --project=walkthrough --grep preflight
11
+ *
12
+ * It never asserts, so it cannot fail the suite — it prints a report. The real
13
+ * gate is still the recording itself, which refuses to narrate what is not
14
+ * visible; this just makes getting there quick.
15
+ */
16
+ import { test } from '@playwright/test';
17
+ import { LOC, signInAsAdmin } from './signin';
18
+
19
+ interface Target {
20
+ where: string;
21
+ what: string;
22
+ find: (page: any) => any;
23
+ }
24
+
25
+ /** Everything the specs point at, grouped by the screen it must appear on. */
26
+ const SCREENS: Array<{ route: string; settle: number; targets: Target[] }> = [
27
+ {
28
+ route: `/location/${LOC}/dashboard`,
29
+ // 6-8 SECONDS, NOT 5 — a 5s read reports a populated page as empty.
30
+ settle: 8_000,
31
+ targets: [
32
+ { where: 'dashboard', what: '{a heading}', find: (p) => p.getByText(/{Heading}/i) },
33
+ { where: 'dashboard', what: '{a card}', find: (p) => p.getByText(/{Card Title}/i) },
34
+ ],
35
+ },
36
+ ];
37
+
38
+ /**
39
+ * Screens reached by opening a record and clicking its tabs, rather than by URL.
40
+ * COUNT THE TABS ON SCREEN — a record with 7 tabs narrated as 3 ships an error.
41
+ */
42
+ const DRILLDOWNS: Array<{
43
+ name: string;
44
+ open: (page: any) => Promise<void>;
45
+ tabs: Array<{ tab: string; targets: Target[] }>;
46
+ }> = [
47
+ // {
48
+ // name: 'aircraft',
49
+ // open: async (p) => {
50
+ // await p.goto(`/location/${LOC}/{route}`, { waitUntil: 'domcontentloaded' });
51
+ // await p.waitForTimeout(8_000);
52
+ // await p.getByRole('button', { name: /{Open Record}/i }).first().click();
53
+ // await p.waitForTimeout(6_000);
54
+ // },
55
+ // tabs: [{ tab: '{Tab Name}', targets: [...] }],
56
+ // },
57
+ ];
58
+
59
+ async function firstVisible(loc: any): Promise<{ visible: boolean; n: number }> {
60
+ const n = await loc.count().catch(() => 0);
61
+ for (let i = 0; i < n; i += 1) {
62
+ if (await loc.nth(i).isVisible().catch(() => false)) return { visible: true, n };
63
+ }
64
+ return { visible: false, n };
65
+ }
66
+
67
+ test('walkthrough — preflight target check', async ({ page }) => {
68
+ test.setTimeout(25 * 60_000);
69
+ await signInAsAdmin(page);
70
+
71
+ const missing: string[] = [];
72
+ let checked = 0;
73
+
74
+ for (const screen of SCREENS) {
75
+ await page.goto(screen.route, { waitUntil: 'domcontentloaded' }).catch(() => {});
76
+ await page.waitForTimeout(screen.settle);
77
+
78
+ for (const t of screen.targets) {
79
+ checked += 1;
80
+ const { visible, n } = await firstVisible(t.find(page));
81
+ if (!visible) missing.push(`${t.where.padEnd(14)} ${t.what} (${n} matched, none visible)`);
82
+ }
83
+ }
84
+
85
+ for (const d of DRILLDOWNS) {
86
+ await d.open(page).catch(() => {});
87
+ for (const group of d.tabs) {
88
+ const tab = page
89
+ .getByRole('button', { name: new RegExp(`^${group.tab}$`, 'i') })
90
+ .first();
91
+ if (await tab.isVisible().catch(() => false)) {
92
+ await tab.click().catch(() => {});
93
+ await page.waitForTimeout(5_000);
94
+ } else {
95
+ missing.push(`${d.name.padEnd(14)} tab "${group.tab}" not reachable`);
96
+ continue;
97
+ }
98
+ for (const t of group.targets) {
99
+ checked += 1;
100
+ const { visible, n } = await firstVisible(t.find(page));
101
+ if (!visible) missing.push(`${t.where.padEnd(14)} ${t.what} (${n} matched, none visible)`);
102
+ }
103
+ }
104
+ }
105
+
106
+ console.log(`\n── preflight: ${checked} targets checked, ${missing.length} missing`);
107
+ for (const m of missing) console.log(` MISSING ${m}`);
108
+ if (!missing.length) console.log(' all targets visible');
109
+ });