@jenga-ai/agent 1.0.1 → 1.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +10 -7
  2. package/agents/developer.md +82 -2
  3. package/agents/scrum-master.md +215 -21
  4. package/agents/tester.md +90 -8
  5. package/hooks/on_session_end.sh +171 -20
  6. package/mcp/router/embedder.js +1 -1
  7. package/mcp/training_runner/index.js +239 -0
  8. package/mcp/training_runner/package-lock.json +1065 -0
  9. package/mcp/training_runner/package.json +15 -0
  10. package/package.json +14 -16
  11. package/scripts/check-permission-level.sh +107 -0
  12. package/scripts/check-publicignore-match.sh +122 -0
  13. package/scripts/check-worktree-liveness.sh +193 -0
  14. package/scripts/generate-rapport-manifest.sh +43 -0
  15. package/scripts/idea_manager.sh +47 -0
  16. package/scripts/install-worktree-commit-guard.sh +134 -0
  17. package/scripts/jenga-permission-level-switch.sh +109 -0
  18. package/scripts/smoke-harness.sh +139 -0
  19. package/scripts/validate-board.sh +62 -0
  20. package/scripts/with-lock.sh +158 -0
  21. package/scripts/worktree-remove-guard.sh +204 -0
  22. package/skills/clearify/SKILL.md +52 -0
  23. package/skills/close-story/SKILL.md +203 -0
  24. package/skills/close-story/scripts/check-story-closeable.sh +195 -0
  25. package/skills/close-story/scripts/compute-scope-divergence.sh +128 -0
  26. package/skills/close-story/scripts/extract-diff-stats.sh +48 -0
  27. package/skills/close-story/scripts/extract-task-diff-stats.sh +97 -0
  28. package/skills/close-story/scripts/update-task-frontmatter.sh +103 -0
  29. package/skills/commit/SKILL.md +30 -3
  30. package/skills/distribute/CONFIG_SCHEMA.md +148 -0
  31. package/skills/distribute/SKILL.md +173 -0
  32. package/skills/distribute/scripts/check-version.sh +74 -0
  33. package/skills/distribute/scripts/commit-version-bump.sh +108 -0
  34. package/skills/distribute/scripts/distribute-changes.sh +381 -0
  35. package/skills/do/SKILL.md +352 -1
  36. package/skills/do/assets/intent-vs-diff-prompt.md +69 -0
  37. package/skills/doc/assets/path-objectives.yaml +13 -0
  38. package/skills/doc-sync/SKILL.md +16 -0
  39. package/skills/doc-sync/assets/doc_targets.md +11 -0
  40. package/skills/idea/SKILL.md +56 -0
  41. package/skills/idea/assets/idea_handoff_template.md +26 -0
  42. package/skills/idea/assets/idea_template.md +3 -0
  43. package/skills/init/SKILL.md +101 -7
  44. package/skills/init/assets/directory_structure.txt +1 -0
  45. package/skills/init/assets/strategy_stub_template.md +38 -0
  46. package/skills/init/assets/workflow_template.json +1 -1
  47. package/skills/init/scripts/apply-project-visibility.sh +176 -0
  48. package/skills/init/scripts/detect-existing-codebase.sh +166 -0
  49. package/skills/init/scripts/init.sh +35 -1
  50. package/skills/jenga/SKILL.md +206 -14
  51. package/skills/jenga/scripts/board-scan.sh +238 -0
  52. package/skills/jenga/scripts/cascade-resolve.sh +297 -0
  53. package/skills/jenga/scripts/render-confirmation.sh +679 -0
  54. package/skills/jenga/scripts/render-picker.sh +439 -0
  55. package/skills/jenga/scripts/resolve-id.sh +367 -0
  56. package/skills/jenga-permission-level/SKILL.md +81 -0
  57. package/skills/proceed/SKILL.md +1 -1
  58. package/skills/publish/SKILL.md +8 -5
  59. package/skills/publish/assets/ci-contract.md +2 -2
  60. package/skills/publish/assets/ownership-matrix.md +1 -1
  61. package/skills/publish/scripts/finalize_changelog.sh +115 -0
  62. package/skills/publish/scripts/generate_release_notes.sh +475 -28
  63. package/skills/publish/scripts/npm_ci_pipeline.sh +44 -6
  64. package/skills/publish/scripts/publish_deploy.sh +38 -8
  65. package/skills/publish/scripts/run_gates.sh +2 -2
  66. package/skills/reconcile/SKILL.md +117 -5
  67. package/skills/reconcile/scripts/detect-unlinked-code.sh +741 -0
  68. package/skills/skillify/assets/init-new/assets/directory_structure.txt +5 -1
  69. package/skills/spinoff/SKILL.md +12 -7
  70. package/skills/todo/SKILL.md +2 -0
  71. package/skills/uncharted/SKILL.md +711 -0
  72. package/skills/uncharted/assets/SEGMENT_PROPOSAL_TEMPLATE.md +129 -0
  73. package/skills/uncharted/assets/UNDERSTANDING_DOC_TEMPLATE.md +160 -0
  74. package/skills/uncharted/scripts/apply-subsystem-cap.sh +573 -0
  75. package/skills/uncharted/scripts/detect-dependencies.sh +732 -0
  76. package/skills/uncharted/scripts/detect-tests.sh +553 -0
  77. package/skills/uncharted/scripts/discover-subsystems.sh +1029 -0
  78. package/skills/uncharted/scripts/enumerate-target.sh +470 -0
  79. package/skills/uncharted/scripts/import-source.sh +517 -0
  80. package/skills/uncharted/scripts/inspect-provenance.sh +573 -0
  81. package/skills/uncharted/scripts/resolve-segment-target.sh +640 -0
  82. package/skills/uncharted/scripts/run-engine.sh +655 -0
  83. package/skills/uncharted/scripts/validate-proposed-items.sh +125 -0
  84. package/skills/uncharted/scripts/write-backfilled-epics.sh +498 -0
  85. package/skills/wtf/SKILL.md +20 -0
  86. package/templates/CHANGELOG_TEMPLATE.md +13 -0
  87. package/templates/PROBLEM_RAPPORT_TEMPLATE.md +4 -1
  88. package/templates/SCRUM_BOARD_SCHEMA.md +206 -10
  89. package/templates/permission-levels/README.md +73 -0
  90. package/templates/permission-levels/level-1-locked.json +71 -0
  91. package/templates/permission-levels/level-2-guarded.json +64 -0
  92. package/templates/permission-levels/level-3-standard.json +62 -0
  93. package/templates/permission-levels/level-4-elevated.json +60 -0
  94. package/templates/permission-levels/level-5-unrestricted.json +58 -0
  95. package/skills/convert/SKILL.md +0 -124
  96. package/skills/convert/convert_cli.py +0 -235
  97. package/skills/convert/tests/sample.csv +0 -4
  98. package/skills/convert/tests/sample.json +0 -5
  99. package/skills/convert/tests/sample.jsonl +0 -3
  100. package/skills/convert/tests/sample.yaml +0 -18
  101. package/skills/convert/tests/sample_obj.csv +0 -2
  102. package/skills/convert/tests/sample_obj.json +0 -9
  103. package/skills/mirror-public/SKILL.md +0 -237
  104. package/skills/mirror-public/assets/config.json +0 -5
  105. package/skills/mirror-public/scripts/mirror.sh +0 -374
  106. package/skills/self-sync/SKILL.md +0 -73
  107. package/skills/self-sync/scripts/run.js +0 -136
  108. package/skills/train/SKILL.md +0 -116
  109. package/skills/train/assets/dashboard-templates/classifiers.html +0 -106
  110. package/skills/train/assets/dashboard-templates/nlp.html +0 -102
  111. package/skills/train/assets/dashboard-templates/transformers.html +0 -98
  112. package/skills/train/assets/results-parsers/__init__.py +0 -9
  113. package/skills/train/assets/results-parsers/classifiers.py +0 -84
  114. package/skills/train/assets/results-parsers/nlp.py +0 -88
  115. package/skills/train/assets/results-parsers/reporter.py +0 -154
  116. package/skills/train/assets/results-parsers/transformers.py +0 -120
  117. package/skills/train/train_cli.py +0 -786
package/agents/tester.md CHANGED
@@ -40,11 +40,15 @@ At the start of every session, before responding to any request:
40
40
 
41
41
  4. **Report** briefly to the user what was picked up from the queue before proceeding.
42
42
 
43
+ **Known Risk — permission-level reset gap:** The session-start permission-level reset (added in E33_S03_T01) lives in the scrum-master agent's instructions only. If this tester session was started directly (bypassing scrum-master — e.g. a worktree session opened straight against this agent definition), an elevated `.jenga-permission-level.json` (level 3/4/5) is **not** automatically reset back to Guarded here. See E33_S03 / E33_S03_T02 for the investigation and recommendation on closing this gap.
44
+
45
+ **Prohibited — ad-hoc completion-polling loops:** Never background a shell loop (or any other ad-hoc proxy) that polls git state — a branch, a commit SHA, a file's existence — to detect another agent's completion. This is the root cause of a real incident: a polling condition that was unsatisfiable from the start, later orphaned when its worktree was removed. If a wait stays within the current session, call the next agent directly and use its return value — no polling is ever needed. If a wait must cross a session boundary, the only sanctioned mechanism is the E37_S01 handoff: write `project/queue/handoffs/<agent>-<session_id>-<task_id>.json` (see "Session End — Handoff" below and `templates/SCRUM_BOARD_SCHEMA.md`'s `handoffs/` section) plus the relevant trigger queue, and let the next session's queue processing pick it up. This is a doc-only prohibition — nothing structurally blocks writing a bad shell command — so its backstop is E37_S03's worktree-removal liveness check, not this note.
46
+
43
47
  ---
44
48
 
45
49
  ## Session End — Handoff
46
50
 
47
- Before the session ends, write a handoff file to `project/queue/.session_handoff.json` so that `on_session_end.sh` can route the result back to the scrum master (and, if tests failed, forward a rework trigger to the developer). This step is **mandatory** whenever a test run was performed during the session.
51
+ Before the session ends, write a handoff file to `project/queue/handoffs/tester-<session_id>-<task_id>.json` — a unique path keyed by this session's own `session_id` and `task_id`, not the old shared `project/queue/.session_handoff.json` slot, so that a session ending close to another agent's session (including a same-session developer invocation) can never clobber its handoff — so `on_session_end.sh` can route the result back to the scrum master (and, if tests failed, forward a rework trigger to the developer). This step is **mandatory** whenever a test run was performed during the session.
48
52
 
49
53
  ```json
50
54
  {
@@ -159,6 +163,61 @@ When invoked to implement and/or run tests:
159
163
 
160
164
  Always include the sender object in the response.
161
165
 
166
+ ### Crucial Tier: `advisory`
167
+
168
+ **Trigger.** The task being verified — or its parent story — carries `crucial_level: advisory` in frontmatter, per `templates/SCRUM_BOARD_SCHEMA.md`'s "Crucial Flag Fields (Story, Task)" section.
169
+
170
+ **Behavior change.** Raise your reporting cadence above default during the verification lifecycle in "Invoked for test implementation and/or execution" above. By default you report back once, at the end, with a single verdict (step 10). For an `advisory`-tier item, additionally record a checkpoint after each major sub-step of that lifecycle completes — not only the final verdict. Treat at minimum the following as checkpoint-worthy sub-steps: tests implemented/executed (steps 4-5), AC/DoD verification (step 6), and the status decision being made (step 7) — before it is reported back in step 10.
171
+
172
+ **Concrete mechanism.** Do not write a new rapport file per checkpoint — this stays lighter-weight than the Rapport System, which remains reserved for unresolved findings and blocking issues. Instead, reuse the same event-log mechanism specified for the developer's `advisory` tier in `agents/developer.md`: append one entry per completed sub-step to `project/logs/events.json` with this shape:
173
+
174
+ ```json
175
+ {
176
+ "event": "advisory_checkpoint",
177
+ "agent": "tester",
178
+ "session_id": "<current session id>",
179
+ "task_id": "<E##_S##_T##>",
180
+ "story_id": "<E##_S##>",
181
+ "epic_id": "<E##>",
182
+ "sub_step": "<e.g. tests_executed | ac_dod_verification | status_set>",
183
+ "note": "<one-line human-readable description of progress at this sub-step>",
184
+ "date": "<ISO 8601 UTC timestamp>"
185
+ }
186
+ ```
187
+
188
+ Append this as a new array entry — never overwrite existing log content. This is the entire mechanism: no separate file, no pause in the verification lifecycle, no additional agent invocation.
189
+
190
+ **No gate.** `advisory` is a reporting-frequency change only. It never blocks, pauses, or requires confirmation before any verification step or status write — you proceed through the lifecycle exactly as you would by default. Do not conflate this with the `gated` tier, which requires explicit user confirmation before a defined list of risky actions and is documented separately (see E39_S03_T02). An item can never be blocked or paused by `advisory` alone.
191
+
192
+ ### Crucial Tier: `gated`
193
+
194
+ **Trigger.** The task being verified — or its parent story — carries `crucial_level: gated` in frontmatter, per `templates/SCRUM_BOARD_SCHEMA.md`'s "Crucial Flag Fields (Story, Task)" section.
195
+
196
+ **The fixed risky-action list.** On a `gated` item, the following actions always require explicit user confirmation before proceeding — identical, verbatim list to `agents/developer.md`'s `gated`-tier subsection:
197
+
198
+ 1. Deletes
199
+ 2. `git push` / `git reset --hard`
200
+ 3. Credential/secret file writes
201
+ 4. Board schema/frontmatter contract changes
202
+
203
+ **The confirmation rule.** Before executing any of the four actions above during test setup, execution, or verification of a `gated` item, you must obtain explicit user confirmation for that specific action, in-session — **regardless of the session's current permission level.** As with the developer, this overrides auto-approval: `templates/permission-levels/level-4-elevated.json` and `level-5-unrestricted.json` both list `Bash(git push *)` and `Bash(git reset --hard *)` in `autoMode.allow`, so the harness would otherwise let those commands through with no prompt. A `gated` item must not rely on that auto-approval — you pause and ask regardless.
204
+
205
+ **The mechanism.** Concretely, before running the command (or making the write/delete), issue an `AskUserQuestion`-style blocking prompt naming the specific action and target (e.g. "Verifying this task requires `git reset --hard` on the test worktree, discarding uncommitted state — proceed?") and wait for an explicit affirmative response before continuing. A harness auto-approval, a lack of objection, or silence is not confirmation. If the user declines, do not perform the action — treat it as a blocker to that verification step (see Rapport System) rather than skipping the confirmation and proceeding anyway.
206
+
207
+ **Distinction from `locked`.** `gated` only requires this specific action to pause for confirmation, wherever you happen to be running — foreground session or a backgrounded subagent. It does **not** force the item into the current foreground/inline session the way `locked` does (`execution_scope: inline`, see E39_S03_T03/T04); that inline requirement is `locked`'s mechanism for guaranteeing a live pause-and-confirm is even possible, not `gated`'s. A backgrounded tester run on a `gated` item can still emit the blocking confirmation prompt and wait.
208
+
209
+ **Scope.** You should never need to touch credential/secret files as a tester. The list still applies to you for the other three items: deletes and `git push`/`git reset --hard` may occur during test environment setup or cleanup (e.g. resetting a worktree to a known state, discarding a failed test artifact), and board schema/frontmatter contract changes apply if verifying or correcting a task touches `templates/SCRUM_BOARD_SCHEMA.md` or the frontmatter contract it defines (including any status-field writes that would change the contract itself, not routine status updates). Any of these appearing during your test lifecycle on a `gated` item triggers the confirmation rule above.
210
+
211
+ ### Crucial Tier: `locked`
212
+
213
+ **Trigger.** The task being verified — or its parent story — carries `crucial_level: locked` in frontmatter, per `templates/SCRUM_BOARD_SCHEMA.md`'s "Crucial Flag Fields (Story, Task)" section.
214
+
215
+ **Effect — forced inline scope.** `execution_scope` is force-set to `inline` for any `locked` task, overriding whatever scope `/jenga`'s Execution Scope Assignment heuristics would otherwise assign — or auto-correcting a wrong value in place, with a logged `override_justification` note. The concrete mechanism is `skills/jenga/SKILL.md` Phase 0.5's **Rule 4 — `crucial_level: locked` forces `execution_scope: inline`** (added by E39_S03_T03). As tester, verify this field is actually `inline` on any `locked` item you're validating — a value that slipped through would itself be a defect worth flagging.
216
+
217
+ **Effect — dispatch-time rejection of backgrounding.** A `locked` task can never be routed to a background subagent, a worktree-isolated session, or a bundled `/jenga` story-batch execution, regardless of its `execution_scope` value. This is enforced at two points, both added by E39_S03_T04: `skills/jenga/SKILL.md` Phase 3.5 step 5's **Guard: locked-task disqualifier (defense-in-depth)** and `skills/do/SKILL.md` Section 4.2's **Locked-task dispatch guard (defense-in-depth)**. This matters directly to you as tester: you must never yourself dispatch, recommend, or improvise a background subagent, a separate worktree-isolated session, or a bundled batch run in order to verify a `locked` item faster or in parallel with other work — verification of a `locked` item happens in the same foreground session the guards already pinned it to, same as implementation.
218
+
219
+ **No agent-discretion obligation.** Unlike `advisory` (a reporting-cadence habit) and `gated` (a confirmation you must actively pause and perform), `locked` requires no judgment call from you. It is fully enforced by pre-flight validation (Rule 4) and dispatch-time guards (the Phase 3.5 and `/do` guards above) before the developer ever begins work — none of this depends on you noticing or remembering anything mid-verification. Your only obligation is to recognize that a `locked` task always runs (and was always verified) in the current foreground session, and to never suggest or perform a workaround that would route around that guarantee. If you find evidence during verification that a `locked` item was actually run in a backgrounded or worktree-isolated context, treat that as a guard failure worth flagging (see Rapport System), not something to silently pass.
220
+
162
221
  ### Invoked for analysis or comparison testing
163
222
  When invoked to run an analysis or comparison:
164
223
 
@@ -254,12 +313,13 @@ Update the status directly on the scrum board after each test run. Follow the fi
254
313
 
255
314
  ### Scrum Board Concurrency Control
256
315
 
257
- Before writing to any scrum board file, follow this locking protocol:
316
+ Before writing to any scrum board file, wrap the write through `scripts/with-lock.sh` — do not read/write a `.lock` file by hand. The script acquires an atomic, cross-platform (Linux + macOS) exclusive lock keyed to the target file, runs the wrapped write, and always releases the lock afterward, on success or failure:
258
317
 
259
- 1. Check for a `<filename>.lock` file adjacent to the target file.
260
- 2. If the lock file exists and is less than 60 seconds old — wait 10 seconds and retry once. If still locked, abort and write a problem rapport.
261
- 3. If no lock exists (or it is stale, older than 60 seconds) — create the lock file, perform the write, then delete the lock file.
262
- 4. Always delete the lock file in both success and error paths.
318
+ ```bash
319
+ scripts/with-lock.sh <target-file> -- <command-that-performs-the-write>
320
+ ```
321
+
322
+ If the script exits non-zero (it could not acquire the lock within its timeout), it never ran the write — abort and write a problem rapport rather than retrying the write outside the script or bypassing it. See `templates/SCRUM_BOARD_SCHEMA.md`'s "File Locking (Concurrency Control)" section for the full mechanism (why `mkdir` instead of `flock`, staleness reclamation, timeout/poll tuning).
263
323
 
264
324
  ---
265
325
 
@@ -279,6 +339,28 @@ After every status update to a task or story, check whether a parent rollup is w
279
339
 
280
340
  ## Rapport System
281
341
 
342
+ ### Commit the rapport immediately
343
+ A rapport is the only record of a finding until it is committed — an untracked file does not survive `git clean`, and if the parent story ends up blocked on a human, the exposure window is unbounded rather than the few hours a normal rollup takes.
344
+
345
+ Immediately after writing any rapport file (problem or analysis), commit it yourself, in the same session, before doing anything else with it:
346
+
347
+ - Stage **only the rapport file itself, by explicit path** — e.g. `git add project/rapports/problems/<file>.md`. Never `git add -A` or `git add .` for this commit. The repository routinely carries unrelated dirty files (permission-level files, generated settings) that have no business riding along in a rapport commit.
348
+ - Commit it as **its own standalone commit** — do not fold it into a status-update commit, a rollup commit, or any other commit. The rapport's commit message should name the finding, e.g. `chore(<E##_S##_T##>): add rapport — <short description>`.
349
+ - Do this before moving on to the next step of the workflow (status update, rollup trigger, etc.), so the rapport is durable the instant it exists on disk.
350
+
351
+ ### Verification commit target
352
+ Any commit you make while verifying a task — a fix-up, a test file, a rapport, anything — must land on the branch you are verifying, inside that task's worktree. Never commit to `main` or to any branch other than the one under test. A `pre-commit` hook installed at worktree creation (see `scripts/install-worktree-commit-guard.sh`, E37_S02_T01) rejects commits made on the wrong branch as a mechanical backstop, but do not rely on the hook alone — always confirm you are on the correct branch (`git branch --show-current`) before committing.
353
+
354
+ ### Escalation rapports (`crucial_escalation`)
355
+
356
+ **Trigger — non-blocking.** During verification, you discover something that makes the item riskier than its current `crucial_level` reflects (or riskier than warranted by having no `crucial_level` at all) — e.g. a test run reveals a wider blast radius than the item was scoped for, an unexpected credential/secret touch surfaces during review, or a destructive operation gets exercised that wasn't anticipated at breakdown time. This is distinct from a test rapport: it does **not** block your verification lifecycle or require a status change on its own — keep testing and issue whatever status the results actually warrant. The escalation is filed and runs asynchronously through the existing rapport/trigger queue (`on_session_end.sh` → `scrum_triggers.jsonl`) rather than as a synchronous interrupt — no live pause-and-confirm channel exists for a backgrounded subagent (see `agents/developer.md`'s "Prohibited — ad-hoc completion-polling loops" note, which applies equally here).
357
+
358
+ **Concrete-reason requirement.** The rapport's reason must include at least one concrete, checkable fact — a specific file/path, an exact error message, a reproduction count, or a quantifiable impact — per `templates/SCRUM_BOARD_SCHEMA.md`'s `crucial_escalation` subsection. A generic statement like "this seems risky" is not acceptable and will be rejected by scrum-master at review time; do not file one expecting it to be actioned. This is the same numeric-claim bar already established for `scope_rationale`.
359
+
360
+ **Never write `crucial_level` yourself.** Regardless of how confident you are that the escalation is warranted, you must never write `crucial_level`, `crucial_set_by`, or `crucial_note` to any board file directly — not even alongside a status update you are otherwise authorized to make. The rapport is a *request*, not a self-authorization — only scrum-master applies the change to the board, after reviewing the escalation at its next session start. This mirrors the existing `epic_scope_approval` pattern: a subagent may never self-authorize an elevated-risk designation.
361
+
362
+ **Mechanism.** Use `templates/PROBLEM_RAPPORT_TEMPLATE.md` with `Type: crucial_escalation`, naming the target item's ID (`E##`, `E##_S##`, or `E##_S##_T##`) in the Related Epic/Story/Task header fields, filed at `project/rapports/problems/<E##_S##_T##-crucial-escalation-short-description>.md`. Commit it immediately per "Commit the rapport immediately" above — no new commit convention applies.
363
+
282
364
  ### Test rapports (unresolved findings)
283
365
  Write a test rapport when there are unresolved findings, errors, or issues from a test run.
284
366
 
@@ -287,7 +369,7 @@ Location:
287
369
  project/rapports/problems/<E##_S##_T##-short-problem-description>.md
288
370
  ```
289
371
 
290
- Create folders if they do not exist. Follow the rapport template at `templates/PROBLEM_RAPPORT_TEMPLATE.md`.
372
+ Create folders if they do not exist. Follow the rapport template at `templates/PROBLEM_RAPPORT_TEMPLATE.md`. Commit it immediately per "Commit the rapport immediately" above.
291
373
 
292
374
  ### IGNORE.md — skipping resolved rapports
293
375
  During any test run or rapport scan, **skip all files whose name ends in `.IGNORE.md`**. These have been reviewed and explicitly dismissed by the developer. Do not re-flag, re-report, or reference them as open findings.
@@ -307,7 +389,7 @@ Location:
307
389
  project/rapports/analysis/<E##_S##-short-analysis-description>.md
308
390
  ```
309
391
 
310
- Create folders if they do not exist. Follow the same template structure as test rapports, adapted for analysis findings and conclusions.
392
+ Create folders if they do not exist. Follow the same template structure as test rapports, adapted for analysis findings and conclusions. Commit it immediately per "Commit the rapport immediately" above.
311
393
 
312
394
  ---
313
395
 
@@ -7,8 +7,10 @@
7
7
  # 2. Detect new problem rapports using a manifest (not a timestamp)
8
8
  # and write trigger payloads to the scrum master queue
9
9
  # 3. Write a status-review trigger to the scrum master queue
10
- # 4. Read .session_handoff.json (if present) and route the assignment
11
- # to the correct next agent queue, then delete the handoff file
10
+ # 4. Process every pending file under project/queue/handoffs/ (if any),
11
+ # routing each one's assignment to the correct next agent queue, then
12
+ # deleting that individual file immediately after it is routed — so a
13
+ # problem with one file never blocks or loses any of the others
12
14
  #
13
15
  # Pipeline routing (section 4):
14
16
  # scrum-master planning_complete → developer_triggers.jsonl
@@ -26,18 +28,27 @@ source "$(git rev-parse --show-toplevel)/lib/resolve-project-dir.sh"
26
28
 
27
29
  PROJECT_DIR="$JENGA_PROJECT_DIR"
28
30
  RAPPORT_DIR="$PROJECT_DIR/project/rapports/problems"
29
- MANIFEST="$PROJECT_DIR/.claude/hooks/.rapport_manifest.json"
31
+ # NOTE: this manifest lives under project/ (not .claude/) because .claude/
32
+ # and .agents/ are generated build outputs, clobbered on every /distribute
33
+ # or /self-sync run — see CLAUDE.md. Canonical, committed state must not
34
+ # live inside a distribution target. Seed generated via, and kept in sync
35
+ # by, scripts/generate-rapport-manifest.sh (see section 2 below).
36
+ MANIFEST="$PROJECT_DIR/project/data/rapport_manifest.json"
30
37
  QUEUE_DIR="$PROJECT_DIR/project/queue"
31
38
  QUEUE_FILE="$QUEUE_DIR/scrum_triggers.jsonl"
32
39
  DEV_QUEUE="$QUEUE_DIR/developer_triggers.jsonl"
33
40
  TESTER_QUEUE="$QUEUE_DIR/tester_triggers.jsonl"
34
- HANDOFF_FILE="$QUEUE_DIR/.session_handoff.json"
41
+ # Per-session handoff files (E37_S01_T01) replace the old single-slot
42
+ # .session_handoff.json. Every file under this directory is processed in
43
+ # section 4 below; see templates/SCRUM_BOARD_SCHEMA.md's handoffs/ section
44
+ # for the write-side filename convention.
45
+ HANDOFF_DIR="$QUEUE_DIR/handoffs"
35
46
  EVENTS_LOG="$PROJECT_DIR/project/logs/events.json"
36
47
  AGENT="${JENGA_AGENT_TYPE:-unknown}"
37
48
  SESSION_ID="${JENGA_SESSION_ID:-}"
38
49
  TIMESTAMP=$(date -u +"%Y-%m-%dT%H:%M:%SZ")
39
50
 
40
- mkdir -p "$PROJECT_DIR/project/logs" "$QUEUE_DIR"
51
+ mkdir -p "$PROJECT_DIR/project/logs" "$QUEUE_DIR" "$HANDOFF_DIR"
41
52
 
42
53
  # --- 1. Log sender object ---
43
54
 
@@ -58,18 +69,55 @@ SENDER=$(jq -n \
58
69
  }
59
70
  }')
60
71
 
61
- if [ -f "$EVENTS_LOG" ]; then
62
- jq --argjson entry "$SENDER" '. += [$entry]' "$EVENTS_LOG" > /tmp/events_tmp.json \
63
- && mv /tmp/events_tmp.json "$EVENTS_LOG"
72
+ # The read-modify-write below is wrapped in scripts/with-lock.sh (E37_S01_T03)
73
+ # so concurrent on_session_end.sh invocations never interleave their read and
74
+ # write of $EVENTS_LOG — see E37_S01_T04 and the two rapports it links
75
+ # (project/rapports/problems/E37_S01-concurrent-session-end-race-verification-gap.md,
76
+ # project/rapports/problems/E39_S03_T05-events-json-concurrent-write-corruption.md)
77
+ # for the concurrent-invocation data loss (events.json truncated to 0 bytes)
78
+ # this replaces. The temp file used for the atomic `mv` is created via
79
+ # `mktemp` in the SAME directory as $EVENTS_LOG (not a shared
80
+ # /tmp/events_tmp.json path) so it is (a) unique per invocation — no two
81
+ # concurrent processes can ever collide on the same temp filename — and
82
+ # (b) on the same filesystem as the destination, so the final `mv` stays an
83
+ # atomic rename rather than a cross-filesystem copy. The update logic is
84
+ # captured as a standalone script string (via a quoted heredoc, so jq's own
85
+ # `$entry` and quoting pass through untouched by this outer shell) and run
86
+ # through `bash -c` with $EVENTS_LOG/$SENDER passed as positional args
87
+ # rather than interpolated into the script text, avoiding fragile
88
+ # nested-quote escaping while keeping this fix self-contained in this file.
89
+ APPEND_EVENT_SCRIPT=$(cat <<'EOS'
90
+ events_log="$1"
91
+ sender_json="$2"
92
+ tmp_file=$(mktemp "$(dirname "$events_log")/events_tmp.XXXXXX") || exit 1
93
+ if [ -s "$events_log" ]; then
94
+ jq --argjson entry "$sender_json" '. += [$entry]' "$events_log" > "$tmp_file" \
95
+ && mv "$tmp_file" "$events_log"
64
96
  else
65
- echo "[$SENDER]" > "$EVENTS_LOG"
97
+ printf '[%s]' "$sender_json" > "$tmp_file" \
98
+ && mv "$tmp_file" "$events_log"
66
99
  fi
100
+ rc=$?
101
+ [ -f "$tmp_file" ] && rm -f "$tmp_file"
102
+ exit "$rc"
103
+ EOS
104
+ )
105
+
106
+ "$PROJECT_DIR/scripts/with-lock.sh" "$EVENTS_LOG" -- bash -c "$APPEND_EVENT_SCRIPT" _ "$EVENTS_LOG" "$SENDER"
67
107
 
68
108
  # --- 2. Manifest-based rapport detection ---
69
109
  # Use a JSON array of known filenames instead of a mtime sentinel file.
70
110
  # This prevents silently missing rapports written before the session ends
71
111
  # but after the sentinel was last touched.
112
+ #
113
+ # The manifest itself is committed to the repo (seeded and kept in sync via
114
+ # scripts/generate-rapport-manifest.sh), so on a normal checkout it already
115
+ # exists and reflects the rapports present at commit time. The bootstrap
116
+ # below is a non-destructive fallback for the genuinely unexpected case
117
+ # where the file is missing (e.g. project/data/ was pruned) — it does not
118
+ # run on a fresh clone with the manifest intact.
72
119
 
120
+ mkdir -p "$(dirname "$MANIFEST")"
73
121
  if [ ! -f "$MANIFEST" ]; then
74
122
  echo "[]" > "$MANIFEST"
75
123
  fi
@@ -77,12 +125,28 @@ fi
77
125
  if [ -d "$RAPPORT_DIR" ]; then
78
126
  # Collect current rapport files (exclude .IGNORE.md files — already resolved)
79
127
  # Results are stored in a bash array to avoid word-splitting on paths.
80
- mapfile -t CURRENT_FILES < <(find "$RAPPORT_DIR" -name "*.md" ! -name "*.IGNORE.md" 2>/dev/null | sort)
128
+ # Uses a portable `while read` loop rather than mapfile/readarray: macOS
129
+ # ships bash 3.2 (no mapfile support), and this hook must run there —
130
+ # matches the convention already established in scripts/smoke-harness.sh,
131
+ # skills/publish/scripts/generate_release_notes.sh, and
132
+ # skills/publish/scripts/finalize_changelog.sh. Fixed incidentally here
133
+ # because this task's acceptance criteria require the hook to actually
134
+ # execute end-to-end (mapfile silently failed on stock macOS bash,
135
+ # leaving CURRENT_FILES empty and masking real detection results).
136
+ CURRENT_FILES=()
137
+ while IFS= read -r line; do
138
+ [ -n "$line" ] && CURRENT_FILES+=("$line")
139
+ done < <(find "$RAPPORT_DIR" -name "*.md" ! -name "*.IGNORE.md" 2>/dev/null | sort)
81
140
 
82
- # Identify new files not present in the manifest
141
+ # Identify new files not present in the manifest. The manifest stores
142
+ # paths relative to $PROJECT_DIR (portable across clones/worktrees), so
143
+ # each absolute CURRENT_FILES entry is relativized before the lookup.
144
+ # NEW_FILES itself stays absolute — it feeds rapport_files in the
145
+ # trigger below, which the scrum master reads directly.
83
146
  NEW_FILES=()
84
147
  for file in "${CURRENT_FILES[@]}"; do
85
- known=$(jq --arg f "$file" 'index($f) != null' "$MANIFEST" 2>/dev/null)
148
+ rel="${file#"$PROJECT_DIR"/}"
149
+ known=$(jq --arg f "$rel" 'index($f) != null' "$MANIFEST" 2>/dev/null)
86
150
  if [ "$known" != "true" ]; then
87
151
  NEW_FILES+=("$file")
88
152
  fi
@@ -109,8 +173,11 @@ if [ -d "$RAPPORT_DIR" ]; then
109
173
 
110
174
  echo "$TRIGGER" >> "$QUEUE_FILE"
111
175
 
112
- # Update the manifest to include all current files
113
- printf '%s\n' "${CURRENT_FILES[@]}" | jq -R . | jq -s . > "$MANIFEST"
176
+ # Update the manifest to include all current files. Delegates to the
177
+ # shared generator script (rather than re-inlining the same jq
178
+ # pipeline) so the runtime update and the committed seed can never
179
+ # drift in how they compute "current rapport files".
180
+ bash "$PROJECT_DIR/scripts/generate-rapport-manifest.sh" "$MANIFEST" >/dev/null
114
181
  fi
115
182
  fi
116
183
 
@@ -129,15 +196,97 @@ TRIGGER=$(jq -n \
129
196
 
130
197
  echo "$TRIGGER" >> "$QUEUE_FILE"
131
198
 
199
+ # --- Helper: has this handoff's referenced task already reached a
200
+ # terminal board status? ---------------------------------------------------
201
+ # Guards against resurrecting a stale handoff file (e.g. one that was
202
+ # accidentally committed to git alongside unrelated work, or one left over
203
+ # from before this directory was ever consumed) as a fresh routing signal
204
+ # for work that has already been fully verified. Terminal here means "no
205
+ # further routing action should occur": Passed / Passed with remarks /
206
+ # Rejected / Done are all end states, and Blocked is included because a
207
+ # blocked task must not be silently re-touched by any agent per
208
+ # agents/developer.md's Blocked-halt contract (only a human may clear it).
209
+ # Deliberately fails open (returns 1 / "not terminal") when the task file
210
+ # can't be found or has no readable status, so a lookup miss falls back to
211
+ # today's behavior (route it) rather than silently dropping a handoff whose
212
+ # task genuinely can't be identified.
213
+ is_task_terminal() {
214
+ local task_id="$1" task_file task_status
215
+ task_file=$(find "$PROJECT_DIR/project/board/tasks" -maxdepth 1 -iname "${task_id}_*.md" 2>/dev/null | head -1)
216
+ [ -z "$task_file" ] && return 1
217
+ task_status=$(awk -F': ' '/^status:/ {print $2; exit}' "$task_file" 2>/dev/null)
218
+ case "$task_status" in
219
+ Passed|"Passed with remarks"|Rejected|Done|Blocked) return 0 ;;
220
+ *) return 1 ;;
221
+ esac
222
+ }
223
+
132
224
  # --- 4. Session handoff routing ---
133
- # Each agent writes .session_handoff.json before its session ends.
134
- # This section reads that file and routes the work to the correct next
135
- # agent queue so the pipeline continues automatically.
225
+ # Each agent writes a per-session handoff file under project/queue/handoffs/
226
+ # before its session ends (see E37_S01_T01). This section processes every
227
+ # file currently present in that directory — not just one fixed path — so
228
+ # concurrent sessions ending close together each get routed instead of the
229
+ # last writer silently clobbering the others. Every file is routed through
230
+ # the same per-agent logic below, then deleted individually right after
231
+ # routing, so a problem with one file can't block or lose any of the rest.
232
+ #
233
+ # The glob is intentionally generic (*.json, not a pattern anchored to the
234
+ # <agent>-<session_id>-<task_id> convention) so it also picks up older
235
+ # ad-hoc-named files written before that convention existed — routing below
236
+ # only ever reads the .agent/.status fields from file contents, never the
237
+ # filename, so this is safe.
238
+ #
239
+ # Before routing, a developer/tester handoff whose referenced task is
240
+ # already terminal on the board is treated as stale (deleted, not routed)
241
+ # rather than as a live signal — see is_task_terminal() above. This closes
242
+ # a real regression found during E37_S01_T02 testing: this directory can
243
+ # accumulate committed leftover files for already-shipped work (nothing
244
+ # consumed them before this task existed), and a bare generic glob would
245
+ # otherwise resurrect all of them as fresh test_assignment/story_rollup
246
+ # triggers the moment this hook first runs against such a directory.
247
+ for HANDOFF_FILE in "$HANDOFF_DIR"/*.json; do
248
+ # Guard against the glob matching nothing (no nullglob dependency, so
249
+ # this stays portable to stock macOS bash 3.2 — same rationale as the
250
+ # portable `while read` loop in section 2 above).
251
+ [ -e "$HANDOFF_FILE" ] || continue
252
+
253
+ # Atomic per-file claim (E37_S01_T04) — closes the TOCTOU window between
254
+ # this per-process glob snapshot and the eventual delete at the bottom of
255
+ # the loop. `mv` between two paths on the same filesystem is a rename(2)
256
+ # syscall: the kernel guarantees that when multiple concurrent processes
257
+ # race to rename the same source path, exactly one succeeds and every
258
+ # other attempt fails because the source no longer exists. No GNU-only
259
+ # flags are involved, so this is portable to macOS/BSD as well as Linux.
260
+ # Without this claim, concurrent invocations could each see the same
261
+ # not-yet-deleted file, read it, and route it before either deleted it —
262
+ # confirmed in project/rapports/problems/E37_S01-concurrent-session-end-race-verification-gap.md
263
+ # as duplicate (2x/2x/4x observed) trigger entries for the same handoff.
264
+ # If the rename fails, another concurrent process already claimed this
265
+ # exact file — skip it without reading, routing, or deleting anything;
266
+ # that other process now owns it.
267
+ CLAIMED_HANDOFF_FILE="${HANDOFF_FILE}.claimed.$$"
268
+ if ! mv "$HANDOFF_FILE" "$CLAIMED_HANDOFF_FILE" 2>/dev/null; then
269
+ echo "[on_session_end] handoff already claimed by a concurrent invocation — skipping $(basename "$HANDOFF_FILE")"
270
+ continue
271
+ fi
272
+ HANDOFF_FILE="$CLAIMED_HANDOFF_FILE"
136
273
 
137
- if [ -f "$HANDOFF_FILE" ]; then
138
274
  HANDOFF_AGENT=$(jq -r '.agent // empty' "$HANDOFF_FILE" 2>/dev/null)
139
275
  HANDOFF_STATUS=$(jq -r '.status // empty' "$HANDOFF_FILE" 2>/dev/null)
140
276
 
277
+ # Staleness check — only meaningful for developer/tester handoffs, which
278
+ # each reference exactly one task_id. scrum-master's planning_complete
279
+ # handoff carries a task_ids array of just-created tasks that can't
280
+ # already be terminal in practice, so it is not checked here.
281
+ if [ "$HANDOFF_AGENT" = "developer" ] || [ "$HANDOFF_AGENT" = "tester" ]; then
282
+ CHECK_TASK_ID=$(jq -r '.task_id // empty' "$HANDOFF_FILE" 2>/dev/null)
283
+ if [ -n "$CHECK_TASK_ID" ] && is_task_terminal "$CHECK_TASK_ID"; then
284
+ echo "[on_session_end] stale handoff for already-terminal task $CHECK_TASK_ID — skipping routing, deleting $(basename "$HANDOFF_FILE")"
285
+ rm -f "$HANDOFF_FILE"
286
+ continue
287
+ fi
288
+ fi
289
+
141
290
  case "$HANDOFF_AGENT" in
142
291
 
143
292
  scrum-master)
@@ -228,9 +377,11 @@ if [ -f "$HANDOFF_FILE" ]; then
228
377
 
229
378
  esac
230
379
 
231
- # Consume the handoff file — it is single-use
380
+ # Consume this handoff file — it is single-use. Deleted immediately after
381
+ # routing (not batched after the loop) so a later file's failure can never
382
+ # cause an earlier, already-routed file to be left unconsumed or vice versa.
232
383
  rm -f "$HANDOFF_FILE"
233
- fi
384
+ done
234
385
 
235
386
  # --- 5. Todo cleanup ---
236
387
  # Remove project/todo.md if it is effectively empty (only blanks, # Todo, and HTML comments).
@@ -1,4 +1,4 @@
1
- import { pipeline } from "@xenova/transformers";
1
+ import { pipeline } from "@huggingface/transformers";
2
2
 
3
3
  let _pipe = null;
4
4