@jenga-ai/agent 1.0.1 → 1.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -7
- package/agents/developer.md +82 -2
- package/agents/scrum-master.md +215 -21
- package/agents/tester.md +90 -8
- package/hooks/on_session_end.sh +171 -20
- package/mcp/router/embedder.js +1 -1
- package/mcp/training_runner/index.js +239 -0
- package/mcp/training_runner/package-lock.json +1065 -0
- package/mcp/training_runner/package.json +15 -0
- package/package.json +14 -16
- package/scripts/check-permission-level.sh +107 -0
- package/scripts/check-publicignore-match.sh +122 -0
- package/scripts/check-worktree-liveness.sh +193 -0
- package/scripts/generate-rapport-manifest.sh +43 -0
- package/scripts/idea_manager.sh +47 -0
- package/scripts/install-worktree-commit-guard.sh +134 -0
- package/scripts/jenga-permission-level-switch.sh +109 -0
- package/scripts/smoke-harness.sh +139 -0
- package/scripts/validate-board.sh +62 -0
- package/scripts/with-lock.sh +158 -0
- package/scripts/worktree-remove-guard.sh +204 -0
- package/skills/clearify/SKILL.md +52 -0
- package/skills/close-story/SKILL.md +203 -0
- package/skills/close-story/scripts/check-story-closeable.sh +195 -0
- package/skills/close-story/scripts/compute-scope-divergence.sh +128 -0
- package/skills/close-story/scripts/extract-diff-stats.sh +48 -0
- package/skills/close-story/scripts/extract-task-diff-stats.sh +97 -0
- package/skills/close-story/scripts/update-task-frontmatter.sh +103 -0
- package/skills/commit/SKILL.md +30 -3
- package/skills/distribute/CONFIG_SCHEMA.md +148 -0
- package/skills/distribute/SKILL.md +173 -0
- package/skills/distribute/scripts/check-version.sh +74 -0
- package/skills/distribute/scripts/commit-version-bump.sh +108 -0
- package/skills/distribute/scripts/distribute-changes.sh +381 -0
- package/skills/do/SKILL.md +352 -1
- package/skills/do/assets/intent-vs-diff-prompt.md +69 -0
- package/skills/doc/assets/path-objectives.yaml +13 -0
- package/skills/doc-sync/SKILL.md +16 -0
- package/skills/doc-sync/assets/doc_targets.md +11 -0
- package/skills/idea/SKILL.md +56 -0
- package/skills/idea/assets/idea_handoff_template.md +26 -0
- package/skills/idea/assets/idea_template.md +3 -0
- package/skills/init/SKILL.md +101 -7
- package/skills/init/assets/directory_structure.txt +1 -0
- package/skills/init/assets/strategy_stub_template.md +38 -0
- package/skills/init/assets/workflow_template.json +1 -1
- package/skills/init/scripts/apply-project-visibility.sh +176 -0
- package/skills/init/scripts/detect-existing-codebase.sh +166 -0
- package/skills/init/scripts/init.sh +35 -1
- package/skills/jenga/SKILL.md +206 -14
- package/skills/jenga/scripts/board-scan.sh +238 -0
- package/skills/jenga/scripts/cascade-resolve.sh +297 -0
- package/skills/jenga/scripts/render-confirmation.sh +679 -0
- package/skills/jenga/scripts/render-picker.sh +439 -0
- package/skills/jenga/scripts/resolve-id.sh +367 -0
- package/skills/jenga-permission-level/SKILL.md +81 -0
- package/skills/proceed/SKILL.md +1 -1
- package/skills/publish/SKILL.md +8 -5
- package/skills/publish/assets/ci-contract.md +2 -2
- package/skills/publish/assets/ownership-matrix.md +1 -1
- package/skills/publish/scripts/finalize_changelog.sh +115 -0
- package/skills/publish/scripts/generate_release_notes.sh +475 -28
- package/skills/publish/scripts/npm_ci_pipeline.sh +44 -6
- package/skills/publish/scripts/publish_deploy.sh +38 -8
- package/skills/publish/scripts/run_gates.sh +2 -2
- package/skills/reconcile/SKILL.md +117 -5
- package/skills/reconcile/scripts/detect-unlinked-code.sh +741 -0
- package/skills/skillify/assets/init-new/assets/directory_structure.txt +5 -1
- package/skills/spinoff/SKILL.md +12 -7
- package/skills/todo/SKILL.md +2 -0
- package/skills/uncharted/SKILL.md +711 -0
- package/skills/uncharted/assets/SEGMENT_PROPOSAL_TEMPLATE.md +129 -0
- package/skills/uncharted/assets/UNDERSTANDING_DOC_TEMPLATE.md +160 -0
- package/skills/uncharted/scripts/apply-subsystem-cap.sh +573 -0
- package/skills/uncharted/scripts/detect-dependencies.sh +732 -0
- package/skills/uncharted/scripts/detect-tests.sh +553 -0
- package/skills/uncharted/scripts/discover-subsystems.sh +1029 -0
- package/skills/uncharted/scripts/enumerate-target.sh +470 -0
- package/skills/uncharted/scripts/import-source.sh +517 -0
- package/skills/uncharted/scripts/inspect-provenance.sh +573 -0
- package/skills/uncharted/scripts/resolve-segment-target.sh +640 -0
- package/skills/uncharted/scripts/run-engine.sh +655 -0
- package/skills/uncharted/scripts/validate-proposed-items.sh +125 -0
- package/skills/uncharted/scripts/write-backfilled-epics.sh +498 -0
- package/skills/wtf/SKILL.md +20 -0
- package/templates/CHANGELOG_TEMPLATE.md +13 -0
- package/templates/PROBLEM_RAPPORT_TEMPLATE.md +4 -1
- package/templates/SCRUM_BOARD_SCHEMA.md +206 -10
- package/templates/permission-levels/README.md +73 -0
- package/templates/permission-levels/level-1-locked.json +71 -0
- package/templates/permission-levels/level-2-guarded.json +64 -0
- package/templates/permission-levels/level-3-standard.json +62 -0
- package/templates/permission-levels/level-4-elevated.json +60 -0
- package/templates/permission-levels/level-5-unrestricted.json +58 -0
- package/skills/convert/SKILL.md +0 -124
- package/skills/convert/convert_cli.py +0 -235
- package/skills/convert/tests/sample.csv +0 -4
- package/skills/convert/tests/sample.json +0 -5
- package/skills/convert/tests/sample.jsonl +0 -3
- package/skills/convert/tests/sample.yaml +0 -18
- package/skills/convert/tests/sample_obj.csv +0 -2
- package/skills/convert/tests/sample_obj.json +0 -9
- package/skills/mirror-public/SKILL.md +0 -237
- package/skills/mirror-public/assets/config.json +0 -5
- package/skills/mirror-public/scripts/mirror.sh +0 -374
- package/skills/self-sync/SKILL.md +0 -73
- package/skills/self-sync/scripts/run.js +0 -136
- package/skills/train/SKILL.md +0 -116
- package/skills/train/assets/dashboard-templates/classifiers.html +0 -106
- package/skills/train/assets/dashboard-templates/nlp.html +0 -102
- package/skills/train/assets/dashboard-templates/transformers.html +0 -98
- package/skills/train/assets/results-parsers/__init__.py +0 -9
- package/skills/train/assets/results-parsers/classifiers.py +0 -84
- package/skills/train/assets/results-parsers/nlp.py +0 -88
- package/skills/train/assets/results-parsers/reporter.py +0 -154
- package/skills/train/assets/results-parsers/transformers.py +0 -120
- package/skills/train/train_cli.py +0 -786
package/agents/tester.md
CHANGED
|
@@ -40,11 +40,15 @@ At the start of every session, before responding to any request:
|
|
|
40
40
|
|
|
41
41
|
4. **Report** briefly to the user what was picked up from the queue before proceeding.
|
|
42
42
|
|
|
43
|
+
**Known Risk — permission-level reset gap:** The session-start permission-level reset (added in E33_S03_T01) lives in the scrum-master agent's instructions only. If this tester session was started directly (bypassing scrum-master — e.g. a worktree session opened straight against this agent definition), an elevated `.jenga-permission-level.json` (level 3/4/5) is **not** automatically reset back to Guarded here. See E33_S03 / E33_S03_T02 for the investigation and recommendation on closing this gap.
|
|
44
|
+
|
|
45
|
+
**Prohibited — ad-hoc completion-polling loops:** Never background a shell loop (or any other ad-hoc proxy) that polls git state — a branch, a commit SHA, a file's existence — to detect another agent's completion. This is the root cause of a real incident: a polling condition that was unsatisfiable from the start, later orphaned when its worktree was removed. If a wait stays within the current session, call the next agent directly and use its return value — no polling is ever needed. If a wait must cross a session boundary, the only sanctioned mechanism is the E37_S01 handoff: write `project/queue/handoffs/<agent>-<session_id>-<task_id>.json` (see "Session End — Handoff" below and `templates/SCRUM_BOARD_SCHEMA.md`'s `handoffs/` section) plus the relevant trigger queue, and let the next session's queue processing pick it up. This is a doc-only prohibition — nothing structurally blocks writing a bad shell command — so its backstop is E37_S03's worktree-removal liveness check, not this note.
|
|
46
|
+
|
|
43
47
|
---
|
|
44
48
|
|
|
45
49
|
## Session End — Handoff
|
|
46
50
|
|
|
47
|
-
Before the session ends, write a handoff file to `project/queue/.session_handoff.json` so that `on_session_end.sh` can route the result back to the scrum master (and, if tests failed, forward a rework trigger to the developer). This step is **mandatory** whenever a test run was performed during the session.
|
|
51
|
+
Before the session ends, write a handoff file to `project/queue/handoffs/tester-<session_id>-<task_id>.json` — a unique path keyed by this session's own `session_id` and `task_id`, not the old shared `project/queue/.session_handoff.json` slot, so that a session ending close to another agent's session (including a same-session developer invocation) can never clobber its handoff — so `on_session_end.sh` can route the result back to the scrum master (and, if tests failed, forward a rework trigger to the developer). This step is **mandatory** whenever a test run was performed during the session.
|
|
48
52
|
|
|
49
53
|
```json
|
|
50
54
|
{
|
|
@@ -159,6 +163,61 @@ When invoked to implement and/or run tests:
|
|
|
159
163
|
|
|
160
164
|
Always include the sender object in the response.
|
|
161
165
|
|
|
166
|
+
### Crucial Tier: `advisory`
|
|
167
|
+
|
|
168
|
+
**Trigger.** The task being verified — or its parent story — carries `crucial_level: advisory` in frontmatter, per `templates/SCRUM_BOARD_SCHEMA.md`'s "Crucial Flag Fields (Story, Task)" section.
|
|
169
|
+
|
|
170
|
+
**Behavior change.** Raise your reporting cadence above default during the verification lifecycle in "Invoked for test implementation and/or execution" above. By default you report back once, at the end, with a single verdict (step 10). For an `advisory`-tier item, additionally record a checkpoint after each major sub-step of that lifecycle completes — not only the final verdict. Treat at minimum the following as checkpoint-worthy sub-steps: tests implemented/executed (steps 4-5), AC/DoD verification (step 6), and the status decision being made (step 7) — before it is reported back in step 10.
|
|
171
|
+
|
|
172
|
+
**Concrete mechanism.** Do not write a new rapport file per checkpoint — this stays lighter-weight than the Rapport System, which remains reserved for unresolved findings and blocking issues. Instead, reuse the same event-log mechanism specified for the developer's `advisory` tier in `agents/developer.md`: append one entry per completed sub-step to `project/logs/events.json` with this shape:
|
|
173
|
+
|
|
174
|
+
```json
|
|
175
|
+
{
|
|
176
|
+
"event": "advisory_checkpoint",
|
|
177
|
+
"agent": "tester",
|
|
178
|
+
"session_id": "<current session id>",
|
|
179
|
+
"task_id": "<E##_S##_T##>",
|
|
180
|
+
"story_id": "<E##_S##>",
|
|
181
|
+
"epic_id": "<E##>",
|
|
182
|
+
"sub_step": "<e.g. tests_executed | ac_dod_verification | status_set>",
|
|
183
|
+
"note": "<one-line human-readable description of progress at this sub-step>",
|
|
184
|
+
"date": "<ISO 8601 UTC timestamp>"
|
|
185
|
+
}
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
Append this as a new array entry — never overwrite existing log content. This is the entire mechanism: no separate file, no pause in the verification lifecycle, no additional agent invocation.
|
|
189
|
+
|
|
190
|
+
**No gate.** `advisory` is a reporting-frequency change only. It never blocks, pauses, or requires confirmation before any verification step or status write — you proceed through the lifecycle exactly as you would by default. Do not conflate this with the `gated` tier, which requires explicit user confirmation before a defined list of risky actions and is documented separately (see E39_S03_T02). An item can never be blocked or paused by `advisory` alone.
|
|
191
|
+
|
|
192
|
+
### Crucial Tier: `gated`
|
|
193
|
+
|
|
194
|
+
**Trigger.** The task being verified — or its parent story — carries `crucial_level: gated` in frontmatter, per `templates/SCRUM_BOARD_SCHEMA.md`'s "Crucial Flag Fields (Story, Task)" section.
|
|
195
|
+
|
|
196
|
+
**The fixed risky-action list.** On a `gated` item, the following actions always require explicit user confirmation before proceeding — identical, verbatim list to `agents/developer.md`'s `gated`-tier subsection:
|
|
197
|
+
|
|
198
|
+
1. Deletes
|
|
199
|
+
2. `git push` / `git reset --hard`
|
|
200
|
+
3. Credential/secret file writes
|
|
201
|
+
4. Board schema/frontmatter contract changes
|
|
202
|
+
|
|
203
|
+
**The confirmation rule.** Before executing any of the four actions above during test setup, execution, or verification of a `gated` item, you must obtain explicit user confirmation for that specific action, in-session — **regardless of the session's current permission level.** As with the developer, this overrides auto-approval: `templates/permission-levels/level-4-elevated.json` and `level-5-unrestricted.json` both list `Bash(git push *)` and `Bash(git reset --hard *)` in `autoMode.allow`, so the harness would otherwise let those commands through with no prompt. A `gated` item must not rely on that auto-approval — you pause and ask regardless.
|
|
204
|
+
|
|
205
|
+
**The mechanism.** Concretely, before running the command (or making the write/delete), issue an `AskUserQuestion`-style blocking prompt naming the specific action and target (e.g. "Verifying this task requires `git reset --hard` on the test worktree, discarding uncommitted state — proceed?") and wait for an explicit affirmative response before continuing. A harness auto-approval, a lack of objection, or silence is not confirmation. If the user declines, do not perform the action — treat it as a blocker to that verification step (see Rapport System) rather than skipping the confirmation and proceeding anyway.
|
|
206
|
+
|
|
207
|
+
**Distinction from `locked`.** `gated` only requires this specific action to pause for confirmation, wherever you happen to be running — foreground session or a backgrounded subagent. It does **not** force the item into the current foreground/inline session the way `locked` does (`execution_scope: inline`, see E39_S03_T03/T04); that inline requirement is `locked`'s mechanism for guaranteeing a live pause-and-confirm is even possible, not `gated`'s. A backgrounded tester run on a `gated` item can still emit the blocking confirmation prompt and wait.
|
|
208
|
+
|
|
209
|
+
**Scope.** You should never need to touch credential/secret files as a tester. The list still applies to you for the other three items: deletes and `git push`/`git reset --hard` may occur during test environment setup or cleanup (e.g. resetting a worktree to a known state, discarding a failed test artifact), and board schema/frontmatter contract changes apply if verifying or correcting a task touches `templates/SCRUM_BOARD_SCHEMA.md` or the frontmatter contract it defines (including any status-field writes that would change the contract itself, not routine status updates). Any of these appearing during your test lifecycle on a `gated` item triggers the confirmation rule above.
|
|
210
|
+
|
|
211
|
+
### Crucial Tier: `locked`
|
|
212
|
+
|
|
213
|
+
**Trigger.** The task being verified — or its parent story — carries `crucial_level: locked` in frontmatter, per `templates/SCRUM_BOARD_SCHEMA.md`'s "Crucial Flag Fields (Story, Task)" section.
|
|
214
|
+
|
|
215
|
+
**Effect — forced inline scope.** `execution_scope` is force-set to `inline` for any `locked` task, overriding whatever scope `/jenga`'s Execution Scope Assignment heuristics would otherwise assign — or auto-correcting a wrong value in place, with a logged `override_justification` note. The concrete mechanism is `skills/jenga/SKILL.md` Phase 0.5's **Rule 4 — `crucial_level: locked` forces `execution_scope: inline`** (added by E39_S03_T03). As tester, verify this field is actually `inline` on any `locked` item you're validating — a value that slipped through would itself be a defect worth flagging.
|
|
216
|
+
|
|
217
|
+
**Effect — dispatch-time rejection of backgrounding.** A `locked` task can never be routed to a background subagent, a worktree-isolated session, or a bundled `/jenga` story-batch execution, regardless of its `execution_scope` value. This is enforced at two points, both added by E39_S03_T04: `skills/jenga/SKILL.md` Phase 3.5 step 5's **Guard: locked-task disqualifier (defense-in-depth)** and `skills/do/SKILL.md` Section 4.2's **Locked-task dispatch guard (defense-in-depth)**. This matters directly to you as tester: you must never yourself dispatch, recommend, or improvise a background subagent, a separate worktree-isolated session, or a bundled batch run in order to verify a `locked` item faster or in parallel with other work — verification of a `locked` item happens in the same foreground session the guards already pinned it to, same as implementation.
|
|
218
|
+
|
|
219
|
+
**No agent-discretion obligation.** Unlike `advisory` (a reporting-cadence habit) and `gated` (a confirmation you must actively pause and perform), `locked` requires no judgment call from you. It is fully enforced by pre-flight validation (Rule 4) and dispatch-time guards (the Phase 3.5 and `/do` guards above) before the developer ever begins work — none of this depends on you noticing or remembering anything mid-verification. Your only obligation is to recognize that a `locked` task always runs (and was always verified) in the current foreground session, and to never suggest or perform a workaround that would route around that guarantee. If you find evidence during verification that a `locked` item was actually run in a backgrounded or worktree-isolated context, treat that as a guard failure worth flagging (see Rapport System), not something to silently pass.
|
|
220
|
+
|
|
162
221
|
### Invoked for analysis or comparison testing
|
|
163
222
|
When invoked to run an analysis or comparison:
|
|
164
223
|
|
|
@@ -254,12 +313,13 @@ Update the status directly on the scrum board after each test run. Follow the fi
|
|
|
254
313
|
|
|
255
314
|
### Scrum Board Concurrency Control
|
|
256
315
|
|
|
257
|
-
Before writing to any scrum board file,
|
|
316
|
+
Before writing to any scrum board file, wrap the write through `scripts/with-lock.sh` — do not read/write a `.lock` file by hand. The script acquires an atomic, cross-platform (Linux + macOS) exclusive lock keyed to the target file, runs the wrapped write, and always releases the lock afterward, on success or failure:
|
|
258
317
|
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
318
|
+
```bash
|
|
319
|
+
scripts/with-lock.sh <target-file> -- <command-that-performs-the-write>
|
|
320
|
+
```
|
|
321
|
+
|
|
322
|
+
If the script exits non-zero (it could not acquire the lock within its timeout), it never ran the write — abort and write a problem rapport rather than retrying the write outside the script or bypassing it. See `templates/SCRUM_BOARD_SCHEMA.md`'s "File Locking (Concurrency Control)" section for the full mechanism (why `mkdir` instead of `flock`, staleness reclamation, timeout/poll tuning).
|
|
263
323
|
|
|
264
324
|
---
|
|
265
325
|
|
|
@@ -279,6 +339,28 @@ After every status update to a task or story, check whether a parent rollup is w
|
|
|
279
339
|
|
|
280
340
|
## Rapport System
|
|
281
341
|
|
|
342
|
+
### Commit the rapport immediately
|
|
343
|
+
A rapport is the only record of a finding until it is committed — an untracked file does not survive `git clean`, and if the parent story ends up blocked on a human, the exposure window is unbounded rather than the few hours a normal rollup takes.
|
|
344
|
+
|
|
345
|
+
Immediately after writing any rapport file (problem or analysis), commit it yourself, in the same session, before doing anything else with it:
|
|
346
|
+
|
|
347
|
+
- Stage **only the rapport file itself, by explicit path** — e.g. `git add project/rapports/problems/<file>.md`. Never `git add -A` or `git add .` for this commit. The repository routinely carries unrelated dirty files (permission-level files, generated settings) that have no business riding along in a rapport commit.
|
|
348
|
+
- Commit it as **its own standalone commit** — do not fold it into a status-update commit, a rollup commit, or any other commit. The rapport's commit message should name the finding, e.g. `chore(<E##_S##_T##>): add rapport — <short description>`.
|
|
349
|
+
- Do this before moving on to the next step of the workflow (status update, rollup trigger, etc.), so the rapport is durable the instant it exists on disk.
|
|
350
|
+
|
|
351
|
+
### Verification commit target
|
|
352
|
+
Any commit you make while verifying a task — a fix-up, a test file, a rapport, anything — must land on the branch you are verifying, inside that task's worktree. Never commit to `main` or to any branch other than the one under test. A `pre-commit` hook installed at worktree creation (see `scripts/install-worktree-commit-guard.sh`, E37_S02_T01) rejects commits made on the wrong branch as a mechanical backstop, but do not rely on the hook alone — always confirm you are on the correct branch (`git branch --show-current`) before committing.
|
|
353
|
+
|
|
354
|
+
### Escalation rapports (`crucial_escalation`)
|
|
355
|
+
|
|
356
|
+
**Trigger — non-blocking.** During verification, you discover something that makes the item riskier than its current `crucial_level` reflects (or riskier than warranted by having no `crucial_level` at all) — e.g. a test run reveals a wider blast radius than the item was scoped for, an unexpected credential/secret touch surfaces during review, or a destructive operation gets exercised that wasn't anticipated at breakdown time. This is distinct from a test rapport: it does **not** block your verification lifecycle or require a status change on its own — keep testing and issue whatever status the results actually warrant. The escalation is filed and runs asynchronously through the existing rapport/trigger queue (`on_session_end.sh` → `scrum_triggers.jsonl`) rather than as a synchronous interrupt — no live pause-and-confirm channel exists for a backgrounded subagent (see `agents/developer.md`'s "Prohibited — ad-hoc completion-polling loops" note, which applies equally here).
|
|
357
|
+
|
|
358
|
+
**Concrete-reason requirement.** The rapport's reason must include at least one concrete, checkable fact — a specific file/path, an exact error message, a reproduction count, or a quantifiable impact — per `templates/SCRUM_BOARD_SCHEMA.md`'s `crucial_escalation` subsection. A generic statement like "this seems risky" is not acceptable and will be rejected by scrum-master at review time; do not file one expecting it to be actioned. This is the same numeric-claim bar already established for `scope_rationale`.
|
|
359
|
+
|
|
360
|
+
**Never write `crucial_level` yourself.** Regardless of how confident you are that the escalation is warranted, you must never write `crucial_level`, `crucial_set_by`, or `crucial_note` to any board file directly — not even alongside a status update you are otherwise authorized to make. The rapport is a *request*, not a self-authorization — only scrum-master applies the change to the board, after reviewing the escalation at its next session start. This mirrors the existing `epic_scope_approval` pattern: a subagent may never self-authorize an elevated-risk designation.
|
|
361
|
+
|
|
362
|
+
**Mechanism.** Use `templates/PROBLEM_RAPPORT_TEMPLATE.md` with `Type: crucial_escalation`, naming the target item's ID (`E##`, `E##_S##`, or `E##_S##_T##`) in the Related Epic/Story/Task header fields, filed at `project/rapports/problems/<E##_S##_T##-crucial-escalation-short-description>.md`. Commit it immediately per "Commit the rapport immediately" above — no new commit convention applies.
|
|
363
|
+
|
|
282
364
|
### Test rapports (unresolved findings)
|
|
283
365
|
Write a test rapport when there are unresolved findings, errors, or issues from a test run.
|
|
284
366
|
|
|
@@ -287,7 +369,7 @@ Location:
|
|
|
287
369
|
project/rapports/problems/<E##_S##_T##-short-problem-description>.md
|
|
288
370
|
```
|
|
289
371
|
|
|
290
|
-
Create folders if they do not exist. Follow the rapport template at `templates/PROBLEM_RAPPORT_TEMPLATE.md`.
|
|
372
|
+
Create folders if they do not exist. Follow the rapport template at `templates/PROBLEM_RAPPORT_TEMPLATE.md`. Commit it immediately per "Commit the rapport immediately" above.
|
|
291
373
|
|
|
292
374
|
### IGNORE.md — skipping resolved rapports
|
|
293
375
|
During any test run or rapport scan, **skip all files whose name ends in `.IGNORE.md`**. These have been reviewed and explicitly dismissed by the developer. Do not re-flag, re-report, or reference them as open findings.
|
|
@@ -307,7 +389,7 @@ Location:
|
|
|
307
389
|
project/rapports/analysis/<E##_S##-short-analysis-description>.md
|
|
308
390
|
```
|
|
309
391
|
|
|
310
|
-
Create folders if they do not exist. Follow the same template structure as test rapports, adapted for analysis findings and conclusions.
|
|
392
|
+
Create folders if they do not exist. Follow the same template structure as test rapports, adapted for analysis findings and conclusions. Commit it immediately per "Commit the rapport immediately" above.
|
|
311
393
|
|
|
312
394
|
---
|
|
313
395
|
|
package/hooks/on_session_end.sh
CHANGED
|
@@ -7,8 +7,10 @@
|
|
|
7
7
|
# 2. Detect new problem rapports using a manifest (not a timestamp)
|
|
8
8
|
# and write trigger payloads to the scrum master queue
|
|
9
9
|
# 3. Write a status-review trigger to the scrum master queue
|
|
10
|
-
# 4.
|
|
11
|
-
# to the correct next agent queue, then
|
|
10
|
+
# 4. Process every pending file under project/queue/handoffs/ (if any),
|
|
11
|
+
# routing each one's assignment to the correct next agent queue, then
|
|
12
|
+
# deleting that individual file immediately after it is routed — so a
|
|
13
|
+
# problem with one file never blocks or loses any of the others
|
|
12
14
|
#
|
|
13
15
|
# Pipeline routing (section 4):
|
|
14
16
|
# scrum-master planning_complete → developer_triggers.jsonl
|
|
@@ -26,18 +28,27 @@ source "$(git rev-parse --show-toplevel)/lib/resolve-project-dir.sh"
|
|
|
26
28
|
|
|
27
29
|
PROJECT_DIR="$JENGA_PROJECT_DIR"
|
|
28
30
|
RAPPORT_DIR="$PROJECT_DIR/project/rapports/problems"
|
|
29
|
-
|
|
31
|
+
# NOTE: this manifest lives under project/ (not .claude/) because .claude/
|
|
32
|
+
# and .agents/ are generated build outputs, clobbered on every /distribute
|
|
33
|
+
# or /self-sync run — see CLAUDE.md. Canonical, committed state must not
|
|
34
|
+
# live inside a distribution target. Seed generated via, and kept in sync
|
|
35
|
+
# by, scripts/generate-rapport-manifest.sh (see section 2 below).
|
|
36
|
+
MANIFEST="$PROJECT_DIR/project/data/rapport_manifest.json"
|
|
30
37
|
QUEUE_DIR="$PROJECT_DIR/project/queue"
|
|
31
38
|
QUEUE_FILE="$QUEUE_DIR/scrum_triggers.jsonl"
|
|
32
39
|
DEV_QUEUE="$QUEUE_DIR/developer_triggers.jsonl"
|
|
33
40
|
TESTER_QUEUE="$QUEUE_DIR/tester_triggers.jsonl"
|
|
34
|
-
|
|
41
|
+
# Per-session handoff files (E37_S01_T01) replace the old single-slot
|
|
42
|
+
# .session_handoff.json. Every file under this directory is processed in
|
|
43
|
+
# section 4 below; see templates/SCRUM_BOARD_SCHEMA.md's handoffs/ section
|
|
44
|
+
# for the write-side filename convention.
|
|
45
|
+
HANDOFF_DIR="$QUEUE_DIR/handoffs"
|
|
35
46
|
EVENTS_LOG="$PROJECT_DIR/project/logs/events.json"
|
|
36
47
|
AGENT="${JENGA_AGENT_TYPE:-unknown}"
|
|
37
48
|
SESSION_ID="${JENGA_SESSION_ID:-}"
|
|
38
49
|
TIMESTAMP=$(date -u +"%Y-%m-%dT%H:%M:%SZ")
|
|
39
50
|
|
|
40
|
-
mkdir -p "$PROJECT_DIR/project/logs" "$QUEUE_DIR"
|
|
51
|
+
mkdir -p "$PROJECT_DIR/project/logs" "$QUEUE_DIR" "$HANDOFF_DIR"
|
|
41
52
|
|
|
42
53
|
# --- 1. Log sender object ---
|
|
43
54
|
|
|
@@ -58,18 +69,55 @@ SENDER=$(jq -n \
|
|
|
58
69
|
}
|
|
59
70
|
}')
|
|
60
71
|
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
72
|
+
# The read-modify-write below is wrapped in scripts/with-lock.sh (E37_S01_T03)
|
|
73
|
+
# so concurrent on_session_end.sh invocations never interleave their read and
|
|
74
|
+
# write of $EVENTS_LOG — see E37_S01_T04 and the two rapports it links
|
|
75
|
+
# (project/rapports/problems/E37_S01-concurrent-session-end-race-verification-gap.md,
|
|
76
|
+
# project/rapports/problems/E39_S03_T05-events-json-concurrent-write-corruption.md)
|
|
77
|
+
# for the concurrent-invocation data loss (events.json truncated to 0 bytes)
|
|
78
|
+
# this replaces. The temp file used for the atomic `mv` is created via
|
|
79
|
+
# `mktemp` in the SAME directory as $EVENTS_LOG (not a shared
|
|
80
|
+
# /tmp/events_tmp.json path) so it is (a) unique per invocation — no two
|
|
81
|
+
# concurrent processes can ever collide on the same temp filename — and
|
|
82
|
+
# (b) on the same filesystem as the destination, so the final `mv` stays an
|
|
83
|
+
# atomic rename rather than a cross-filesystem copy. The update logic is
|
|
84
|
+
# captured as a standalone script string (via a quoted heredoc, so jq's own
|
|
85
|
+
# `$entry` and quoting pass through untouched by this outer shell) and run
|
|
86
|
+
# through `bash -c` with $EVENTS_LOG/$SENDER passed as positional args
|
|
87
|
+
# rather than interpolated into the script text, avoiding fragile
|
|
88
|
+
# nested-quote escaping while keeping this fix self-contained in this file.
|
|
89
|
+
APPEND_EVENT_SCRIPT=$(cat <<'EOS'
|
|
90
|
+
events_log="$1"
|
|
91
|
+
sender_json="$2"
|
|
92
|
+
tmp_file=$(mktemp "$(dirname "$events_log")/events_tmp.XXXXXX") || exit 1
|
|
93
|
+
if [ -s "$events_log" ]; then
|
|
94
|
+
jq --argjson entry "$sender_json" '. += [$entry]' "$events_log" > "$tmp_file" \
|
|
95
|
+
&& mv "$tmp_file" "$events_log"
|
|
64
96
|
else
|
|
65
|
-
|
|
97
|
+
printf '[%s]' "$sender_json" > "$tmp_file" \
|
|
98
|
+
&& mv "$tmp_file" "$events_log"
|
|
66
99
|
fi
|
|
100
|
+
rc=$?
|
|
101
|
+
[ -f "$tmp_file" ] && rm -f "$tmp_file"
|
|
102
|
+
exit "$rc"
|
|
103
|
+
EOS
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
"$PROJECT_DIR/scripts/with-lock.sh" "$EVENTS_LOG" -- bash -c "$APPEND_EVENT_SCRIPT" _ "$EVENTS_LOG" "$SENDER"
|
|
67
107
|
|
|
68
108
|
# --- 2. Manifest-based rapport detection ---
|
|
69
109
|
# Use a JSON array of known filenames instead of a mtime sentinel file.
|
|
70
110
|
# This prevents silently missing rapports written before the session ends
|
|
71
111
|
# but after the sentinel was last touched.
|
|
112
|
+
#
|
|
113
|
+
# The manifest itself is committed to the repo (seeded and kept in sync via
|
|
114
|
+
# scripts/generate-rapport-manifest.sh), so on a normal checkout it already
|
|
115
|
+
# exists and reflects the rapports present at commit time. The bootstrap
|
|
116
|
+
# below is a non-destructive fallback for the genuinely unexpected case
|
|
117
|
+
# where the file is missing (e.g. project/data/ was pruned) — it does not
|
|
118
|
+
# run on a fresh clone with the manifest intact.
|
|
72
119
|
|
|
120
|
+
mkdir -p "$(dirname "$MANIFEST")"
|
|
73
121
|
if [ ! -f "$MANIFEST" ]; then
|
|
74
122
|
echo "[]" > "$MANIFEST"
|
|
75
123
|
fi
|
|
@@ -77,12 +125,28 @@ fi
|
|
|
77
125
|
if [ -d "$RAPPORT_DIR" ]; then
|
|
78
126
|
# Collect current rapport files (exclude .IGNORE.md files — already resolved)
|
|
79
127
|
# Results are stored in a bash array to avoid word-splitting on paths.
|
|
80
|
-
|
|
128
|
+
# Uses a portable `while read` loop rather than mapfile/readarray: macOS
|
|
129
|
+
# ships bash 3.2 (no mapfile support), and this hook must run there —
|
|
130
|
+
# matches the convention already established in scripts/smoke-harness.sh,
|
|
131
|
+
# skills/publish/scripts/generate_release_notes.sh, and
|
|
132
|
+
# skills/publish/scripts/finalize_changelog.sh. Fixed incidentally here
|
|
133
|
+
# because this task's acceptance criteria require the hook to actually
|
|
134
|
+
# execute end-to-end (mapfile silently failed on stock macOS bash,
|
|
135
|
+
# leaving CURRENT_FILES empty and masking real detection results).
|
|
136
|
+
CURRENT_FILES=()
|
|
137
|
+
while IFS= read -r line; do
|
|
138
|
+
[ -n "$line" ] && CURRENT_FILES+=("$line")
|
|
139
|
+
done < <(find "$RAPPORT_DIR" -name "*.md" ! -name "*.IGNORE.md" 2>/dev/null | sort)
|
|
81
140
|
|
|
82
|
-
# Identify new files not present in the manifest
|
|
141
|
+
# Identify new files not present in the manifest. The manifest stores
|
|
142
|
+
# paths relative to $PROJECT_DIR (portable across clones/worktrees), so
|
|
143
|
+
# each absolute CURRENT_FILES entry is relativized before the lookup.
|
|
144
|
+
# NEW_FILES itself stays absolute — it feeds rapport_files in the
|
|
145
|
+
# trigger below, which the scrum master reads directly.
|
|
83
146
|
NEW_FILES=()
|
|
84
147
|
for file in "${CURRENT_FILES[@]}"; do
|
|
85
|
-
|
|
148
|
+
rel="${file#"$PROJECT_DIR"/}"
|
|
149
|
+
known=$(jq --arg f "$rel" 'index($f) != null' "$MANIFEST" 2>/dev/null)
|
|
86
150
|
if [ "$known" != "true" ]; then
|
|
87
151
|
NEW_FILES+=("$file")
|
|
88
152
|
fi
|
|
@@ -109,8 +173,11 @@ if [ -d "$RAPPORT_DIR" ]; then
|
|
|
109
173
|
|
|
110
174
|
echo "$TRIGGER" >> "$QUEUE_FILE"
|
|
111
175
|
|
|
112
|
-
# Update the manifest to include all current files
|
|
113
|
-
|
|
176
|
+
# Update the manifest to include all current files. Delegates to the
|
|
177
|
+
# shared generator script (rather than re-inlining the same jq
|
|
178
|
+
# pipeline) so the runtime update and the committed seed can never
|
|
179
|
+
# drift in how they compute "current rapport files".
|
|
180
|
+
bash "$PROJECT_DIR/scripts/generate-rapport-manifest.sh" "$MANIFEST" >/dev/null
|
|
114
181
|
fi
|
|
115
182
|
fi
|
|
116
183
|
|
|
@@ -129,15 +196,97 @@ TRIGGER=$(jq -n \
|
|
|
129
196
|
|
|
130
197
|
echo "$TRIGGER" >> "$QUEUE_FILE"
|
|
131
198
|
|
|
199
|
+
# --- Helper: has this handoff's referenced task already reached a
|
|
200
|
+
# terminal board status? ---------------------------------------------------
|
|
201
|
+
# Guards against resurrecting a stale handoff file (e.g. one that was
|
|
202
|
+
# accidentally committed to git alongside unrelated work, or one left over
|
|
203
|
+
# from before this directory was ever consumed) as a fresh routing signal
|
|
204
|
+
# for work that has already been fully verified. Terminal here means "no
|
|
205
|
+
# further routing action should occur": Passed / Passed with remarks /
|
|
206
|
+
# Rejected / Done are all end states, and Blocked is included because a
|
|
207
|
+
# blocked task must not be silently re-touched by any agent per
|
|
208
|
+
# agents/developer.md's Blocked-halt contract (only a human may clear it).
|
|
209
|
+
# Deliberately fails open (returns 1 / "not terminal") when the task file
|
|
210
|
+
# can't be found or has no readable status, so a lookup miss falls back to
|
|
211
|
+
# today's behavior (route it) rather than silently dropping a handoff whose
|
|
212
|
+
# task genuinely can't be identified.
|
|
213
|
+
is_task_terminal() {
|
|
214
|
+
local task_id="$1" task_file task_status
|
|
215
|
+
task_file=$(find "$PROJECT_DIR/project/board/tasks" -maxdepth 1 -iname "${task_id}_*.md" 2>/dev/null | head -1)
|
|
216
|
+
[ -z "$task_file" ] && return 1
|
|
217
|
+
task_status=$(awk -F': ' '/^status:/ {print $2; exit}' "$task_file" 2>/dev/null)
|
|
218
|
+
case "$task_status" in
|
|
219
|
+
Passed|"Passed with remarks"|Rejected|Done|Blocked) return 0 ;;
|
|
220
|
+
*) return 1 ;;
|
|
221
|
+
esac
|
|
222
|
+
}
|
|
223
|
+
|
|
132
224
|
# --- 4. Session handoff routing ---
|
|
133
|
-
# Each agent writes
|
|
134
|
-
#
|
|
135
|
-
#
|
|
225
|
+
# Each agent writes a per-session handoff file under project/queue/handoffs/
|
|
226
|
+
# before its session ends (see E37_S01_T01). This section processes every
|
|
227
|
+
# file currently present in that directory — not just one fixed path — so
|
|
228
|
+
# concurrent sessions ending close together each get routed instead of the
|
|
229
|
+
# last writer silently clobbering the others. Every file is routed through
|
|
230
|
+
# the same per-agent logic below, then deleted individually right after
|
|
231
|
+
# routing, so a problem with one file can't block or lose any of the rest.
|
|
232
|
+
#
|
|
233
|
+
# The glob is intentionally generic (*.json, not a pattern anchored to the
|
|
234
|
+
# <agent>-<session_id>-<task_id> convention) so it also picks up older
|
|
235
|
+
# ad-hoc-named files written before that convention existed — routing below
|
|
236
|
+
# only ever reads the .agent/.status fields from file contents, never the
|
|
237
|
+
# filename, so this is safe.
|
|
238
|
+
#
|
|
239
|
+
# Before routing, a developer/tester handoff whose referenced task is
|
|
240
|
+
# already terminal on the board is treated as stale (deleted, not routed)
|
|
241
|
+
# rather than as a live signal — see is_task_terminal() above. This closes
|
|
242
|
+
# a real regression found during E37_S01_T02 testing: this directory can
|
|
243
|
+
# accumulate committed leftover files for already-shipped work (nothing
|
|
244
|
+
# consumed them before this task existed), and a bare generic glob would
|
|
245
|
+
# otherwise resurrect all of them as fresh test_assignment/story_rollup
|
|
246
|
+
# triggers the moment this hook first runs against such a directory.
|
|
247
|
+
for HANDOFF_FILE in "$HANDOFF_DIR"/*.json; do
|
|
248
|
+
# Guard against the glob matching nothing (no nullglob dependency, so
|
|
249
|
+
# this stays portable to stock macOS bash 3.2 — same rationale as the
|
|
250
|
+
# portable `while read` loop in section 2 above).
|
|
251
|
+
[ -e "$HANDOFF_FILE" ] || continue
|
|
252
|
+
|
|
253
|
+
# Atomic per-file claim (E37_S01_T04) — closes the TOCTOU window between
|
|
254
|
+
# this per-process glob snapshot and the eventual delete at the bottom of
|
|
255
|
+
# the loop. `mv` between two paths on the same filesystem is a rename(2)
|
|
256
|
+
# syscall: the kernel guarantees that when multiple concurrent processes
|
|
257
|
+
# race to rename the same source path, exactly one succeeds and every
|
|
258
|
+
# other attempt fails because the source no longer exists. No GNU-only
|
|
259
|
+
# flags are involved, so this is portable to macOS/BSD as well as Linux.
|
|
260
|
+
# Without this claim, concurrent invocations could each see the same
|
|
261
|
+
# not-yet-deleted file, read it, and route it before either deleted it —
|
|
262
|
+
# confirmed in project/rapports/problems/E37_S01-concurrent-session-end-race-verification-gap.md
|
|
263
|
+
# as duplicate (2x/2x/4x observed) trigger entries for the same handoff.
|
|
264
|
+
# If the rename fails, another concurrent process already claimed this
|
|
265
|
+
# exact file — skip it without reading, routing, or deleting anything;
|
|
266
|
+
# that other process now owns it.
|
|
267
|
+
CLAIMED_HANDOFF_FILE="${HANDOFF_FILE}.claimed.$$"
|
|
268
|
+
if ! mv "$HANDOFF_FILE" "$CLAIMED_HANDOFF_FILE" 2>/dev/null; then
|
|
269
|
+
echo "[on_session_end] handoff already claimed by a concurrent invocation — skipping $(basename "$HANDOFF_FILE")"
|
|
270
|
+
continue
|
|
271
|
+
fi
|
|
272
|
+
HANDOFF_FILE="$CLAIMED_HANDOFF_FILE"
|
|
136
273
|
|
|
137
|
-
if [ -f "$HANDOFF_FILE" ]; then
|
|
138
274
|
HANDOFF_AGENT=$(jq -r '.agent // empty' "$HANDOFF_FILE" 2>/dev/null)
|
|
139
275
|
HANDOFF_STATUS=$(jq -r '.status // empty' "$HANDOFF_FILE" 2>/dev/null)
|
|
140
276
|
|
|
277
|
+
# Staleness check — only meaningful for developer/tester handoffs, which
|
|
278
|
+
# each reference exactly one task_id. scrum-master's planning_complete
|
|
279
|
+
# handoff carries a task_ids array of just-created tasks that can't
|
|
280
|
+
# already be terminal in practice, so it is not checked here.
|
|
281
|
+
if [ "$HANDOFF_AGENT" = "developer" ] || [ "$HANDOFF_AGENT" = "tester" ]; then
|
|
282
|
+
CHECK_TASK_ID=$(jq -r '.task_id // empty' "$HANDOFF_FILE" 2>/dev/null)
|
|
283
|
+
if [ -n "$CHECK_TASK_ID" ] && is_task_terminal "$CHECK_TASK_ID"; then
|
|
284
|
+
echo "[on_session_end] stale handoff for already-terminal task $CHECK_TASK_ID — skipping routing, deleting $(basename "$HANDOFF_FILE")"
|
|
285
|
+
rm -f "$HANDOFF_FILE"
|
|
286
|
+
continue
|
|
287
|
+
fi
|
|
288
|
+
fi
|
|
289
|
+
|
|
141
290
|
case "$HANDOFF_AGENT" in
|
|
142
291
|
|
|
143
292
|
scrum-master)
|
|
@@ -228,9 +377,11 @@ if [ -f "$HANDOFF_FILE" ]; then
|
|
|
228
377
|
|
|
229
378
|
esac
|
|
230
379
|
|
|
231
|
-
# Consume
|
|
380
|
+
# Consume this handoff file — it is single-use. Deleted immediately after
|
|
381
|
+
# routing (not batched after the loop) so a later file's failure can never
|
|
382
|
+
# cause an earlier, already-routed file to be left unconsumed or vice versa.
|
|
232
383
|
rm -f "$HANDOFF_FILE"
|
|
233
|
-
|
|
384
|
+
done
|
|
234
385
|
|
|
235
386
|
# --- 5. Todo cleanup ---
|
|
236
387
|
# Remove project/todo.md if it is effectively empty (only blanks, # Todo, and HTML comments).
|
package/mcp/router/embedder.js
CHANGED