@ngockhoale/ukit 3.3.3 → 3.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/CHANGELOG.md +44 -0
  2. package/manifests/engineConformance.yaml +17 -1
  3. package/manifests/hostCapabilities.yaml +68 -1
  4. package/manifests/platform.full.yaml +138 -0
  5. package/manifests/platform.user.yaml +255 -3
  6. package/package.json +1 -1
  7. package/scripts/bench/subagent-orchestrator-corpus.mjs +275 -0
  8. package/scripts/bench/subagent-orchestrator-eval.mjs +565 -0
  9. package/scripts/probe/codex-capability-probe.mjs +169 -0
  10. package/src/cli/commands/doctor.js +168 -0
  11. package/src/cli/commands/indexTools.js +7 -0
  12. package/src/cli/commands/metrics.js +66 -2
  13. package/src/cli/commands/playbook.js +4 -4
  14. package/src/cli/commands/vm.js +49 -8
  15. package/src/core/agentRuntime/adapters.js +328 -27
  16. package/src/core/agentRuntime/artifacts.js +89 -0
  17. package/src/core/agentRuntime/context.js +345 -1
  18. package/src/core/agentRuntime/contract.js +296 -0
  19. package/src/core/agentRuntime/eventStore.js +176 -0
  20. package/src/core/agentRuntime/shadowRun.js +481 -5
  21. package/src/core/agentRuntime/telemetry.js +121 -0
  22. package/src/core/observability/emit/lifecycle.js +68 -1
  23. package/src/core/observability/emit/sessionBoot.js +393 -0
  24. package/src/core/observability/privacy/allowlist.js +10 -1
  25. package/src/core/observability/schema/registry.js +10 -0
  26. package/src/core/runtimeConfig.js +133 -0
  27. package/src/core/userPlaybooks.js +18 -3
  28. package/src/decision/registry.js +19 -0
  29. package/src/diagnostics/feedbackEvents.js +7 -4
  30. package/src/diagnostics/routeOutcomes.js +51 -6
  31. package/src/diagnostics/skillAccuracy.js +43 -3
  32. package/src/index/crossCheckMatrix.js +412 -0
  33. package/src/index/fixLoopEscalation.js +453 -0
  34. package/src/index/playbookRegistry.js +691 -0
  35. package/src/index/reviewPolicy.js +368 -0
  36. package/src/index/routeResolver.js +915 -0
  37. package/src/index/sessionHistoryExtractor.js +359 -0
  38. package/src/index/taskRouting.js +764 -581
  39. package/src/index/tierSelection.js +308 -0
  40. package/src/index/verificationMap.js +404 -0
  41. package/template_project/.claude/hooks/observability-emit.mjs +14 -0
  42. package/template_project/.claude/hooks/record-execution.mjs +19 -1
  43. package/template_project/.claude/hooks/skill-router.sh +691 -25
  44. package/template_project/.claude/hooks/verification-guard.sh +230 -1
  45. package/template_project/.claude/settings.json +2 -2
  46. package/template_project/.claude/ukit/index/cross-check-matrix.mjs +415 -0
  47. package/template_project/.claude/ukit/index/fix-loop-escalation.mjs +456 -0
  48. package/template_project/.claude/ukit/index/playbook-registry.mjs +690 -0
  49. package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +20 -2
  50. package/template_project/.claude/ukit/index/review-policy.mjs +376 -0
  51. package/template_project/.claude/ukit/index/route-resolver.mjs +1059 -0
  52. package/template_project/.claude/ukit/index/route-task.mjs +1253 -846
  53. package/template_project/.claude/ukit/index/session-history-extractor.mjs +362 -0
  54. package/template_project/.claude/ukit/index/tier-selection.mjs +309 -0
  55. package/template_project/.claude/ukit/index/verification-map.mjs +403 -0
  56. package/template_project/.claude/ukit/index/worktree-sweep.mjs +195 -0
  57. package/template_project/.claude/ukit/runtime/execution-ledger.mjs +789 -11
  58. package/template_project/.claude/ukit/runtime/observability-emit.mjs +1102 -0
  59. package/template_project/.claude/ukit/runtime/reinject-context.mjs +9 -1
  60. package/template_project/.claude/ukit/runtime/resumable-run.mjs +149 -5
  61. package/template_project/.claude/ukit/runtime/stop-coordinator.mjs +323 -6
  62. package/template_project/.codex/README.md +8 -0
  63. package/template_project/.omp/hooks/pre/ukit-bridge.js +8 -1
  64. package/template_project/ukit/README.md +1 -1
  65. package/template_project/ukit/storage/config.json +20 -0
  66. package/template_user/playbooks/architecture-decision.md +28 -0
  67. package/template_user/playbooks/autonomous-run.md +43 -0
  68. package/template_user/playbooks/autopilot-full.md +59 -0
  69. package/template_user/playbooks/autopilot-stack.md +54 -0
  70. package/template_user/playbooks/babysit.md +39 -0
  71. package/template_user/playbooks/bug-fix.md +3 -1
  72. package/template_user/playbooks/{issue-implementation.md → feature-implementation.md} +4 -2
  73. package/template_user/playbooks/hillclimb.md +44 -0
  74. package/template_user/playbooks/investigation.md +21 -0
  75. package/template_user/playbooks/migration.md +21 -0
  76. package/template_user/playbooks/open-pr.md +48 -0
  77. package/template_user/playbooks/orchestrate.md +45 -0
  78. package/template_user/playbooks/performance.md +33 -0
  79. package/template_user/playbooks/prototype.md +28 -0
  80. package/template_user/playbooks/refactor.md +19 -0
  81. package/template_user/playbooks/release.md +28 -0
  82. package/template_user/playbooks/runtime-forensics.md +23 -0
  83. package/template_user/playbooks/session-pickup.md +31 -0
  84. package/template_user/playbooks/shipping.md +53 -0
  85. package/template_user/playbooks/skill-evaluation.md +48 -0
  86. package/template_user/playbooks/small-feature.md +20 -0
  87. package/template_user/playbooks/verification-map.json +153 -0
  88. package/template_user/playbooks/verification.md +22 -0
  89. package/template_user/playbooks/worktree-cleanup.md +37 -0
@@ -0,0 +1,28 @@
1
+ ---
2
+ id: architecture-decision
3
+ lanes: [map-impact, shared-edit]
4
+ ---
5
+ You own this fork. Compare the alternatives on the stated constraints and leave a
6
+ decision a later session can check — never settle the fork by just building one.
7
+ If the user says "new task", re-route — do not treat the message as the next step.
8
+ 1. Gather constraints: restate the fork as a checkable question, then the forces
9
+ that govern it (scale, team, ecosystem fit, migration cost, blast radius) and
10
+ the non-negotiables — what no acceptable answer may violate.
11
+ 2. Enumerate the real alternatives, including "keep the status quo" when it is a
12
+ live option; kill each option with a stated reason or keep it in play.
13
+ 3. Compare the survivors on equal terms, constraint by constraint; split each
14
+ alternative into `observed` traits (seen in docs, source, or a probe) and
15
+ `inferred` ones — never upgrade a guess past its evidence.
16
+ 4. Record the decision: the call, the alternatives considered with their kill or
17
+ survive reasons, and a falsifiable rationale — the specific evidence that
18
+ would change the call ("revisit if load crosses X", "revisit if Y ships Z").
19
+ 5. Settle the fork, then stop: the recorded decision, not a spike, is the
20
+ deliverable. A fork whose settlement mechanism is a disposable artifact is a
21
+ `prototype` ask — the spike produces the evidence; this playbook produces the
22
+ decision. The decided path hands off to feature-implementation or migration
23
+ as a new routed task; it is not executed here.
24
+ Reply: the decision, each alternative and why it won or lost, the falsifiable
25
+ rationale in plain terms, and the recommended follow-on lane.
26
+ Ask the human only for: a genuine preference call no evidence settles, a
27
+ constraint only they can supply, or a real dead end. Everything else: compare,
28
+ record, report it.
@@ -0,0 +1,43 @@
1
+ ---
2
+ id: autonomous-run
3
+ lanes: [shared-edit, map-impact]
4
+ ---
5
+ You own this run. Loop work → verify → re-check-finish until the condition
6
+ holds or the escape hatch fires — never fabricate a finish.
7
+ If the user says "new task", re-route — do not treat the message as the next step.
8
+ 1. Bound the goal before touching anything: write the finish condition as a
9
+ checkable predicate you can re-run (e.g. "zero legacy callers", "all N
10
+ items verified"). An unbounded goal is not an autonomous run — name the
11
+ bound or stop, never loop on a vague ask.
12
+ 2. Work the loop: act, verify the step's evidence, re-check the finish
13
+ predicate. A phase is a checkpointed segment of this same loop with its
14
+ own predicate (PB-U21 merged), not a separate planner. Checkpoint each
15
+ segment by appending a milestone Progress entry —
16
+ `<ISO-8601> · milestone: <name> · last-green: <what passed> · files: <...>
17
+ · drift: none|<why>` — and persist a resumable record via
18
+ `resumable-run.mjs suspend <projectRoot> <taskId> --phase <phase> --next
19
+ <action> --completed <a,b,...>`; an interrupted session resumes through
20
+ session-pickup's `resumable-run.mjs read` state-verify, not this playbook.
21
+ 3. On repeated failure of the same step or segment predicate, escalate into a
22
+ deeper debug/verify lane — do not retry the same lane a fourth time, and
23
+ do not exit "blocked" on a fix-loop.
24
+ 4. Escalate (escape hatch): a blocker outside scope, an authorization
25
+ boundary, or the round cap reached — stop the loop, name the blocker
26
+ verbatim, and hand off the ledgered progress: the checkpointed milestone
27
+ entries plus a suspended resumable record. "Done" is only the finish
28
+ condition verified OR the hatch fired with its named blocker — nothing
29
+ else is a terminal report.
30
+ 5. On finish: run the final verification against the step-1 predicate and
31
+ report it. `stop-coordinator.mjs` is the single Stop authority — this
32
+ playbook declares no second stop/finish gate; your loop re-checks the
33
+ predicate, the coordinator decides whether Stop holds.
34
+ Machinery: `resumable-run.mjs suspend|read` for checkpoints, the milestone
35
+ Progress-entry format for the phase ledger, `stop-coordinator.mjs` as the ONE
36
+ Stop evaluator. On Codex (no hook surface) run the same commands manually —
37
+ enforcement stays advisory, never a live claim.
38
+ Sub-playbooks: `orchestrate` when the ask is a standing multi-workstream
39
+ program needing a coordinator; a single bounded goal stays here.
40
+ Reply: the finish predicate, per-segment verification evidence, and the
41
+ terminal state — verified finish or Escalate with the named blocker.
42
+ Ask the human only for: irreversible writes, a genuine preference call no
43
+ experiment settles, or a real dead end. Everything else: do it, report it.
@@ -0,0 +1,59 @@
1
+ ---
2
+ id: autopilot-full
3
+ lanes: [review-release]
4
+ ---
5
+ You own a set of independent changes that should each become its own
6
+ verified PR. You split the work into independent units, assign one owner per
7
+ PR, run the units as sequential worktree waves, verify each merge-ready, and
8
+ land each independently behind per-action authorization.
9
+ If the user says "new task", re-route — do not treat the message as the next step.
10
+ 1. Confirm this is autopilot-full territory: two or more independent changes
11
+ that should each ship as its own verified PR, and you can reach
12
+ CI/review state plus git history. Dependent or stacked changes re-route
13
+ to `autopilot-stack`; a single change goes down the feature path — say
14
+ so and stop. If `gh` is missing or unauthenticated, report the CLI
15
+ failure verbatim — never fabricate check, review, or head state.
16
+ 2. Split the work into independent units: each unit stands alone on the
17
+ base branch — its own branch, diff, checks, and PR. Name the seam
18
+ between units and the files each touches; an overlap that makes order
19
+ matter is a dependency — re-route that pair to `autopilot-stack`.
20
+ 3. Assign one owner per PR and run sequential worktree waves (RCC-01
21
+ classes 1–3: branch/commit, feature-branch push, `gh pr create`). The
22
+ upstream provider connector plus parallel background agents fall back
23
+ here to sequential worktree waves — the pattern the handoff pipeline
24
+ already proves with its executor worktrees — plus `gh` per PR.
25
+ Parallelism is a cost optimization, not a correctness requirement; the
26
+ sequential wave preserves the semantics.
27
+ 4. Verify each unit merge-ready independently (RCC-01 classes 4–5): `gh pr
28
+ checks`, `gh pr view`, review threads via
29
+ `gh api repos/{owner}/{repo}/pulls/{n}/comments`, plus the unit's own
30
+ test/build run where the repo defines one. Record a per-PR verification
31
+ receipt — commands run, output, result. A green unit never vouches for
32
+ its neighbor; a unit that fails verification goes back to its owner, it
33
+ is never landed unverified.
34
+ 5. Land each independently — every merge/land is an irreversible write
35
+ behind the RCC-01 §Land-authorization boundary. Merge/land to the
36
+ default branch and push-to-default are gated: stop and `Ask the human`
37
+ before each one; authorization is per-action and never carried forward —
38
+ "land the set" authorizes the described sequence once, and a new merge
39
+ is a new authorization. The fallback land step is `gh pr merge` or
40
+ `git merge` into the default branch plus `git push` — no connector-only
41
+ step. A model never authorizes; the gate stays deterministic and human.
42
+ Record a per-PR land receipt: the authorization granted, the land
43
+ command, the resulting head.
44
+ 6. Finish when all units are landed as independent verified heads: `git
45
+ log`/`git rev-parse` the default branch and confirm every intended unit
46
+ is in. If authorization is missing or a unit fails verification, report
47
+ BLOCKED at that step naming the boundary or the failed unit — the
48
+ remaining units stay unlanded; never soft-fail around it.
49
+ Every step here is `gh`/git over Bash (RCC-01 §Capability matrix) — full on
50
+ Claude Code and omp; on Codex the gate posture is advisory /
51
+ instruction-mediated (RCC-01 §Hook-free enforcement): the model is asked, not
52
+ stopped.
53
+ Reply: the unit map, each per-PR verification receipt, the land
54
+ authorizations asked and granted, the per-PR land receipts with resulting
55
+ heads, and the terminal state — all units landed, or BLOCKED with the named
56
+ cause.
57
+ Ask the human at every irreversible write — per-action, even mid-set — and
58
+ for a genuine preference call no experiment settles or a real dead end.
59
+ Everything else: do it, report it.
@@ -0,0 +1,54 @@
1
+ ---
2
+ id: autopilot-stack
3
+ lanes: [review-release]
4
+ ---
5
+ You build and verify a linear stacked-PR series on one base branch — a set
6
+ of ordered changes where each layer depends on the layer beneath it. You
7
+ order the stack, chain the branches, verify each layer on its parent, and
8
+ hand the verified stack to the operator for review/land. You never land the
9
+ stack yourself.
10
+ If the user says "new task", re-route — do not treat the message as the next step.
11
+ 1. Confirm this is autopilot-stack territory: a stacked-PR series on one
12
+ base branch where the layers genuinely depend on each other and a linear
13
+ order exists. Independent changes with no ordering requirement are
14
+ `autopilot-full`, and a single change is the feature path — re-route
15
+ both and stop. If you cannot reach git or `gh`, report the CLI failure
16
+ verbatim and stop — never fabricate stack, check, or PR state.
17
+ 2. Order the stack: lay out the layers base-to-tip from the dependency
18
+ order the user gave or the code requires, and record it — a stack
19
+ ordering record naming each layer, its branch name, and its parent. A
20
+ stack with no linear order is not a stack; if the ordering is
21
+ ambiguous, resolve it before the first branch exists.
22
+ 3. Chain the branches: each layer branches off its parent's head — `git
23
+ checkout -b <layer-branch> <parent-branch>` — never off the default
24
+ branch mid-stack. Open each PR with `gh pr create --base <parent>` so
25
+ the PR base is the parent layer's branch, not the default branch. This
26
+ is the RCC-01 §Provider-API fallback: connector stack metadata is
27
+ convenience, explicit git branch chaining is the capability — no step
28
+ is connector-only.
29
+ 4. Build and verify each layer on its parent: the layer's own test/build
30
+ run plus the diff against its parent branch. Record a per-layer
31
+ verification receipt — commands run, output, result — in stack order.
32
+ A layer that fails stops the chain there: later layers are never built
33
+ on an unverified parent, the failing layer is named verbatim, and the
34
+ run reports BLOCKED — never continue past it and never fabricate a
35
+ green.
36
+ 5. Hand the verified stack to the operator for review/land — the RCC-01
37
+ §Land-authorization boundary. The agent never lands the stack itself:
38
+ merge/land to the default branch is an irreversible write, gated per
39
+ action — if the operator delegates a land step to you, `Ask the human`
40
+ before each merge and never carry authorization forward; a model never
41
+ authorizes. Finish at the boundary: stack fully built and verified,
42
+ ordering record and receipts attached, operator reviewing — or BLOCKED
43
+ at the named layer.
44
+ Every step is `gh`/git over Bash (RCC-01 §Capability matrix) — full on
45
+ Claude Code and omp; on Codex the gate posture is advisory /
46
+ instruction-mediated (RCC-01 §Hook-free enforcement): the model is asked,
47
+ not stopped.
48
+ Reply: the stack ordering record, each per-layer verification receipt, the
49
+ PR list with its `--base` chain, and the terminal state — stack built and
50
+ verified, handed to the operator for review/land, or BLOCKED at the named
51
+ layer.
52
+ Ask the human only for: a land authorization the operator delegates to you
53
+ (per merge, never carried forward), a genuine ordering call the code cannot
54
+ settle, or a real dead end. Everything else: do it, report it.
@@ -0,0 +1,39 @@
1
+ ---
2
+ id: babysit
3
+ lanes: [review-release]
4
+ ---
5
+ You own this PR's journey to merge-ready. Loop read-state → fix → push →
6
+ re-check until CI is green, every review thread is resolved, and no conflicts
7
+ remain — then hand back. The merge itself is out of scope.
8
+ If the user says "new task", re-route — do not treat the message as the next step.
9
+ 1. Confirm this is babysit territory: a PR is already open (upstream's
10
+ PR-create step class, RCC-01 class 3, is done) and you can reach CI/review
11
+ state. No open PR, or no CI-or-review access → not babysit — say so and
12
+ stop. If `gh` is missing or unauthenticated, report the CLI failure
13
+ verbatim — never fabricate check or review state.
14
+ 2. Read state (RCC-01 class 4): `gh pr checks` for CI, `gh pr view` for PR
15
+ status, and review threads via
16
+ `gh api repos/{owner}/{repo}/pulls/{n}/comments`. Record a snapshot — CI
17
+ statuses, unresolved threads, conflict state.
18
+ 3. Fix what the snapshot names: resolve failing checks and review comments,
19
+ rebase or merge-base-resolve conflicts, then `git push` — class 5's
20
+ conflict/comment resolution loop composing commit + push + status reads.
21
+ Feature-branch pushes are reversible; run them autonomously.
22
+ 4. Re-check: re-read the same three surfaces and take a new snapshot. Repeat
23
+ the loop until the PR reports merge-ready — CI green, threads resolved, no
24
+ conflicts. Polling cadence is manual on the `gh`/git-over-Bash fallback
25
+ (RCC-01 §Provider-API fallback row): you re-run the reads; nothing polls
26
+ for you.
27
+ 5. Finish at merge-ready and hand back — landing belongs to `shipping` behind
28
+ the RCC-01 §Land-authorization boundary (an irreversible write), so this
29
+ playbook never performs the merge. If authorization or capability is
30
+ missing at any step, report BLOCKED at that step with the named cause —
31
+ do not soften it.
32
+ Every step here is `gh`/git over Bash (RCC-01 §Capability matrix) — full on
33
+ Claude Code and omp; on Codex the gate posture is advisory /
34
+ instruction-mediated (RCC-01 §Hook-free enforcement): the model is asked, not
35
+ stopped.
36
+ Reply: the PR number, the per-round CI + review-thread snapshots, and the
37
+ terminal state — merge-ready handed back, or BLOCKED with the named cause.
38
+ Ask the human only for: irreversible writes, a genuine preference call no
39
+ experiment settles, or a real dead end. Everything else: do it, report it.
@@ -12,7 +12,9 @@ If the user says "new task", re-route — do not treat the message as the next s
12
12
  "might help" is a hypothesis, not a fix; it does not ship.
13
13
  4. Verify on the same surface: the original repro now passes. "Inconclusive" or
14
14
  wrong-surface is not a pass. A unit test shows branch behavior, not bug absence.
15
- 5. Keep the rejected hypotheses — one line each, why ruled out.
15
+ 5. Keep the rejected hypotheses — one line each, why ruled out. When rejections on
16
+ this same defect cross the fix-loop threshold, stop retrying this lane —
17
+ escalate to runtime-forensics (live symptom) or a deeper debug/verify lane.
16
18
  Reply: what was broken, root cause, fix, how verified — paste failing-then-passing
17
19
  repro output verbatim.
18
20
  Ask the human only for: irreversible writes, a genuine preference call no experiment settles, or a real dead end. Everything else: do it, report it.
@@ -1,14 +1,16 @@
1
1
  ---
2
- id: issue-implementation
2
+ id: feature-implementation
3
3
  lanes: [local-build, shared-edit, map-impact]
4
4
  ---
5
5
  You own this task. Normalize the goal, build, verify.
6
6
  If the user says "new task", re-route — do not treat the message as the next step.
7
7
  1. State the done condition as a checkable predicate before writing code.
8
8
  2. Find the established analog — follow it unless you name why it does not fit.
9
+ (Skippable only when you can name why no analog exists.)
9
10
  3. Name the data shape and its organizing structure before writing logic.
10
11
  4. Implement the smallest change satisfying the predicate.
11
12
  5. Verify against the predicate on the real artifact — not "it compiles".
12
13
  6. Widen once: check the impact surface the route named, no broader.
13
14
  Reply: what changed, the predicate, the evidence it now holds.
14
- Ask the human only for: irreversible writes, a genuine preference call no experiment settles, or a real dead end. Everything else: do it, report it.
15
+ Ask the human only for: irreversible writes, a genuine preference call no experiment
16
+ settles, or a real dead end. Everything else: do it, report it.
@@ -0,0 +1,44 @@
1
+ ---
2
+ id: hillclimb
3
+ lanes: [local-build, find-cause]
4
+ ---
5
+ You own a sustained optimization program — one metric, one target, a
6
+ hypothesis loop, and a ledger. A hillclimb ends at the target or at the
7
+ stated stop criterion; it never drifts open-ended.
8
+ If the ask is a single measured perf issue, re-route to performance —
9
+ Performance Fix != Hillclimb; this playbook is a program, not one fix.
10
+ If the metric is not yet measurable (no baseline, no harness), re-route to
11
+ investigation and stand up the measurement first — never climb blind.
12
+ If the user says "new task", re-route — do not treat the message as the next
13
+ step.
14
+ 1. Pin the program: name the metric, the target, and the measurement harness
15
+ (the benchmark or timed run that produces the number), and state the stop
16
+ criterion up front — an iteration budget, a time box, or a plateau rule.
17
+ A program without a written stop criterion is not a hillclimb — bound it
18
+ or stop.
19
+ 2. Propose one hypothesis per iteration — a single change with the mechanism
20
+ it should move ("inlining X cuts allocation churn"). Never batch
21
+ hypotheses: a mixed step cannot be attributed, so it cannot be learned
22
+ from.
23
+ 3. Measure: run the same harness and record the number against the
24
+ iteration's baseline. Keep the harness untouched across iterations so
25
+ every delta stays comparable.
26
+ 4. Accept or reject the change on the measured delta alone and record the
27
+ verdict — accept keeps the change as the new baseline; reject reverts it.
28
+ A rejected hypothesis still gets a verdict line: negative results are
29
+ ledger evidence, not silence.
30
+ 5. Repeat from step 2 until the target is reached or the pinned budget/stop
31
+ criterion is exhausted — then stop and report the ledger either way.
32
+ Ledger: every iteration appends one entry to the exec-ledger surface under
33
+ `.ukit/storage/cache/exec-ledger/` in the milestone Progress-entry grammar —
34
+ `<ISO-8601> · milestone: hillclimb-<n> · last-green: <what passed> · files:
35
+ <...> · drift: none|<why>` — carrying hypothesis → measurement → verdict, so
36
+ an interrupted program resumes from the ledger, not from memory.
37
+ Machinery: plain Bash for harness runs and ledger appends on all three
38
+ engines; on Codex the ledger stays advisory — no hook surface enforces it.
39
+ Reply: the metric and target, the per-iteration ledger (hypothesis →
40
+ measurement → verdict), and the terminal state — target reached or the stop
41
+ criterion exhausted.
42
+ Ask the human only for: irreversible changes, a metric/target preference call
43
+ no measurement settles, or a real dead end. Everything else: run the loop,
44
+ report the ledger.
@@ -0,0 +1,21 @@
1
+ ---
2
+ id: investigation
3
+ lanes: [find-cause, map-impact]
4
+ ---
5
+ You own this question. Answer it with citations or falsify it — you may not mutate
6
+ the repo to get there.
7
+ If the user says "new task", re-route — do not treat the message as the next step.
8
+ 1. Frame the question as a checkable claim — what would confirm it, what would
9
+ falsify it.
10
+ 2. Gather evidence read-only: index-first lookups, source reads, non-mutating
11
+ commands. No edits, no scaffolds.
12
+ 3. Classify every finding: `observed` (seen in source or output), `inferred` (it
13
+ follows), `unverified` (cannot be checked here) — never upgrade a claim past
14
+ its evidence.
15
+ 4. Answer the question, or mark it unresolved with the exact missing evidence
16
+ named. (Skippable: the unresolved branch only when every claim is observed.)
17
+ 5. Confirm zero mutation — `git status` clean, no stray files.
18
+ Reply: the answer with `path:symbol` or command-output citations per claim, the
19
+ `unverified` items and why, and confirmation the repo is untouched.
20
+ Ask the human only for: irreversible writes, a genuine preference call no experiment
21
+ settles, or a real dead end. Everything else: do it, report it.
@@ -0,0 +1,21 @@
1
+ ---
2
+ id: migration
3
+ lanes: [shared-edit, map-impact]
4
+ ---
5
+ You own this move. Land it on the new shape, verify it there, and prove the way
6
+ back.
7
+ If the user says "new task", re-route — do not treat the message as the next step.
8
+ 1. Pin the current state and behavior — counts, checksums, or the characterization
9
+ suite you will re-check after the move.
10
+ 2. Define the target shape and the rollback path before touching anything — a
11
+ migration with no demonstrated way back is not ready.
12
+ 3. Migrate in verifiable steps; re-check the pin on the target after each step,
13
+ not from memory.
14
+ 4. Verify on the target: read and exercise the moved data/system through the new
15
+ shape; the pin holds.
16
+ 5. Demonstrate rollback on a safe surface — run it or dry-run it and capture the
17
+ output; then schedule the legacy path for removal — it does not linger.
18
+ Reply: what moved, the verification evidence on the target, the rollback
19
+ demonstration, and the legacy-path removal plan.
20
+ Ask the human only for: irreversible writes, a genuine preference call no experiment
21
+ settles, or a real dead end. Everything else: do it, report it.
@@ -0,0 +1,48 @@
1
+ ---
2
+ id: open-pr
3
+ lanes: [review-release]
4
+ ---
5
+ You own packaging this finished work into a review-ready PR — ordered commits,
6
+ a conventional title, a briefing-style body, and a PR a reviewer can open
7
+ cold. Landing the PR is not part of this playbook.
8
+ If the user says "new task", re-route — do not treat the message as the next step.
9
+ 1. Confirm this is open-pr territory: a finished unit of work exists locally —
10
+ commits already made or a working tree ready to commit — and the repo has a
11
+ PR surface (a `gh`-reachable forge remote). No finished work to package, or
12
+ a repo without a PR surface → not open-pr — say so and stop. If `gh` is
13
+ missing or unauthenticated, report the CLI failure verbatim — never
14
+ fabricate PR or check state.
15
+ 2. Order or split the commits (RCC-01 class 1): read `git status` and `git
16
+ log`/`diff` against the base branch, then shape the history a reviewer
17
+ expects — one logical change per commit, ordered so each commit builds on
18
+ the last, splits that separate mechanical moves from semantic edits. On a
19
+ feature branch this is reversible and runs autonomously; never rewrite or
20
+ reorder commits already pushed to the default branch — that belongs behind
21
+ `Ask the human`.
22
+ 3. Write the title: conventional-commit shape (`type(scope): summary` —
23
+ `feat:`, `fix:`, `chore:` …) in imperative mood, sized to one line. The
24
+ title names what the PR does, not the task that asked for it.
25
+ 4. Write the body: briefing style — what changed and why up front, the
26
+ evidence (commands run, output that proves it), then anything a reviewer
27
+ must know before reading the diff. A reviewer who never saw this session
28
+ should be able to approve from the body alone.
29
+ 5. Open the PR with `gh pr create` (RCC-01 class 3): first push the feature
30
+ branch — a reversible write, autonomous — then create with the title and
31
+ body from steps 3–4. The fallback is forge REST over `gh api`/`curl` via
32
+ Bash (RCC-01 §Provider-API fallback) — the trivial fallback that makes this
33
+ the release group's lead entry; no step is connector-only.
34
+ 6. Confirm ready state: `gh pr view` the new PR — it exists, is non-draft
35
+ unless a draft was asked for, points at the right base, and carries the
36
+ title/body you wrote. The merge itself, any push or merge to the default
37
+ branch, is an irreversible write behind the RCC-01 §Land-authorization
38
+ boundary — this playbook never performs it; if the ask was "open AND land",
39
+ stop at the open PR and `Ask the human` for the land, or hand to `shipping`.
40
+ Every step here is `gh`/git over Bash (RCC-01 §Capability matrix) — full on
41
+ Claude Code and omp; on Codex the gate posture is advisory /
42
+ instruction-mediated (RCC-01 §Hook-free enforcement): the model is asked, not
43
+ stopped.
44
+ Reply: the PR URL plus the title and body verbatim, and the terminal state —
45
+ ready for review, or BLOCKED with the named cause.
46
+ Ask the human only for: irreversible writes (default-branch push/merge, the
47
+ land), a genuine preference call no experiment settles, or a real dead end.
48
+ Everything else: do it, report it.
@@ -0,0 +1,45 @@
1
+ ---
2
+ id: orchestrate
3
+ lanes: [shared-edit, map-impact]
4
+ ---
5
+ You are the coordinator of a standing multi-workstream program. Loop
6
+ decompose → assign → coordinate merges/conflicts → track state → report
7
+ until program-level acceptance holds or a program-level blocker fires.
8
+ If the user says "new task", re-route — do not treat the message as the next step.
9
+ 1. Confirm this is orchestrate territory: a standing program with more than
10
+ one workstream running at once. A single bounded goal is `autonomous-run`,
11
+ and a small parallel task set that fits one worktree wave needs no
12
+ coordinator — route both there and stop.
13
+ 2. Decompose the program into workstreams: each carries a bounded goal, a
14
+ checkable finish predicate (the autonomous-run contract), and declared
15
+ file/lane boundaries — the seams where streams meet are what you own.
16
+ 3. Assign owners: one executor worktree per workstream and a per-task file
17
+ per workstream under `docs/AI_HANDOFF/tasks/` as its owner/agent contract
18
+ — scope, status, predicate, Progress entries. Register every stream in
19
+ `docs/AI_HANDOFF/INDEX.md` and keep the `docs/AI_HANDOFF/RUN.md` cursor
20
+ pointing at the program's next step; that pair is the resumable surface.
21
+ 4. Coordinate merges and conflicts: when a stream's predicate verifies, fold
22
+ its worktree into the program order — name conflicts, classify which
23
+ stream owns the resolution, and record the call in the coordinator
24
+ decision log. You adjudicate seams; inside a worktree the owner executes.
25
+ 5. Track program state and report: milestone Progress entries —
26
+ `<ISO-8601> · milestone: <name> · last-green: <what passed> · files: <...>
27
+ · drift: none|<why>` — are the ledger; never trust `last-green` without
28
+ its verification evidence. Append a coordinator entry per decision so an
29
+ interrupted session resumes from INDEX/RUN, not from memory.
30
+ 6. Finish at program-level acceptance: every workstream predicate verified
31
+ AND cross-stream integration confirmed, then coordinator close-out —
32
+ reconcile INDEX statuses, drain the cursor, and summarize the decision
33
+ log. A blocker outside any stream's scope stops the program; name it
34
+ verbatim and hand back the ledger — never fabricate acceptance.
35
+ This is the file-based pattern, not live multi-agent plumbing: no durable
36
+ background agents, no cross-agent messaging — file handoff through
37
+ INDEX/RUN, task files, and Progress entries is the whole contract. The same
38
+ contract holds on all three engines: full on Claude Code and omp; on Codex
39
+ (no hook surface) it is advisory / instruction-mediated — the model is
40
+ asked, not stopped.
41
+ Reply: the workstream map, the coordinator decision log, and the terminal
42
+ state — program-level acceptance with the close-out, or BLOCKED with the
43
+ named blocker.
44
+ Ask the human only for: irreversible writes, a genuine preference call no
45
+ experiment settles, or a real dead end. Everything else: do it, report it.
@@ -0,0 +1,33 @@
1
+ ---
2
+ id: performance
3
+ lanes: [local-build, find-cause]
4
+ ---
5
+ You own a measured slowness claim. Turn it into a before/after delta on one
6
+ harness — the measured delta is the finish; guessing is not.
7
+ If the ask carries no measurable signal yet (no number, baseline, or harness),
8
+ re-route to investigation and measure first — never optimize blind.
9
+ If the symptom is live or intermittent (hangs, freezes, "sometimes"), re-route
10
+ to runtime-forensics — that playbook names the mechanism first.
11
+ If the user says "new task", re-route — do not treat the message as the next
12
+ step.
13
+ 1. Baseline: identify the harness (benchmark, timed request, profiling run) or
14
+ build the smallest one that reproduces the symptom; record the measurement
15
+ and state the target margin explicitly.
16
+ 2. Trace: instrument the same harness — profile, flame graph, query plan,
17
+ timing breakdown — and name the hot path that dominates the symptom.
18
+ 3. Improve: make the smallest change aimed at that hot path; keep the harness
19
+ untouched so before/after stay comparable.
20
+ 4. Re-measure: run the same harness again. Finished only when the measured
21
+ delta meets the stated margin; otherwise iterate from the new profile or
22
+ hand back with the numbers you have.
23
+ 5. Leave the harness runnable: store captures under artifacts, keep the
24
+ commands copy-pasteable for the next run.
25
+ One run proves one delta — repeated hillclimb passes beyond the stated margin
26
+ are a follow-on run, not part of this one.
27
+ Sub-playbooks: `hillclimb` when the ask is a sustained metric program with a
28
+ target and an iteration budget; a single measured issue stays here.
29
+ Reply: the before/after numbers, the harness + commands that produced them,
30
+ the hot path found, and the change made.
31
+ Ask the human only for: irreversible changes, a cost/complexity preference
32
+ call no measurement settles, or a real dead end. Everything else: do it,
33
+ report it.
@@ -0,0 +1,28 @@
1
+ ---
2
+ id: prototype
3
+ lanes: [local-build]
4
+ ---
5
+ You own this spike. Build the cheapest thing that settles the fork, then dispose
6
+ or promote it — never leave it ambiguous.
7
+ If the user says "new task", re-route — do not treat the message as the next step.
8
+ 1. State the design question this spike answers and the observation that settles
9
+ it — the measured number, render, or behavior that picks a side.
10
+ 2. Build the cheapest artifact that produces that evidence — hardcode freely;
11
+ hardening, tests, and wiring are deliberately skipped unless they are the
12
+ discriminating evidence.
13
+ 3. Measure or observe against the criterion from step 1; capture the output
14
+ verbatim — observed evidence, never an estimate.
15
+ 4. Record the verdict — which side the evidence picked and why the runner-up
16
+ lost; a fork decidable without building stops early with the verdict.
17
+ 5. Dispose or promote: mark the artifact throwaway (remove or quarantine it) or
18
+ flag it for promotion — a prototype silently left in place is a defect. The
19
+ spike answers a question; it does not ship a feature — no integration,
20
+ verification-map ceremony, or production-diff obligation applies here. When
21
+ the artifact must ship, hand off to feature-implementation with the findings.
22
+ A fork needing a recorded decision instead of a build is
23
+ architecture-decision territory — re-route rather than spike.
24
+ Reply: the design question, the experiment output verbatim, the verdict, and the
25
+ artifact disposition (disposable | promoted).
26
+ Ask the human only for: irreversible writes, a genuine preference call no
27
+ experiment settles, or a real dead end. Everything else: run it, record it,
28
+ dispose of it.
@@ -0,0 +1,19 @@
1
+ ---
2
+ id: refactor
3
+ lanes: [shared-edit, map-impact]
4
+ ---
5
+ You own this restructure. Behavior must not change — pin it, move it, prove it.
6
+ If the user says "new task", re-route — do not treat the message as the next step.
7
+ 1. Pin the current behavior: the existing suite, or characterization checks you
8
+ write first. A pin you cannot run is not a pin. (Skippable: writing
9
+ characterization checks only when a runnable pin already exists — cite it.)
10
+ 2. Apply the structural change in verifiable steps — re-run the pin after each,
11
+ not once at the end.
12
+ 3. Migrate every caller before deleting the old surface — a compatibility shim
13
+ still counts as a caller; sweep until zero legacy references remain.
14
+ 4. Re-run the pin with the new structure in place: green, zero behavior delta.
15
+ 5. Delete the old surface and its dead code — obsolete aliases and re-exports go
16
+ in the same change, not a follow-up.
17
+ Reply: what moved, the pin evidence before-and-after, the caller sweep result.
18
+ Ask the human only for: irreversible writes, a genuine preference call no experiment
19
+ settles, or a real dead end. Everything else: do it, report it.
@@ -0,0 +1,28 @@
1
+ ---
2
+ id: release
3
+ lanes: [review-release]
4
+ ---
5
+ You own this ship chain — this repo's own release only. Done means all three
6
+ version probes equal: package.json, git tag/GitHub Release, npm registry.
7
+ If the user says "new task", re-route — do not treat the message as the next step.
8
+ 1. Update the version and changelog, commit, and run `yarn test:release-core`.
9
+ 2. `git push origin <branch>` and `git push --tags` with a tag matching the
10
+ version.
11
+ 3. Create the GitHub Release from that tag.
12
+ 4. `npm publish --dry-run` first, then `npm publish` in the same cycle.
13
+ 5. Probe all three sources — `npm view @ngockhoale/ukit version`, `gh release
14
+ list`, `git tag` — all must equal package.json; run
15
+ `node scripts/release/verify-release.mjs --post-publish`.
16
+ 6. If parity fails or publish never ran, say "not shipped" and name the exact
17
+ missing step — git-only or GitHub-only is not a release; stopping early is a
18
+ named miss, never a quiet success.
19
+ Sub-playbooks: `open-pr` when finished local work needs packaging into a
20
+ review-ready PR; `babysit` when an already-open PR must be driven to
21
+ merge-ready; `shipping` when a stack of ready changes must be verified and
22
+ landed; `autopilot-full` when independent changes should each ship as their
23
+ own verified PR; `autopilot-stack` when ordered layers build a stacked-PR
24
+ series.
25
+ Reply: the version, the three probe outputs verbatim, and shipped | not shipped with
26
+ the exact missing step.
27
+ Ask the human only for: irreversible writes, a genuine preference call no experiment
28
+ settles, or a real dead end. Everything else: do it, report it.
@@ -0,0 +1,23 @@
1
+ ---
2
+ id: runtime-forensics
3
+ lanes: [find-cause]
4
+ ---
5
+ You own this anomaly. Turn it into a named mechanism with captured evidence —
6
+ naming the mechanism is the finish; fixing it is the follow-on run.
7
+ If the user says "new task", re-route — do not treat the message as the next step.
8
+ 1. Pick the input branch: live process → attach and observe it; captured artifact
9
+ (.cpuprofile, trace, spindump, heap snapshot) → load it with the matching
10
+ analyzer.
11
+ 2. Capture the evidence: live = samples, heap, open handles over an observation
12
+ window; artifact = hot frames, dominators, blocked threads cited into the
13
+ capture.
14
+ 3. Isolate the anomaly — the state or hot path that explains the reported
15
+ symptom; rule out the candidates that do not.
16
+ 4. Name the mechanism and hand off — the fix goes to bug-fix or performance as a
17
+ follow-on run, carrying this evidence.
18
+ 5. Release the patient: detach or kill instrumentation, store captures under
19
+ artifacts, leave nothing running.
20
+ Reply: the anomaly, the named mechanism, the captures/derived readings that prove
21
+ it, and the recommended follow-on lane.
22
+ Ask the human only for: irreversible writes, a genuine preference call no
23
+ experiment settles, or a real dead end. Everything else: do it, report it.
@@ -0,0 +1,31 @@
1
+ ---
2
+ id: session-pickup
3
+ lanes: [local-fix, local-build]
4
+ ---
5
+ You own this continuation. Resume at the confirmed cursor — never fabricate
6
+ continuity.
7
+ If the user says "new task", re-route — do not treat the message as the next step.
8
+ Resume (a resumable record exists):
9
+ 1. Load the resumable record — goal, done-so-far, next step — via
10
+ `resumable-run.mjs read <projectRoot> <taskId>`; a `status: absent` or no
11
+ record means there is nothing to resume: re-route as the underlying task
12
+ type, do not resume.
13
+ 2. Verify the record against the actual worktree before trusting it — the
14
+ `stateVerify` receipt already flags divergent index/config fingerprints;
15
+ where it reports `fail` or the tree disagrees, the record is stale:
16
+ reconcile or report the divergence, never silently pick a side.
17
+ 3. Continue from the confirmed cursor — no re-done steps, no skipped ones.
18
+ Suspend (interruption, deliberate pause, session end):
19
+ 4. Checkpoint progress — what is done, what is next, open questions.
20
+ 5. Persist the resumable record via `resumable-run.mjs suspend <projectRoot>
21
+ <taskId> --phase <phase> --next <action> --completed <a,b,...>` and
22
+ self-check sufficiency: a cold session could resume from the record alone.
23
+ Machinery: `resumable-run.mjs read|suspend`; stage `off` writes nothing and the
24
+ pickup record is advisory context. On Codex (no hook surface) run the same
25
+ commands manually — enforcement stays advisory, never a live claim.
26
+ Sub-playbooks: `worktree-cleanup` when the session tail is reclaiming the
27
+ stale worktrees past sessions spawned — the gated sweep, not a resume.
28
+ Reply: for resume — the cursor you verified and the evidence the worktree agrees;
29
+ for suspend — the record written and the sufficiency check.
30
+ Ask the human only for: irreversible writes, a genuine preference call no experiment
31
+ settles, or a real dead end. Everything else: do it, report it.