@orkestrel/scaffold 0.0.77 → 0.0.78

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (157) hide show
  1. package/dist/agents/skills/orkestrel-dispatch/scripts/bench.js +204 -0
  2. package/dist/agents/skills/orkestrel-dispatch/scripts/brief.js +102 -0
  3. package/dist/agents/skills/orkestrel-dispatch/scripts/cite.js +95 -0
  4. package/dist/agents/skills/orkestrel-dispatch/scripts/helpers.js +207 -0
  5. package/dist/agents/skills/orkestrel-dispatch/scripts/launch.js +108 -0
  6. package/dist/agents/skills/orkestrel-dispatch/scripts/login.js +114 -0
  7. package/dist/agents/skills/orkestrel-dispatch/scripts/result.js +108 -0
  8. package/dist/agents/skills/orkestrel-dispatch/scripts/sweep.js +156 -0
  9. package/dist/agents/skills/orkestrel-harden/scripts/discovery.js +196 -0
  10. package/dist/agents/skills/orkestrel-publish/scripts/compare.js +206 -0
  11. package/dist/agents/skills/orkestrel-publish/scripts/pins.js +93 -0
  12. package/dist/agents/skills/orkestrel-publish/scripts/wave.js +458 -0
  13. package/dist/agents/skills/orkestrel-publish/scripts/window.js +188 -0
  14. package/dist/agents/skills/orkestrel-scout/scripts/map.js +300 -0
  15. package/dist/agents/templates/brief.md +55 -0
  16. package/dist/bin/main.js +4 -2
  17. package/dist/bin/main.js.map +1 -1
  18. package/dist/host/AGENTS.md +77 -135
  19. package/dist/host/agents/orchestration.md +147 -998
  20. package/dist/host/agents/skills/enterprise-bootstrap/SKILL.md +2 -2
  21. package/dist/host/agents/skills/enterprise-bootstrap/references/inspection.md +1 -1
  22. package/dist/host/agents/skills/{orkestrel-align-packages → orkestrel-align}/SKILL.md +6 -13
  23. package/dist/host/agents/skills/{orkestrel-align-packages → orkestrel-align}/agents/openai.yaml +1 -1
  24. package/dist/host/agents/skills/{orkestrel-align-packages → orkestrel-align}/references/fleet.md +5 -7
  25. package/dist/host/agents/skills/{orkestrel-build-application → orkestrel-build}/SKILL.md +11 -22
  26. package/dist/host/agents/skills/{orkestrel-build-application → orkestrel-build}/agents/openai.yaml +1 -1
  27. package/dist/host/agents/skills/orkestrel-debrief/SKILL.md +8 -16
  28. package/dist/host/agents/skills/orkestrel-debrief/references/instruction-audit.md +3 -3
  29. package/dist/host/agents/skills/orkestrel-debrief/references/retention.md +13 -13
  30. package/dist/host/agents/skills/orkestrel-dispatch/SKILL.md +61 -0
  31. package/dist/host/agents/skills/orkestrel-dispatch/agents/openai.yaml +4 -0
  32. package/dist/host/agents/skills/orkestrel-dispatch/references/bench.md +25 -0
  33. package/dist/host/agents/skills/orkestrel-dispatch/references/launch.md +32 -0
  34. package/dist/host/agents/skills/orkestrel-dispatch/scripts/bench.ts +259 -0
  35. package/dist/host/agents/skills/orkestrel-dispatch/scripts/brief.ts +110 -0
  36. package/dist/host/agents/skills/orkestrel-dispatch/scripts/cite.ts +115 -0
  37. package/dist/host/agents/skills/orkestrel-dispatch/scripts/helpers.ts +239 -0
  38. package/dist/host/agents/skills/orkestrel-dispatch/scripts/launch.ts +124 -0
  39. package/dist/host/agents/skills/orkestrel-dispatch/scripts/login.ts +123 -0
  40. package/dist/host/agents/skills/orkestrel-dispatch/scripts/result.ts +129 -0
  41. package/dist/host/agents/skills/orkestrel-dispatch/scripts/sweep.ts +157 -0
  42. package/dist/host/agents/skills/orkestrel-falsify/SKILL.md +42 -193
  43. package/dist/host/agents/skills/orkestrel-falsify/references/brief.md +38 -108
  44. package/dist/host/agents/skills/orkestrel-falsify/references/reconcile.md +35 -134
  45. package/dist/host/agents/skills/{orkestrel-harden-package → orkestrel-harden}/SKILL.md +10 -14
  46. package/dist/host/agents/skills/{orkestrel-harden-package → orkestrel-harden}/agents/openai.yaml +1 -1
  47. package/dist/host/agents/skills/{orkestrel-harden-package → orkestrel-harden}/references/hardening.md +3 -4
  48. package/dist/host/agents/skills/orkestrel-harden/scripts/discovery.ts +228 -0
  49. package/dist/host/agents/skills/{orkestrel-prove-journey → orkestrel-journey}/SKILL.md +15 -23
  50. package/dist/host/agents/skills/orkestrel-journey/agents/openai.yaml +4 -0
  51. package/dist/host/agents/skills/{orkestrel-prove-journey → orkestrel-journey}/references/captures.md +1 -1
  52. package/dist/host/agents/skills/{orkestrel-polish-surface → orkestrel-polish}/SKILL.md +25 -33
  53. package/dist/host/agents/skills/{orkestrel-polish-surface → orkestrel-polish}/agents/openai.yaml +1 -1
  54. package/dist/host/agents/skills/{orkestrel-polish-surface → orkestrel-polish}/references/capture-harness.md +3 -3
  55. package/dist/host/agents/skills/orkestrel-publish/SKILL.md +33 -20
  56. package/dist/host/agents/skills/orkestrel-publish/references/release.md +39 -0
  57. package/dist/host/agents/skills/orkestrel-publish/references/wave.md +22 -21
  58. package/dist/host/agents/skills/orkestrel-publish/references/window.md +27 -14
  59. package/dist/host/agents/skills/orkestrel-publish/scripts/compare.ts +220 -0
  60. package/dist/host/agents/skills/orkestrel-publish/scripts/pins.ts +114 -0
  61. package/dist/host/agents/skills/orkestrel-publish/scripts/wave.ts +629 -0
  62. package/dist/host/agents/skills/orkestrel-publish/scripts/window.ts +242 -0
  63. package/dist/host/agents/skills/orkestrel-scout/SKILL.md +28 -0
  64. package/dist/host/agents/skills/orkestrel-scout/agents/openai.yaml +4 -0
  65. package/dist/host/agents/skills/orkestrel-scout/scripts/map.ts +352 -0
  66. package/dist/host/agents/templates/brief.md +21 -142
  67. package/dist/host/agents/transports/claude-cli.md +21 -0
  68. package/dist/host/agents/transports/codex.md +38 -159
  69. package/dist/host/agents/transports/cursor.md +16 -65
  70. package/dist/host/claude/AGENTS.md +38 -0
  71. package/dist/host/claude/agents/analyst.md +14 -53
  72. package/dist/host/claude/agents/astra.md +26 -0
  73. package/dist/host/claude/agents/builder.md +14 -30
  74. package/dist/host/claude/agents/checker.md +13 -57
  75. package/dist/host/claude/agents/distiller.md +11 -26
  76. package/dist/host/claude/agents/grok.md +12 -35
  77. package/dist/host/claude/agents/opus.md +14 -30
  78. package/dist/host/claude/agents/planner.md +10 -44
  79. package/dist/host/claude/agents/researcher.md +11 -30
  80. package/dist/host/claude/agents/reviewer.md +11 -95
  81. package/dist/host/claude/agents/scout.md +9 -23
  82. package/dist/host/claude/agents/verifier.md +15 -33
  83. package/dist/host/claude/rules/documentation.md +8 -2
  84. package/dist/host/claude/rules/portability.md +7 -1
  85. package/dist/host/claude/rules/quality.md +36 -96
  86. package/dist/host/claude/rules/styles.md +3 -0
  87. package/dist/host/claude/rules/tests.md +6 -3
  88. package/dist/host/claude/rules/workspace.md +19 -15
  89. package/dist/host/claude/rules/writing.md +57 -108
  90. package/dist/host/claude/settings.json +5 -3
  91. package/dist/host/claude/skills/enterprise-bootstrap/SKILL.md +1 -1
  92. package/dist/host/claude/skills/{orkestrel-align-packages → orkestrel-align}/SKILL.md +2 -2
  93. package/dist/host/claude/skills/{orkestrel-build-application → orkestrel-build}/SKILL.md +2 -2
  94. package/dist/host/claude/skills/orkestrel-dispatch/SKILL.md +11 -0
  95. package/dist/host/claude/skills/orkestrel-falsify/SKILL.md +2 -1
  96. package/dist/host/claude/skills/{orkestrel-harden-package → orkestrel-harden}/SKILL.md +2 -2
  97. package/dist/host/claude/skills/{orkestrel-prove-journey → orkestrel-journey}/SKILL.md +2 -2
  98. package/dist/host/claude/skills/orkestrel-polish/SKILL.md +12 -0
  99. package/dist/host/claude/skills/orkestrel-scout/SKILL.md +11 -0
  100. package/dist/host/codex/agents/analyst.toml +14 -31
  101. package/dist/host/codex/agents/astra.toml +25 -0
  102. package/dist/host/codex/agents/builder.toml +13 -20
  103. package/dist/host/codex/agents/checker.toml +13 -27
  104. package/dist/host/codex/agents/distiller.toml +9 -22
  105. package/dist/host/codex/agents/grok.toml +11 -30
  106. package/dist/host/codex/agents/opus.toml +14 -22
  107. package/dist/host/codex/agents/orkestrel.toml +1 -1
  108. package/dist/host/codex/agents/planner.toml +11 -28
  109. package/dist/host/codex/agents/researcher.toml +10 -22
  110. package/dist/host/codex/agents/reviewer.toml +11 -27
  111. package/dist/host/codex/agents/scout.toml +11 -17
  112. package/dist/host/codex/agents/verifier.toml +16 -12
  113. package/dist/host/codex/config.toml +18 -21
  114. package/dist/host/cursor/mcp.json +0 -4
  115. package/dist/host/cursor/rules/orchestration.mdc +12 -20
  116. package/dist/host/dotfiles/mcp.json +0 -4
  117. package/dist/host/dotfiles/oxlintrc.json +7 -0
  118. package/dist/host/guides/probe.md +9 -9
  119. package/dist/host/guides/scaffold.md +117 -71
  120. package/dist/host/guides/test.md +1 -1
  121. package/dist/host/manifest.json +321 -184
  122. package/dist/host/scripts/codex.sh +0 -0
  123. package/dist/host/scripts/cursor.sh +0 -0
  124. package/dist/host/scripts/deps.sh +0 -0
  125. package/dist/host/scripts/ollama.sh +0 -0
  126. package/dist/host/tests/config.test.ts +68 -46
  127. package/dist/host/tests/policy.test.ts +1 -5
  128. package/dist/host/tests/setupPolicy.ts +179 -4
  129. package/dist/src/core/index.cjs +255 -84
  130. package/dist/src/core/index.cjs.map +1 -1
  131. package/dist/src/core/index.d.cts +94 -29
  132. package/dist/src/core/index.d.ts +94 -29
  133. package/dist/src/core/index.js +253 -85
  134. package/dist/src/core/index.js.map +1 -1
  135. package/dist/src/server/index.cjs +55 -9
  136. package/dist/src/server/index.cjs.map +1 -1
  137. package/dist/src/server/index.d.cts +29 -4
  138. package/dist/src/server/index.d.ts +29 -4
  139. package/dist/src/server/index.js +56 -11
  140. package/dist/src/server/index.js.map +1 -1
  141. package/package.json +15 -11
  142. package/dist/host/CLAUDE.md +0 -61
  143. package/dist/host/agents/skills/orkestrel-prove-journey/agents/openai.yaml +0 -4
  144. package/dist/host/agents/transports/claude.md +0 -49
  145. package/dist/host/claude/agents/application.md +0 -36
  146. package/dist/host/claude/agents/sol.md +0 -61
  147. package/dist/host/claude/skills/orkestrel-polish-surface/SKILL.md +0 -12
  148. package/dist/host/codex/agents/application.toml +0 -25
  149. package/dist/host/codex/agents/sol.toml +0 -19
  150. /package/dist/host/agents/skills/{orkestrel-align-packages → orkestrel-align}/references/integration.md +0 -0
  151. /package/dist/host/agents/skills/{orkestrel-harden-package → orkestrel-harden}/references/centralization.md +0 -0
  152. /package/dist/host/agents/skills/{orkestrel-harden-package → orkestrel-harden}/references/contract.md +0 -0
  153. /package/dist/host/agents/skills/{orkestrel-harden-package → orkestrel-harden}/references/research.md +0 -0
  154. /package/dist/host/agents/skills/{orkestrel-prove-journey → orkestrel-journey}/references/decide.md +0 -0
  155. /package/dist/host/agents/skills/{orkestrel-prove-journey → orkestrel-journey}/references/layer.md +0 -0
  156. /package/dist/host/agents/skills/{orkestrel-prove-journey → orkestrel-journey}/references/statechart.md +0 -0
  157. /package/dist/host/agents/skills/{orkestrel-prove-journey → orkestrel-journey}/references/styles.md +0 -0
@@ -1,156 +1,57 @@
1
1
  # Reconciling a round and ruling on it
2
2
 
3
- The auditors return. Neither accepts; the orchestrator does. This is where a round becomes a
4
- decision, and it is not delegable.
3
+ The Orchestrator reconciles and accepts; no lane does. Run a finding rather than argue it.
5
4
 
6
- ## Reproduce before you act
5
+ ## Reproduce before acting
7
6
 
8
- The rule beneath this whole section: **run it rather than argue it.** Every judgement that follows
9
- is cheap once the probe exists and unreliable until it does.
7
+ - Treat a lane's finding as a hypothesis until you have run it against the built output. Reproduce every `BROKEN` and every outside finding by hand before it enters a fix brief.
8
+ - Construct the hostile input outside the `try` and guard only the call under test, so a harness fault (a missing import, a wrong arity, a `require` in ESM) crashes loudly instead of reading as the finding.
9
+ - Record which outcome the reproduction produced: confirmed and wider than reported; confirmed and bounded smaller; or evaporated because the input could not exercise what it claimed.
10
+ - Separate a dead finding from a dead vector: a vector the compiler rejects refutes the vector alone. Re-derive one the types admit and record which vector was tested.
11
+ - Apply the same discipline to your own probes; a probe whose input cannot reach the code under test reads exactly like a real pass.
10
12
 
11
- An auditor's finding is a **hypothesis** until the orchestrator has run it. Reproduce every sharp
12
- claim by hand, against the built output, before it enters a fix brief.
13
+ ## Resolve a disagreement
13
14
 
14
- **Build the hostile input outside the `try`.** A probe that wraps construction and invocation in one
15
- catch cannot distinguish _the subject threw_ from _my harness threw_ — a missing import, a wrong
16
- arity, a `require` in an ESM context all surface as the finding you were hoping to see. Construct
17
- first, let harness failures crash loudly, and only guard the call under test.
15
+ When lanes return opposite verdicts on one claim, reproduce first; never average them and never prefer the engine you trust more. Then name the question each lane answered.
18
16
 
19
- Reproduction produces these outcomes, and all of them matter:
17
+ - Both right about different objects: the claim was a universal carrying more than one subject. Split it, keep `BROKEN` on any broken subclaim (`SPLIT-CLAIM` is a note, never a verdict value), and carry the split into the successor brief.
18
+ - Both right about different halves of one claim number: split and renumber.
19
+ - One right on the mechanism, the other on the criterion: take both constraints; the reconciled ruling satisfies both.
20
+ - An argument that an input class is unreachable: run it and show the reachable consequence.
20
21
 
21
- - the finding **confirms** and is often **wider** than reported — the reproduction reaches doors the
22
- auditor did not try;
23
- - the finding **confirms but is bounded smaller** — real, and not where the auditor thought;
24
- - the finding **evaporates**, because the auditor's input could not exercise what it claimed to test.
25
-
26
- Separate a dead finding from a dead vector before evaporating anything. A reported vector the
27
- compiler rejects refutes the vector alone; re-derive one the types admit, and record which vector
28
- was actually tested.
29
-
30
- The same reproduction discipline applies to your own probes. A probe whose input cannot reach the
31
- code under test reports a pass that means nothing, and it will read exactly like a real pass.
32
-
33
- ## A disagreement is rarely a tie
34
-
35
- When auditors return opposite verdicts on one claim, do not average them and do not prefer the
36
- engine you trust more. **Reproduce first** — running the disagreement settles most of them outright,
37
- and it is the only method that can also find what neither auditor saw. Then find the question each
38
- one answered. The common shapes:
39
-
40
- - **Both right about different objects.** One tested a case the other did not construct. This is a
41
- `SPLIT-CLAIM`: the claim was a universal that carried more than one subject. It is **not** a
42
- further verdict value — one falsifying input makes a universal claim `BROKEN`, and succeeding on a
43
- different object does not undo that. Split it, keep the original `BROKEN` if any subclaim is
44
- broken, and carry the split into the successor brief.
45
- - **Both right about different halves of one claim number.** Same resolution: the claim number was
46
- carrying more than one claim. Split and renumber.
47
- - **One right on the mechanism, the other on the criterion.** Take both. The reconciled ruling is
48
- frequently neither proposal, and better than either, because each supplied a constraint the other
49
- violated.
50
- - **The consequence disproves the premise.** An argument that an input class is unreachable is
51
- answered by running it and showing what the reachable consequence is.
52
-
53
- Record which engine was right and on what. A round whose disagreements are smoothed over teaches
54
- nothing to the next one.
22
+ Record which engine was right and on what.
55
23
 
56
24
  ## Evidence custody
57
25
 
58
- Blind reports are immutable, and their independence is a property of the record, not of anyone's
59
- memory. A reader six months out must be able to tell an unbiased blind verdict from one produced
60
- after an auditor saw its counterpart's evidence — otherwise the whole value of running blind is
61
- unverifiable after the fact.
62
-
63
- These rules are enforceable:
64
-
65
- 1. **A returned verdict is never edited** — not by the auditor, not by the orchestrator.
66
- 2. **Anything an auditor says after seeing another's report is a separate file beside that verdict**,
67
- under `.orkestrel/<package>/`, named for the unit and the exposure, recording what was shown and
68
- to whom. It is a durable record, not a journal, so it survives the campaign sweep.
69
-
70
- That exchange is a **fallback, not a phase.** Reproduction comes first and settles most
71
- disagreements. Reach for an exchange only when a specific factual question survives reproduction,
72
- scope it to that question, initiate it yourself, and run it once — a second exchange is negotiation.
73
-
74
- **Ask the auditor to attack the other's evidence on that question. Never ask it to resolve the
75
- disagreement, reconsider its position, or say whether the other changed its mind.** Constraining
76
- when, who, scope and frequency does nothing about convergence if the instruction itself invites it,
77
- and "does their evidence change your claim?" is the convergence prompt in its purest form.
26
+ - Never edit a returned verdict, whoever wrote it.
27
+ - Write anything a lane says after seeing another lane's report as a separate file beside the verdict under `.orkestrel/<package>/`, named for the unit and the exposure, recording what was shown and to whom. That file is a durable record, never a journal: it survives the campaign sweep, and its contents are promoted into the acceptance record before the folder retires.
28
+ - Reach for such an exchange only when a specific factual question survives reproduction; scope it to that question, initiate it yourself, and run it once. Ask the lane to attack the other's evidence on that question; never ask it to resolve the disagreement, reconsider its position, or say whether the other changed its mind.
78
29
 
79
30
  ## Bound the finding
80
31
 
81
- State what is **not** broken, and why the adjacent behaviour that looks identical is correct. A
82
- finding without a boundary is an alarm, and alarms get discounted wholesale — including the true
83
- ones next to them.
84
-
85
- These boundaries earn their keep:
86
-
87
- - **Credit what the round got right.** If the hostile inputs adjacent to the hole are correctly
88
- contained, say so and list them. It sharpens the finding to a point instead of an area.
89
- - **Show where the same-looking answer is correct.** When several exports answer a hostile input the
90
- same way and only one is wrong, name what makes the difference — usually that the correct ones
91
- agree with a documented view, and the wrong one reads on an axis it then ignores.
32
+ - State what is not broken and why the adjacent behavior that looks identical is correct.
33
+ - List the hostile inputs adjacent to the hole that are correctly contained.
34
+ - Where several exports answer a hostile input the same way and one is wrong, name what makes the difference.
92
35
 
93
36
  ## Bound the fix before briefing it
94
37
 
95
- Establish what over-correcting would break, and put it in the fix brief as a constraint. Both ends
96
- are usually wrong:
97
-
98
- - **too little** — a patch to the one function, leaving the package holding conflicting standards
99
- for the same thing, which is the inconsistency that produced the finding;
100
- - **too much** — adopting the strictest sibling's rule verbatim, breaking a legitimate caller
101
- pattern, and tripping "no refusal was widened into a regression" in the next round.
102
-
103
- Find the rule that fits both. It is usually about **agreement** rather than about categories — what
104
- a reader reads, its answer must carry — and it dissolves the special cases rather than enumerating
105
- them.
106
-
107
- Measure a proposed fix before adopting it; it is itself a claim. Run it against the set it must
108
- not break, including every case an earlier round pinned. Where it fails that set, document the
109
- limit on the helper that owns it and pin the limit with a test that names it as one. A heuristic
110
- that trades one wrong answer for another fails quietly; a stated boundary does not.
111
-
112
- Where the choice is genuinely open, it is a design judgement with a subjective and an objective
113
- half, and it goes to a blind design pass before code. Ruling it unilaterally is how a fix round
114
- becomes the next audit's finding.
115
-
116
- ## Certifying an instrument
117
-
118
- The control-population law in `.claude/rules/quality.md` binds here without restatement. What it
119
- leaves this round is the procedure.
120
-
121
- When a round certifies an instrument — a pin, an identity check, a generated sweep — the controls
122
- are usually drawn from whatever the instrument obviously covers, because that is where the examples
123
- take the least construction. That sampling proves discrimination _within_ the population and is
124
- routinely reported as proof the instrument works.
125
-
126
- So before running controls, write down the instrument's **membership rule** in one sentence, then
127
- ask what the rule excludes. Draw at least one control from there. These shapes have already cost a
128
- round each:
129
-
130
- - an AST comparison whose controls were all drawn from the literal classes present in the bodies it
131
- guarded, blind to the classes absent from them;
132
- - a call-closure pin whose controls were all body-reachable functions, green for a function reached
133
- only through a parameter default.
134
-
135
- Then write the sentences that matter: what the controls established, and what they did not. What
136
- they did not establish is the one that gets skipped, and skipping it is how an instrument's
137
- credibility outruns its evidence.
38
+ - Put in the fix brief what over-correcting would break. Refuse both a patch to one function that leaves the package holding two standards for the same thing, and the strictest sibling's rule adopted verbatim, which breaks a legitimate caller.
39
+ - Find the rule that fits both ends: a rule about agreement (what a reader reads, its answer must carry) that dissolves the special cases rather than enumerating them.
40
+ - Measure a proposed fix as a claim: run it against the set it must not break, including every case an earlier round pinned. Where it fails that set, document the limit on the helper that owns it and pin the limit with a test that names it.
41
+ - Where the choice is open, it has a subjective and an objective half: send it to a blind design pass before code.
138
42
 
139
- ## Ruling
43
+ ## Certify an instrument
140
44
 
141
- - Every retained finding names the fix-brief item that carries it. A finding with no carrier is a
142
- dropped finding; walk the list once and check.
143
- - Drop, **on the record**, anything neither auditor can substantiate against the evidence.
144
- - Promote anything that must outlive the round into a durable artifact before the working files are
145
- swept. What lives only in a scratch file did not survive.
146
- - The fix round's auditor is an engine that did not write it.
147
- - The next round's brief is this round's successor.
45
+ `.claude/rules/quality.md` § Instruments binds the control-population law. The procedure:
148
46
 
149
- ## The threshold
47
+ 1. Write the instrument's membership rule in one sentence.
48
+ 2. Name what the rule excludes and draw at least one control from there. Controls drawn only from what the instrument obviously covers prove discrimination inside the population and nothing outside it: an AST comparison blind to literal classes absent from its bodies, a call-closure pin green for a function reached only through a parameter default.
49
+ 3. Write what the controls established and what they did not.
150
50
 
151
- Accept when the brief's claims are **satisfied on evidence** — the `PASS` terminal line the skill
152
- defines, against a claim set that covers what the subject owns. Not green gates.
51
+ ## Rule
153
52
 
154
- A round that finds something is the process working. A round that finds nothing because nobody tried
155
- is the failure; a round re-run because an attack can still be imagined never ends. Bound the claim
156
- set at the brief, rule on what it returned, and close.
53
+ - Every retained finding names the fix-brief item that carries it; walk the list once.
54
+ - Drop, on the record, anything no lane can substantiate against the evidence.
55
+ - Promote anything that must outlive the round into a durable artifact before the working files are swept.
56
+ - The fix round's lane is an engine that did not write the fix. The next round's brief is this round's successor.
57
+ - Accept when the claims are satisfied on evidence (`VERDICT: PASS` against a claim set covering what the subject owns), never on green gates alone. Bound the claim set at the brief, rule on what it returned, and close; never re-run because an attack can still be imagined.
@@ -1,23 +1,19 @@
1
1
  ---
2
- name: orkestrel-harden-package
2
+ name: orkestrel-harden
3
3
  description: Research, audit, refactor, implement, centralize, test, document, and locally verify an individual Orkestrel TypeScript package to enterprise-grade production readiness under the repository's current AGENTS.md. Use when asked to fill missing or deferred capabilities, compare upstream or legacy implementations, salvage prior art, centralize source or test declarations, eliminate nested functions or superfluous wrappers, maximize declared @orkestrel dependencies—especially @orkestrel/contract—or add rigorous real-implementation and live-service tests. Select only the phases required by a narrow request; run the full workflow for production readiness or comprehensive hardening.
4
4
  ---
5
5
 
6
6
  # Harden an Orkestrel package
7
7
 
8
- ## Load authority
8
+ ## Read
9
9
 
10
- Read the current files in this order:
10
+ `AGENTS.md` § Authority and loading names the files every unit reads. This skill adds the references the selected lane names, the contracts, barrels, manifest, and configuration that lane needs, and every decision-bearing implementation file the user names, read first-hand by the decision owner. Delegate bulk reconnaissance and supporting research, never the design decision. Preserve dirty and user-owned work.
11
11
 
12
- 1. `AGENTS.md`.
13
- 2. Every applicable `.claude/rules/*.md`.
14
- 3. Select the work lane § "Select the work lane" names, and read every reference that lane requires.
15
- 4. `guides/README.md`, the governing package/domain guide, and `ROADMAP.md` when present.
16
- 5. The authoritative `*/types.ts`, public barrels, `package.json`, build/test configuration, and decision-bearing implementation files.
12
+ ## Scripts
17
13
 
18
- Treat the current user instruction as authoritative. Treat repository rules as the coding contract and this skill as the workflow. Preserve dirty and user-owned work.
19
-
20
- The decision owner must read the governing types and every implementation file the user names directly. Delegate bulk reconnaissance or supporting research, not the final design decision.
14
+ | Script | Does |
15
+ | ---------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
16
+ | `scripts/discovery.ts` | `[--config vite.config.ts] [--projects a,b] [--json]`, run from the checkout root: follows every root script chain to the Vitest projects and test files it gates, runs `vitest list --json` once, and reports per project the files, tests, and gate, plus every `.skip(`, `.todo(`, `.skipIf(`, `.runIf(`, `retry:`, and `timeout:` marker. Exit 3 when a collected project has no gate, a named project collects nothing, or a test file under `tests/` is collected by no project. |
21
17
 
22
18
  ## Select the work lane
23
19
 
@@ -44,14 +40,14 @@ Load [hardening.md](references/hardening.md) for the hardening lane and for any
44
40
  3. **Establish the intended contract.** Build a capability/defect matrix. Separate verified fact from inference. Mark each row implement, repair, retain, or exclude with a reason. The matrix is fixed at this step and is the campaign's definition of done: every later step serves a row, and work that serves no row belongs to the next campaign. The row set is fixed; the planned work that closes it is not. Re-baseline that work at each phase boundary per `.agents/orchestration.md`, which owns the step and its boundary with rescoping.
45
41
  4. **Design types first.** Update guide/spec intent and `*/types.ts` before implementation, under the root design laws. A contract that needs a compatibility shim is the wrong contract.
46
42
  5. **Implement completely.** Finish every in-scope branch and reuse the exact installed Orkestrel primitives whose semantics match. The root completion law decides what may not be left behind.
47
- 6. **Prove each defect before repairing it.** A repair begins with a test that fails for that defect: record the exact command and its failing count before the fix and the same command's passing count after. A repair with no red-then-green record is unproven.
43
+ 6. **Prove each defect before repairing it** per the defect rule in `AGENTS.md` § Work loop. A repair with no red-then-green record is unproven.
48
44
  7. **Consolidate.** Run the complete centralization and wrapper sweep. Update all call sites to the real symbol rather than leaving aliases or 1:1 delegates.
49
45
  8. **Challenge seams.** Add deterministic tests for invariants, boundaries, failures, lifecycle, cleanup, cancellation, concurrency, hostile input, and resource pressure as applicable, under the test rules' real-implementation law.
50
46
  9. **Use live services deliberately.** Put real external services/models in their dedicated project, require readiness, and make each request minimally sufficient, stable across the service's nondeterminism, and behaviorally meaningful. When the claim is that a foreign client can use this package, drive one representative real client end to end.
51
47
  10. **Document the final behavior.** Update the governing guide, examples, method tables, limitations, and parity coverage. Document architectural limits honestly.
52
- 11. **Audit completion.** Inspect test discovery, `.todo`/`.skip`/conditional skip use, source/test helper duplication, exports, environment isolation, unexpected text corruption, and the entire diff.
48
+ 11. **Audit completion.** Run `node .agents/skills/orkestrel-harden/scripts/discovery.ts` and rule on every flag and marker it reports, then inspect source/test helper duplication, exports, environment isolation, unexpected text corruption, and the entire diff.
53
49
  12. **Verify.** Run the repository-prescribed gates in order and inspect the generated outputs relevant to the request.
54
- 13. **Review independently, and never by the author.** When orchestration is available, run the adversarial pass — subjective design fit and objective correctness — plus a mechanical checker, per `.agents/orchestration.md`. Add a dedicated adversarial round for security, concurrency, destructive paths, or external input. A unit's auditor is an engine that did not write it; same-engine re-review returns the author's own blind spot. Resolve every required finding, then rerun affected verification.
50
+ 13. **Review by the size gate, never by the author.** Run the review `.agents/orchestration.md` § Size gate names for the change, resolve every required finding, then rerun the affected scoped verification.
55
51
 
56
52
  ## Accept the result
57
53
 
@@ -1,4 +1,4 @@
1
1
  interface:
2
2
  display_name: 'Harden Orkestrel Package'
3
3
  short_description: 'Research, centralize, test, and harden one package'
4
- default_prompt: 'Use $orkestrel-harden-package to bring this package to enterprise-grade production readiness.'
4
+ default_prompt: 'Use $orkestrel-harden to bring this package to enterprise-grade production readiness.'
@@ -82,11 +82,10 @@ For every fetch, write, delete, extraction, path, protocol, or authentication bo
82
82
 
83
83
  ## Audit the tests themselves
84
84
 
85
- Verify:
85
+ Run `node .agents/skills/orkestrel-harden/scripts/discovery.ts` for the discovery census: which project collects each file, which gate reaches each project, and where a file carries `.skip`, `.todo`, a conditional skip, a retry, or a timeout. Then verify:
86
86
 
87
- - every test file is discovered by the intended project;
88
- - targeted commands run the expected count and environment;
89
- - `.todo`, `.skip`, conditional skips, retries, and generous timeouts are justified;
87
+ - every flag the census reports is closed or ruled with a reason;
88
+ - every marker it reports is justified;
90
89
  - no current-scope requirement is represented only by a todo;
91
90
  - assertions can fail for the defect they claim to catch;
92
91
  - tests observe public outcomes rather than private implementation;
@@ -0,0 +1,228 @@
1
+ // Census of test discovery: which projects the gates reach, what each collects, and where a suite
2
+ // skips. Run from the checkout root:
3
+ // node .agents/skills/orkestrel-harden/scripts/discovery.ts [--config vite.config.ts] [--projects a,b] [--json]
4
+ // The script follows every root script chain in package.json (a script no other script invokes) to
5
+ // the `--project` names its Vitest scripts gate and the test files its `node` scripts run, runs
6
+ // `vitest list --json` once through the local vitest entry to read what each project collects, and
7
+ // reads every collected file for `.skip(`, `.todo(`, `.skipIf(`, `.runIf(`, `retry:`, and
8
+ // `timeout:`. It flags a collected project no root chain reaches, a named project that collects
9
+ // nothing, and a test file under tests/ that no project collects and no root script runs directly.
10
+ // A project with no test file and no gate is outside the census. Exit 0 with no flag, 3 with one, 2 when Vitest cannot list, 64 on
11
+ // usage.
12
+ import { spawnSync } from 'node:child_process'
13
+ import { existsSync, readFileSync } from 'node:fs'
14
+ import { relative, resolve } from 'node:path'
15
+ import {
16
+ listFiles,
17
+ readMissingFlags,
18
+ readOption,
19
+ } from '../../orkestrel-dispatch/scripts/helpers.ts'
20
+
21
+ const VITEST = 'node_modules/vitest/vitest.mjs'
22
+ const MARKERS: readonly string[] = ['.skip(', '.todo(', '.skipIf(', '.runIf(', 'retry:', 'timeout:']
23
+
24
+ interface Collected {
25
+ readonly file: string
26
+ readonly projectName: string
27
+ }
28
+
29
+ interface Gates {
30
+ readonly projects: ReadonlyMap<string, string>
31
+ readonly files: ReadonlyMap<string, string>
32
+ }
33
+
34
+ interface Project {
35
+ readonly name: string
36
+ readonly gate: string | undefined
37
+ readonly files: number
38
+ readonly tests: number
39
+ }
40
+
41
+ interface Census {
42
+ readonly file: string
43
+ readonly projects: readonly string[]
44
+ readonly markers: Readonly<Record<string, number>>
45
+ }
46
+
47
+ function readScripts(): Readonly<Record<string, string>> {
48
+ const parsed: unknown = JSON.parse(readFileSync('package.json', 'utf8'))
49
+ if (typeof parsed !== 'object' || parsed === null) return {}
50
+ const scripts = Object.fromEntries(Object.entries(parsed)).scripts
51
+ if (typeof scripts !== 'object' || scripts === null) return {}
52
+ const record: Record<string, string> = {}
53
+ for (const [key, value] of Object.entries(scripts))
54
+ if (typeof value === 'string') record[key] = value
55
+ return record
56
+ }
57
+
58
+ function listInvocations(text: string): readonly string[] {
59
+ return [...text.matchAll(/npm run ([A-Za-z0-9:_-]+)/gu)].flatMap((match) =>
60
+ match[1] === undefined ? [] : [match[1]],
61
+ )
62
+ }
63
+
64
+ function readGates(scripts: Readonly<Record<string, string>>): Gates {
65
+ const invoked = new Set(Object.values(scripts).flatMap(listInvocations))
66
+ const projects = new Map<string, string>()
67
+ const files = new Map<string, string>()
68
+ for (const root of Object.keys(scripts).filter((name) => !invoked.has(name))) {
69
+ const visited = new Set<string>()
70
+ const pending = [root]
71
+ while (pending.length > 0) {
72
+ const name = pending.pop()
73
+ if (name === undefined || visited.has(name)) continue
74
+ visited.add(name)
75
+ const text = scripts[name]
76
+ if (text === undefined) continue
77
+ pending.push(...listInvocations(text))
78
+ const gate = root === name ? name : `${root} > ${name}`
79
+ if (/\bvitest\b/u.test(text)) {
80
+ for (const match of text.matchAll(/--project[= ]([A-Za-z0-9:_-]+)/gu)) {
81
+ if (match[1] !== undefined && !projects.has(match[1])) projects.set(match[1], gate)
82
+ }
83
+ }
84
+ for (const match of text.matchAll(
85
+ /(?:^|\s)((?:tests|src|app)\/[A-Za-z0-9_./-]+\.test\.[cm]?ts)\b/gu,
86
+ )) {
87
+ if (match[1] !== undefined && !files.has(match[1])) files.set(match[1], gate)
88
+ }
89
+ }
90
+ }
91
+ return { projects, files }
92
+ }
93
+
94
+ function listCollected(config: string): readonly Collected[] | undefined {
95
+ const result = spawnSync(process.execPath, [VITEST, 'list', '--config', config, '--json'], {
96
+ encoding: 'utf8',
97
+ maxBuffer: 64 * 1024 * 1024,
98
+ windowsHide: true,
99
+ })
100
+ if (result.status !== 0) return undefined
101
+ const start = result.stdout.indexOf('[')
102
+ if (start === -1) return undefined
103
+ const parsed: unknown = JSON.parse(result.stdout.slice(start))
104
+ if (!Array.isArray(parsed)) return undefined
105
+ const collected: Collected[] = []
106
+ for (const entry of parsed) {
107
+ if (typeof entry !== 'object' || entry === null) continue
108
+ const record = Object.fromEntries(Object.entries(entry))
109
+ if (typeof record.file === 'string' && typeof record.projectName === 'string') {
110
+ collected.push({
111
+ file: relative(process.cwd(), record.file).split('\\').join('/'),
112
+ projectName: record.projectName,
113
+ })
114
+ }
115
+ }
116
+ return collected
117
+ }
118
+
119
+ function listTestFiles(directory: string): readonly string[] {
120
+ if (!existsSync(directory)) return []
121
+ return listFiles(directory)
122
+ .map((file) => file.path)
123
+ .filter((path) => /\.test\.[cm]?ts$/u.test(path))
124
+ }
125
+
126
+ function countMarkers(file: string): Readonly<Record<string, number>> {
127
+ const text = readFileSync(file, 'utf8')
128
+ const counts: Record<string, number> = {}
129
+ for (const marker of MARKERS) {
130
+ const count = text.split(marker).length - 1
131
+ if (count > 0) counts[marker] = count
132
+ }
133
+ return counts
134
+ }
135
+
136
+ function main(argv: readonly string[]): number {
137
+ const missing = readMissingFlags(argv, ['--config', '--projects'])
138
+ if (missing.length > 0) {
139
+ console.error(`discovery: ${missing.join(', ')} given with no value`)
140
+ return 64
141
+ }
142
+ const config = readOption(argv, '--config') ?? 'vite.config.ts'
143
+ if (!existsSync(config) || !existsSync('package.json')) {
144
+ console.error('usage: discovery.ts [--config vite.config.ts] [--projects a,b] [--json]')
145
+ return 64
146
+ }
147
+ if (!existsSync(VITEST)) {
148
+ console.error(`discovery: ${VITEST} is missing; run npm ci first`)
149
+ return 2
150
+ }
151
+ const gates = readGates(readScripts())
152
+ const collected = listCollected(config)
153
+ if (collected === undefined) {
154
+ console.error('discovery: vitest list failed; run it bare to read the diagnostic')
155
+ return 2
156
+ }
157
+ const named = (readOption(argv, '--projects') ?? '').split(',').filter((name) => name !== '')
158
+ const names = new Set<string>([
159
+ ...gates.projects.keys(),
160
+ ...collected.map((entry) => entry.projectName),
161
+ ...named,
162
+ ])
163
+ const projects: Project[] = [...names].sort().map((name) => {
164
+ const mine = collected.filter((entry) => entry.projectName === name)
165
+ const fileGate = mine
166
+ .map((entry) => gates.files.get(entry.file))
167
+ .find((gate) => gate !== undefined)
168
+ return {
169
+ name,
170
+ gate: gates.projects.get(name) ?? fileGate,
171
+ files: new Set(mine.map((entry) => entry.file)).size,
172
+ tests: mine.length,
173
+ }
174
+ })
175
+ const collectedFiles = new Set(collected.map((entry) => entry.file))
176
+ const undiscovered = listTestFiles('tests')
177
+ .filter((file) => !collectedFiles.has(file) && !gates.files.has(file))
178
+ .sort()
179
+ const ungated = projects
180
+ .filter((project) => project.gate === undefined && project.tests > 0)
181
+ .map((project) => project.name)
182
+ // A project no chain reaches is a workbench (the `probe` project collects `tmp/probes/**`), so an
183
+ // empty one is reported, never flagged.
184
+ const empty = projects
185
+ .filter((project) => project.tests === 0 && project.gate?.includes(' > ') === true)
186
+ .map((project) => project.name)
187
+ const workbenches = projects
188
+ .filter((project) => project.tests === 0 && project.gate?.includes(' > ') !== true)
189
+ .map((project) => project.name)
190
+ const census: Census[] = [...collectedFiles].sort().map((file) => ({
191
+ file,
192
+ projects: [
193
+ ...new Set(
194
+ collected.filter((entry) => entry.file === file).map((entry) => entry.projectName),
195
+ ),
196
+ ],
197
+ markers: countMarkers(resolve(file)),
198
+ }))
199
+ const flagged = undiscovered.length + ungated.length + empty.length > 0
200
+ if (argv.includes('--json')) {
201
+ console.log(
202
+ JSON.stringify({ config, projects, ungated, empty, workbenches, undiscovered, census }),
203
+ )
204
+ } else {
205
+ for (const project of projects) {
206
+ console.log(
207
+ `discovery: ${project.name.padEnd(14)} ${String(project.files).padStart(3)} file(s) ${String(project.tests).padStart(5)} test(s) gate=${project.gate ?? 'none'}`,
208
+ )
209
+ }
210
+ for (const name of ungated)
211
+ console.log(`discovery: ${name} collects tests but no root script chain reaches it`)
212
+ for (const name of empty) console.log(`discovery: ${name} collects nothing`)
213
+ for (const name of workbenches)
214
+ console.log(`discovery: ${name} is a workbench no chain runs; it collects nothing`)
215
+ for (const file of undiscovered) console.log(`discovery: ${file} is collected by no project`)
216
+ for (const entry of census) {
217
+ const marks = Object.entries(entry.markers)
218
+ if (marks.length > 0) {
219
+ console.log(
220
+ `discovery: ${entry.file} ${marks.map(([marker, count]) => `${marker}×${count}`).join(' ')}`,
221
+ )
222
+ }
223
+ }
224
+ }
225
+ return flagged ? 3 : 0
226
+ }
227
+
228
+ process.exitCode = main(process.argv.slice(2))
@@ -1,28 +1,20 @@
1
1
  ---
2
- name: orkestrel-prove-journey
2
+ name: orkestrel-journey
3
3
  description: Prove a browser application the way a person uses it — real keystrokes, clicks, and Tab/Enter against only what is visible and reachable — through the journey layer @orkestrel/test/browser publishes, and generate the capture portfolio, the resolved-style matrix, and the statechart outcome from those same journeys. Use when accepting a UI build, proving an application end to end, deciding whether a surface is reachable by keyboard alone, proving what a screen refuses as well as what it does, proving the styles a browser actually resolved under each theme and viewport, driving a transition table through the interface and watching it run, auditing whether the interface speaks the user's vocabulary rather than the engine's, producing the screenshots a design review judges, routing a rendered question to an artifact a model can read, or whenever the only evidence a screen works is a test that drove it through JavaScript instead of through the interface.
4
4
  ---
5
5
 
6
6
  # Prove an application through human journeys
7
7
 
8
- ## Load authority
9
-
10
- Read the current files in this order:
11
-
12
- 1. `AGENTS.md`.
13
- 2. `.claude/rules/tests.md` for test law, real implementations, and shared test infrastructure;
14
- `.claude/rules/browser.md` for browser and Vue usage; `.claude/rules/application.md` for app
15
- composition and entries; `.claude/rules/styles.md` for style centralization;
16
- `.claude/rules/quality.md` for the instrument and negative-control law;
17
- `.claude/rules/documentation.md` for parity. Those rules are the contract; this skill is the
18
- workflow.
19
- 3. [layer.md](references/layer.md) before importing, extending, or debugging the journey layer.
20
- 4. [captures.md](references/captures.md) before registering a state or placing a capture.
21
- 5. [styles.md](references/styles.md) before asserting anything the browser resolved.
22
- 6. [statechart.md](references/statechart.md) before declaring a transition or mounting the harness.
23
- 7. [decide.md](references/decide.md) before routing a question to an instrument.
24
- 8. `guides/README.md`, the governing guide for the surface, and `ROADMAP.md` when present.
25
- 9. The `*/types.ts` of every environment the journeys drive, plus the application's root component,
8
+ ## Read
9
+
10
+ `AGENTS.md` § Authority and loading names the files every unit reads. This skill adds:
11
+
12
+ 1. [layer.md](references/layer.md) before importing, extending, or debugging the journey layer.
13
+ 2. [captures.md](references/captures.md) before registering a state or placing a capture.
14
+ 3. [styles.md](references/styles.md) before asserting anything the browser resolved.
15
+ 4. [statechart.md](references/statechart.md) before declaring a transition or mounting the harness.
16
+ 5. [decide.md](references/decide.md) before routing a question to an instrument.
17
+ 6. The `*/types.ts` of every environment the journeys drive, plus the application's root component,
26
18
  route entry, and store contract.
27
19
 
28
20
  Treat a retained readiness verdict as evidence to re-verify against the current tip, never as a plan
@@ -137,8 +129,8 @@ whose text names `vitest`, so a script naming another runner's configuration rai
137
129
  - Put the provided-context declaration in the browser test setup module, and consume the injected
138
130
  data in `tests/app/browser/integration.test.ts` through the following portfolio configuration.
139
131
  Mount the shipped application entry with its real provisions before the acceptance journey runs.
140
- - From `tests/app/browser/integration.test.ts`, pass `../../../tmp/capture/states` as the capture
141
- directory to write into the workspace's `tmp/capture/states` directory. The browser provider
132
+ - From `tests/app/browser/integration.test.ts`, pass `../../../tmp/captures/states` as the capture
133
+ directory to write into the workspace's `tmp/captures/states` directory. The browser provider
142
134
  resolves a custom screenshot path relative to the test file's directory.
143
135
 
144
136
  ```ts
@@ -165,7 +157,7 @@ const PORTFOLIO = createPortfolio({
165
157
  states: ['home'],
166
158
  variants: VARIANTS,
167
159
  variant: VARIANT,
168
- directory: '../../../tmp/capture/states',
160
+ directory: '../../../tmp/captures/states',
169
161
  enabled: CAPTURE,
170
162
  })
171
163
  ```
@@ -330,7 +322,7 @@ exists only during an activation is captured.
330
322
  Where a capture and a green suite disagree, take the capture as the evidence and the fixture as the
331
323
  defect.
332
324
 
333
- Route review of the portfolio to the `orkestrel-polish-surface` campaign. Do not judge it here.
325
+ Route review of the portfolio to the `orkestrel-polish` campaign. Do not judge it here.
334
326
 
335
327
  ## Route the question
336
328
 
@@ -0,0 +1,4 @@
1
+ interface:
2
+ display_name: 'Prove Human Journeys'
3
+ short_description: 'Prove an application through the interface a person uses'
4
+ default_prompt: 'Use $orkestrel-journey to prove this application through the interface a person uses, and generate the capture portfolio, the resolved-style matrix, and the statechart outcome from those journeys.'
@@ -119,5 +119,5 @@ after the click returns.
119
119
  state the interface never rests in.
120
120
  - Regenerate the whole matrix from the journeys after any surface change. Never judge a round
121
121
  against a portfolio that is part old and part new.
122
- - Route review of the portfolio to the `orkestrel-polish-surface` campaign, which owns preflight,
122
+ - Route review of the portfolio to the `orkestrel-polish` campaign, which owns preflight,
123
123
  verdicts, and reconciliation. This reference owns only how the journeys generate it.