shapeup-sdlc 3.5.0 → 3.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +12 -2
- package/README.md +1 -1
- package/kernel/compile.mjs +18 -6
- package/kernel/harness.mjs +3 -2
- package/kernel/init/run-args.mjs +206 -0
- package/kernel/init/run.mjs +10 -0
- package/kernel/lib/paths.mjs +10 -0
- package/kernel/probe/concurrency.mjs +31 -6
- package/kernel/probe/digest.mjs +15 -1
- package/kernel/probe/owner.mjs +4 -1
- package/kernel/probe/resume.mjs +188 -5
- package/kernel/probe/rounds.mjs +104 -0
- package/kernel/reduce/ingest.mjs +53 -12
- package/kernel/reduce/ship.mjs +15 -30
- package/kernel/reduce/snapshot.mjs +23 -2
- package/kernel/report/export.mjs +54 -2
- package/kernel/report/facts.mjs +24 -2
- package/{skills/tech-lead → kernel}/schemas/domain.schema.json +15 -12
- package/kernel/verify/envelope.mjs +2 -2
- package/kernel/verify/skills.mjs +1 -1
- package/package.json +1 -1
- package/skills/coach/SKILL.md +8 -2
- package/skills/hill-chart/SKILL.md +3 -4
- package/skills/scope-hammer/SKILL.md +1 -1
- package/skills/tech-lead/SKILL.md +10 -10
- package/skills/tech-lead/references/gates.md +48 -11
- package/skills/tech-lead/references/protocol.md +4 -2
- package/skills/tech-lead/workflows/shapeup-run.js +133 -37
- package/skills/translator/SKILL.md +1 -1
- /package/{skills/tech-lead → kernel}/schemas/gate-answers.schema.json +0 -0
- /package/{skills/tech-lead → kernel}/schemas/work-order.schema.json +0 -0
- /package/{skills/tech-lead → kernel}/schemas/work-result.schema.json +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "shapeup-sdlc-plugin",
|
|
3
3
|
"displayName": "ShapeUp SDLC Plugin",
|
|
4
|
-
"version": "3.
|
|
4
|
+
"version": "3.6.0",
|
|
5
5
|
"description": "Shape Up SDLC harness for Claude Code: shaping, intake, orient, scope-mapping, building (T0-verified, sandboxed, scope-contracted), evaluation and QA skills orchestrated by a tech-lead.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Liberty Nguyen",
|
package/AGENTS.md
CHANGED
|
@@ -65,11 +65,21 @@ Everything discovered funnels into `.shapeup/<slug>/discovery/ledger.md` (Orient
|
|
|
65
65
|
|
|
66
66
|
## Setup & Execution
|
|
67
67
|
|
|
68
|
-
- Orders/results live in `.shapeup/<slug>/orders|results/`; the envelope schemas ship
|
|
68
|
+
- Orders/results live in `.shapeup/<slug>/orders|results/`; the envelope schemas ship with the plugin runtime, not with any individual skill, so every worker validates against the same copy.
|
|
69
69
|
- The plugin's run entry points need a one-time permission grant — `npx shapeup-sdlc init` writes it into `.claude/settings.json` (`permissions.allow`); without it a headless run stalls at step one. That grant is necessary, not sufficient: it covers the run's own deterministic entry points, not the generic file edits every worker skill makes constantly, or any command a worker reaches for beyond the grant's own exact shape. A truly unattended run also needs a Claude Code permission mode that covers those (`acceptEdits` at minimum) — the plugin cannot grant that on your behalf.
|
|
70
|
+
- The grant is necessary but sits under two more layers this plugin cannot reach either. A fresh
|
|
71
|
+
checkout is an **untrusted workspace**, and Claude Code discards the whole permission grant — every
|
|
72
|
+
rule in it, not only this one — until the workspace is trusted; the installer detects that state and
|
|
73
|
+
tells you, because trusting a directory to run code from is your decision to make, never a package's
|
|
74
|
+
to make for you. Above workspace trust sits Claude Code's own **auto-mode classifier**, which can
|
|
75
|
+
still block a call the grant already covers, invisibly to anything this plugin ships — no rule or
|
|
76
|
+
hook here can see it, let alone override it. When either layer stops an automated run, the
|
|
77
|
+
documented fallback is to drive Build by hand: the same per-dispatch cycle named in the Build
|
|
78
|
+
Vertically step above — compile, dispatch, ingest, verify — run one call at a time instead of
|
|
79
|
+
through the chained launch, until the run can resume unattended again.
|
|
70
80
|
- Two storage tiers (ADR-0001): COMMITTED `shapeup/<slug>/` (shaping, spec, scopes, wiring-map, project-profile, requirements, hill, `REPORT.md` frozen at L4) vs GITIGNORED `.shapeup/` (board, orders/results, T0/eval/QA artifacts, ledgers, metrics, gate answers, and the run scripts staged for launch).
|
|
71
81
|
- **The run launches from a copy inside your project, and it has to.** The Workflow tool loads a script only from a directory the session may already read; the plugin installs outside your project, so naming the shipped path is refused before the run begins and no permission rule repairs it — the grant authorises the tool, not what it may read. Opening a run therefore re-copies the run scripts to `.shapeup/workflows/` and reports the path the launch names. A run in flight keeps the copy it started with: an upgrade reaches the next run, not the current round.
|
|
72
|
-
- **A write the substrate does not cover is denied for as long as the dispatch is in flight, and no longer.** A dispatch is live from the moment its order is compiled until a result for *that* dispatch lands, and liveness is read off the run's order set — not off the pointer that names the run, which outlives it.
|
|
82
|
+
- **A write the substrate does not cover is denied for as long as the dispatch is in flight, and no longer.** A dispatch is live from the moment its order is compiled until a result for *that* dispatch lands, and liveness is read off the run's order set — not off the pointer that names the run, which outlives it. A run closed any other way than shipping — escalated, aborted, killed outright — can leave one still open, and while the run's pointer is still on disk the fence holds on that order exactly as if the run were live. The pointer is what actually switches it off, though: the fence is enforced only while that pointer exists on disk, so a close that removes it releases the fence whatever the order set still says — restoring the pointer flips it straight back to denying. A **ship** close does not itself answer what it leaves outstanding — a ship close can retire the pointer over an order that never got a result — so it is not special because nothing is left unanswered; it is special only because retiring the pointer is the one lever every close needs pulled, and `reduce ship` pulls it for you. Whichever way a run closes, an order it leaves unanswered stays genuinely unresolved — not merely un-fenced — until `init run --force` runs: it writes a synthetic result for every order the closed run left unanswered, so the fence lifts without waiting on a worker that is never coming back. What the fence actually gates is narrower than "the project," too — it is this assistant's own edit path (a direct file write or edit call) outside the live order's substrate; a shell command, `git`, or any other editor still writes straight through it. Re-dispatch is fenced again, and the committed tier is never a worker's to write either way: those files belong to the orchestrator, whose window is a phase boundary rather than the middle of somebody else's order.
|
|
73
83
|
- Every run has a `run_id` — the receipt mints it, and orders, T0 artifacts, trial rows, agent-call journal rows and hook decisions all carry it. It is the only key that separates two runs of the same feature: everything else (`order_id`, round/attempt) repeats. It is **not** a time boundary — a relaunch resumes the same run and reuses the key, so one `run_id` legitimately spans every launch after a paused gate or a kill, with hours of wall clock between them, and `orders/<id>.json` is rewritten by each. Anything measuring elapsed time reads the append-only records, never the span of a key. SHIP S.7 exports the run's records as fact tables under `.shapeup/exports/<run_id>/` before the run trace is superseded; a WorkResult carries no `run_id` and reaches it through `order_id`.
|
|
74
84
|
- Every run projects a **run graph** — `.shapeup/<slug>/graph.jsonl`, append-only, written only by
|
|
75
85
|
`reduce graph`. Two families kept separate: work lineage (Run, Order, Result, Verdict, Trial,
|
package/README.md
CHANGED
|
@@ -332,11 +332,11 @@ claude --plugin-dir . # load this working copy without installing
|
|
|
332
332
|
plugin.json # plugin manifest
|
|
333
333
|
marketplace.json # marketplace listing (points at this repo)
|
|
334
334
|
skills/<name>/SKILL.md # the 13 harness skills (+ references/ and assets/)
|
|
335
|
-
skills/tech-lead/schemas/ # the envelope port: WorkOrder, WorkResult, domain registry
|
|
336
335
|
skills/tech-lead/workflows/shapeup-run.js # the BUILD-phase pipeline, on the native Workflow runtime
|
|
337
336
|
kernel/harness.mjs # ONE entry point for every deterministic step; the whole permission grant
|
|
338
337
|
kernel/{verify,reduce,probe,init,report}/ # its subcommands, plus compile and gate at the root
|
|
339
338
|
kernel/lib/ # argv (the typed CLI boundary), paths (+ the run key), contract (shape)
|
|
339
|
+
kernel/schemas/ # the envelope port: WorkOrder, WorkResult, domain registry
|
|
340
340
|
commands/*.md # slash commands (/ship + the 9 phase commands)
|
|
341
341
|
hooks/ # hooks.json + the four walls: safety-spine, gate-intake, sandbox-guard
|
|
342
342
|
# (PreToolUse) + gate-zerowork (Stop, the one blocking hook)
|
package/kernel/compile.mjs
CHANGED
|
@@ -58,7 +58,7 @@ export const COACHABLE = new Set([
|
|
|
58
58
|
]);
|
|
59
59
|
|
|
60
60
|
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
61
|
-
const ORDER_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "
|
|
61
|
+
const ORDER_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "./schemas/work-order.schema.json"), "utf8"));
|
|
62
62
|
|
|
63
63
|
// --- tiny frontmatter reader (scalar keys + [a, b] inline lists) --------------------------
|
|
64
64
|
/**
|
|
@@ -218,9 +218,15 @@ export function substrateFor(operation, { slug, specDir, scope } = {}) {
|
|
|
218
218
|
const FROZEN_INTAKE = [`${local}/intake.md`, `${local}/breadboard.md`];
|
|
219
219
|
switch (operation) {
|
|
220
220
|
case "execute": case "fix": case "spike":
|
|
221
|
+
// Build legs are the widest window on FROZEN_INTAKE, not an exemption from it: they are the
|
|
222
|
+
// most numerous and longest-lived dispatches in a run, so a doer that can rewrite the staged
|
|
223
|
+
// pitch can rewrite the run's own input truth mid-build. `init run` stages these before any
|
|
224
|
+
// order is live (no live contract yet — nothing to violate) and `translate` writes the
|
|
225
|
+
// COMMITTED copy, not this one, so neither legitimate write is touched by this line.
|
|
221
226
|
return {
|
|
222
227
|
allowed: [...(scope?.allowed_file_substrate || []), `${local}/spikes/**`],
|
|
223
228
|
shared: scope?.shared_substrate || [],
|
|
229
|
+
frozen: [...FROZEN_INTAKE],
|
|
224
230
|
};
|
|
225
231
|
case "analyze":
|
|
226
232
|
return { allowed: [`${spec}/**`, `${local}/**`], frozen: [...FROZEN_INTAKE] };
|
|
@@ -518,24 +524,30 @@ export function bugLocations(bug) {
|
|
|
518
524
|
*
|
|
519
525
|
* OWNERSHIP IS BY SUBSTRATE, because that is what the sandbox enforces: a scope is exactly the set
|
|
520
526
|
* of files its worker may write, so a scope whose substrate excludes the cited line cannot fix it
|
|
521
|
-
* however well it understands the bug.
|
|
527
|
+
* however well it understands the bug. The substrate a scope may write is `allowed ∪ shared` — the
|
|
528
|
+
* same union `sandbox-guard` composes at the fence — so a path declared ONLY in a contract's
|
|
529
|
+
* `shared` list is a file that scope may legitimately write, and the election must see it too —
|
|
530
|
+
* filtering on `allowed` alone elects no one for a shared-only path, and `bugsForScope` then reads
|
|
531
|
+
* that null as "no scope owns this" and fans the bug out to every scope instead of the one or two
|
|
532
|
+
* that declared it.
|
|
522
533
|
*
|
|
523
534
|
* BUT A MATCH IS NOT AN ELECTION. An entry point is routinely SHARED — on the measured run
|
|
524
535
|
* `bin/todo.js` sits in five scopes' substrate at once — so "address it to every scope that
|
|
525
536
|
* matches" hands the same one-line fix to five workers building concurrently against one file.
|
|
526
537
|
* That is a write race the harness sets up itself, and four of the five fixes are waste even when
|
|
527
538
|
* it resolves. So: prefer a scope that owns the file EXCLUSIVELY (allowed, not shared), and among
|
|
528
|
-
* equals
|
|
529
|
-
*
|
|
539
|
+
* equals — every remaining candidate declares it shared, exclusive or not — take the lowest scope
|
|
540
|
+
* id: a rule that needs no coordination to agree with itself, since each leg compiles its own
|
|
541
|
+
* order in its own process.
|
|
530
542
|
*
|
|
531
543
|
* @param {string} path - Repo-relative file the bug cites.
|
|
532
544
|
* @param {Array<{scope_id:string, allowed:string[], shared:string[]}>} scopes - Every scope.
|
|
533
545
|
* @returns {string|null} The elected scope id, or null when no scope may write that file.
|
|
534
546
|
*/
|
|
535
547
|
export function electOwner(path, scopes) {
|
|
536
|
-
const can = (scopes || []).filter((s) => matchesAny(path, s.allowed));
|
|
548
|
+
const can = (scopes || []).filter((s) => matchesAny(path, s.allowed) || matchesAny(path, s.shared || []));
|
|
537
549
|
if (!can.length) return null;
|
|
538
|
-
const exclusive = can.filter((s) => !matchesAny(path, s.shared || []));
|
|
550
|
+
const exclusive = can.filter((s) => matchesAny(path, s.allowed) && !matchesAny(path, s.shared || []));
|
|
539
551
|
return (exclusive.length ? exclusive : can).map((s) => s.scope_id).sort()[0];
|
|
540
552
|
}
|
|
541
553
|
|
package/kernel/harness.mjs
CHANGED
|
@@ -50,7 +50,8 @@
|
|
|
50
50
|
// answers which pitch clause a verdict reached, joined
|
|
51
51
|
// through the plan's own covers: edge — the L4 line and
|
|
52
52
|
// GATE H's census cite it for the same reason.
|
|
53
|
-
// init run · fit
|
|
53
|
+
// init run · fit · run-args Opens a run, or refuses it (exit 3). `run-args`
|
|
54
|
+
// writes GATE L0.9b's launch record and echoes it.
|
|
54
55
|
// report export Projects the run's records as fact tables.
|
|
55
56
|
// compile The WorkOrder: schema-valid or nothing is dispatched.
|
|
56
57
|
//
|
|
@@ -87,7 +88,7 @@ export const ROUTES = {
|
|
|
87
88
|
leg: "./probe/leg.mjs", eval: "./probe/eval.mjs", owner: "./probe/owner.mjs",
|
|
88
89
|
requirements: "./probe/requirements.mjs",
|
|
89
90
|
},
|
|
90
|
-
init: { run: "./init/run.mjs", fit: "./init/fit.mjs" },
|
|
91
|
+
init: { run: "./init/run.mjs", fit: "./init/fit.mjs", "run-args": "./init/run-args.mjs" },
|
|
91
92
|
report: { export: "./report/export.mjs", _default: "export" },
|
|
92
93
|
gate: "./gate.mjs",
|
|
93
94
|
compile: "./compile.mjs",
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// GATE L0.9b — THE LAUNCH RECORD, GIVEN A WRITER.
|
|
3
|
+
//
|
|
4
|
+
// WHY THIS EXISTS. `domain.schema.json`'s `RunArgs` entry calls `.shapeup/<slug>/run-args.json`
|
|
5
|
+
// "the only artifact that records what a run was configured with" and names tech-lead as the
|
|
6
|
+
// writer — but "tech-lead" meant a paragraph of prose telling the orchestrating session to
|
|
7
|
+
// assemble a JSON object by hand and `Write` it. Outside a structural-test fixture, nothing ever
|
|
8
|
+
// did: no kernel module wrote the file, so `probe concurrency`'s `dialFrom()` read a fan-out dial
|
|
9
|
+
// that was never recorded and reported the effective default every time, indistinguishable from a
|
|
10
|
+
// run that genuinely chose it. A registry entry is not an instruction, and an instruction with no
|
|
11
|
+
// enforcer is how a documented record becomes a file that simply never exists.
|
|
12
|
+
//
|
|
13
|
+
// THE FIX. This is the single writer. It takes the resolved GATE L0 values as flags, builds the
|
|
14
|
+
// exact `RunArgs` object the schema describes, writes it to `.shapeup/<slug>/run-args.json`, and
|
|
15
|
+
// prints that SAME object on stdout — so tech-lead passes the printed value straight to
|
|
16
|
+
// `Workflow({args: ...})` rather than re-typing it a second time. One construction, not two: the
|
|
17
|
+
// file on disk and the value the workflow actually launches with can no longer disagree.
|
|
18
|
+
//
|
|
19
|
+
// `wallClockS` is NOT one of this command's flags and never lands in the object it writes.
|
|
20
|
+
// `--wall-clock-budget` is consumed earlier, by `harness init run` (see `RECEIPT_VERSION` and
|
|
21
|
+
// `config.wall_clock_budget_s` in `./run.mjs`) — the deadline breaker reads that receipt field
|
|
22
|
+
// directly and was never going to see this launch's `RunArgs` at all. Naming a third field here
|
|
23
|
+
// that nothing reads is the exact defect this command exists to close, not one to reintroduce.
|
|
24
|
+
//
|
|
25
|
+
// USAGE
|
|
26
|
+
// node `harness init run-args` --slug <slug> --auto-level interactive|auto|unattended \
|
|
27
|
+
// --exec-model <name> [--eval-model <name>] [--qa-model <name>] \
|
|
28
|
+
// --max-rounds N --attempts N --plugin-root <dir> \
|
|
29
|
+
// [--run-id <id>] [--answers <preset|path>] [--lane full|tiny] \
|
|
30
|
+
// [--no-eval] [--no-qa] [--adversarial-verify] [--parallel-scopes N] [--cwd <dir>]
|
|
31
|
+
//
|
|
32
|
+
// `--eval-model` is required unless `--no-eval` is set — an EVAL-skipping run never resolves one.
|
|
33
|
+
// `--run-id`, given no explicit value, is read off the run's own receipt (the run must already be
|
|
34
|
+
// open — see `harness init run`), never invented.
|
|
35
|
+
//
|
|
36
|
+
// Prints the written `RunArgs` object as JSON on stdout. Exit 0 on success, 2 on a usage error,
|
|
37
|
+
// 3 when no run is open at `--slug` (the receipt is what makes this artifact meaningful at all).
|
|
38
|
+
|
|
39
|
+
import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
|
|
40
|
+
import { dirname, join, resolve } from "node:path";
|
|
41
|
+
import { fileURLToPath } from "node:url";
|
|
42
|
+
import { runArgs } from "../lib/argv.mjs";
|
|
43
|
+
import { localRoot, RECEIPT_FILE, runArgsPath, runIdFromReceipt } from "../lib/paths.mjs";
|
|
44
|
+
|
|
45
|
+
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
46
|
+
/** The plugin root, resolved from where this file actually is (`kernel/init/` → repo root). */
|
|
47
|
+
export const PLUGIN_ROOT = resolve(HERE, "../..");
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* The `autoLevel` enum, read from the schema that defines it — never hand-typed here.
|
|
51
|
+
*
|
|
52
|
+
* `domain.schema.json`'s `$defs/RunArgs.properties.autoLevel.enum` is the one place this set is
|
|
53
|
+
* declared. A `new Set([...])` beside it, inside the very module written to close run-argument
|
|
54
|
+
* contract drift (see this file's own banner), would be that drift repeating one level down —
|
|
55
|
+
* the same class of failure this module exists to close, applied here to an *enum's values*
|
|
56
|
+
* instead of RunArgs *field* names. Mirrors `kernel/verify/skills.mjs`'s `roster()`, which
|
|
57
|
+
* derives `WorkerName` the same way.
|
|
58
|
+
*
|
|
59
|
+
* @param {string} [root=PLUGIN_ROOT] - Plugin root the schema is read from — overridable so a test
|
|
60
|
+
* can point this at a scratch copy of the schema and prove the result grows and shrinks with it,
|
|
61
|
+
* never with an edit to this function.
|
|
62
|
+
* @returns {string[]} The enum, in schema order.
|
|
63
|
+
* @throws {Error} When the schema is missing or does not carry the enum.
|
|
64
|
+
*/
|
|
65
|
+
export function autoLevels(root = PLUGIN_ROOT) {
|
|
66
|
+
const schemaPath = join(root, "kernel/schemas/domain.schema.json");
|
|
67
|
+
const schema = JSON.parse(readFileSync(schemaPath, "utf8"));
|
|
68
|
+
const levels = schema?.$defs?.RunArgs?.properties?.autoLevel?.enum;
|
|
69
|
+
if (!Array.isArray(levels) || !levels.length) {
|
|
70
|
+
throw new Error(`${schemaPath} carries no $defs/RunArgs.properties.autoLevel.enum — the auto-level set cannot be derived`);
|
|
71
|
+
}
|
|
72
|
+
return levels;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** The typed argv contract (see `../lib/argv.mjs`). */
|
|
76
|
+
export const ARGV_SPEC = {
|
|
77
|
+
usage: `harness.mjs init run-args --slug <slug> --auto-level ${autoLevels().join("|")} ` +
|
|
78
|
+
'--exec-model <name> [--eval-model <name>] [--qa-model <name>] --max-rounds N --attempts N ' +
|
|
79
|
+
'--plugin-root <dir> [--run-id <id>] [--answers <preset|path>] [--lane full|tiny] ' +
|
|
80
|
+
'[--no-eval] [--no-qa] [--adversarial-verify] [--parallel-scopes N] [--cwd <dir>]',
|
|
81
|
+
_: { arity: 0, max: 0, name: "(no positional operands)" },
|
|
82
|
+
cwd: { type: "path" },
|
|
83
|
+
slug: { type: "str", required: true },
|
|
84
|
+
"run-id": { type: "str" },
|
|
85
|
+
"auto-level": { type: "str", required: true },
|
|
86
|
+
answers: { type: "str" },
|
|
87
|
+
lane: { type: "str" },
|
|
88
|
+
"exec-model": { type: "str", required: true },
|
|
89
|
+
"eval-model": { type: "str" },
|
|
90
|
+
"qa-model": { type: "str" },
|
|
91
|
+
"max-rounds": { type: "int", min: 1, required: true },
|
|
92
|
+
attempts: { type: "int", min: 1, required: true },
|
|
93
|
+
"plugin-root": { type: "path", required: true },
|
|
94
|
+
"no-eval": { type: "flag" },
|
|
95
|
+
"no-qa": { type: "flag" },
|
|
96
|
+
"adversarial-verify": { type: "flag" },
|
|
97
|
+
"parallel-scopes": { type: "int", min: 1 },
|
|
98
|
+
};
|
|
99
|
+
|
|
100
|
+
function fail(code, msg) {
|
|
101
|
+
console.error(msg);
|
|
102
|
+
process.exit(code);
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Drop `undefined`-valued keys, one level deep on the named nested objects — so the JSON this
|
|
107
|
+
* writes never carries a literal `"eval": undefined` for a skipped role or an unset switch.
|
|
108
|
+
*
|
|
109
|
+
* @param {object} o - The candidate RunArgs object.
|
|
110
|
+
* @returns {object} The same shape with every `undefined` leaf and empty nested object removed.
|
|
111
|
+
*/
|
|
112
|
+
function pruned(o) {
|
|
113
|
+
const out = {};
|
|
114
|
+
for (const [k, v] of Object.entries(o)) {
|
|
115
|
+
if (v === undefined) continue;
|
|
116
|
+
if (v && typeof v === "object" && !Array.isArray(v)) {
|
|
117
|
+
const inner = pruned(v);
|
|
118
|
+
if (Object.keys(inner).length) out[k] = inner;
|
|
119
|
+
continue;
|
|
120
|
+
}
|
|
121
|
+
out[k] = v;
|
|
122
|
+
}
|
|
123
|
+
return out;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Build the `RunArgs` object — pure, so the structural suite can assert its shape without a
|
|
128
|
+
* filesystem. Mirrors `domain.schema.json` `$defs/RunArgs` exactly: this is the one place that
|
|
129
|
+
* shape is constructed, so a field this function does not carry cannot reach the file either.
|
|
130
|
+
*
|
|
131
|
+
* @param {object} o - Resolved inputs (destructured); optional fields may be `undefined`.
|
|
132
|
+
* @returns {object} The RunArgs object, pruned of unset optionals.
|
|
133
|
+
*/
|
|
134
|
+
export function buildRunArgs({
|
|
135
|
+
slug, runId, autoLevel, answers, lane, execModel, evalModel, qaModel,
|
|
136
|
+
maxRounds, attemptBudget, pluginRoot, startedAt,
|
|
137
|
+
noEval, noQa, adversarialVerify, maxParallelScopes,
|
|
138
|
+
}) {
|
|
139
|
+
return pruned({
|
|
140
|
+
slug, runId, autoLevel, answers, lane,
|
|
141
|
+
models: { exec: execModel, eval: evalModel, qa: qaModel },
|
|
142
|
+
budgets: { maxRounds, attemptBudget },
|
|
143
|
+
pluginRoot, startedAt,
|
|
144
|
+
noEval, noQa, adversarialVerify, maxParallelScopes,
|
|
145
|
+
});
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* Write `.shapeup/<slug>/run-args.json`, or refuse (exit 3) when no run is open at `--slug`.
|
|
150
|
+
*
|
|
151
|
+
* @param {string[]} rawArgv - The subcommand's own arguments (harness.mjs strips the verb words).
|
|
152
|
+
* @returns {(Promise<void>|void)} Settles when the subcommand has written its output; every path
|
|
153
|
+
* calls `process.exit()` with the subcommand's documented code rather than returning.
|
|
154
|
+
*/
|
|
155
|
+
export function cli(rawArgv) {
|
|
156
|
+
const args = runArgs(ARGV_SPEC, rawArgv);
|
|
157
|
+
const cwd = args.cwd || process.cwd();
|
|
158
|
+
|
|
159
|
+
if (!autoLevels().includes(args.autoLevel)) {
|
|
160
|
+
fail(2, `--auto-level must be one of: ${autoLevels().join(", ")}`);
|
|
161
|
+
}
|
|
162
|
+
if (!args.noEval && !args.evalModel) {
|
|
163
|
+
fail(2, "--eval-model is required unless --no-eval is set — an EVAL-skipping run never resolves one.");
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
const runRoot = localRoot(cwd, args.slug);
|
|
167
|
+
const receiptPath = join(runRoot, RECEIPT_FILE);
|
|
168
|
+
if (!existsSync(receiptPath)) {
|
|
169
|
+
fail(3, [
|
|
170
|
+
`✋ init run-args: no open run at ${runRoot} — run "harness init run" first (GATE L0.1).`,
|
|
171
|
+
"",
|
|
172
|
+
"`run-args.json` records what an OPEN run was launched with; writing one with no receipt",
|
|
173
|
+
"behind it would leave a launch record for a run that, everywhere else in the harness, never",
|
|
174
|
+
"started.",
|
|
175
|
+
].join("\n"));
|
|
176
|
+
}
|
|
177
|
+
let receipt = null;
|
|
178
|
+
try { receipt = JSON.parse(readFileSync(receiptPath, "utf8")); } catch { /* handled below */ }
|
|
179
|
+
const runId = args.runId ?? runIdFromReceipt(receipt) ?? undefined;
|
|
180
|
+
|
|
181
|
+
const runArgsObj = buildRunArgs({
|
|
182
|
+
slug: args.slug,
|
|
183
|
+
runId,
|
|
184
|
+
autoLevel: args.autoLevel,
|
|
185
|
+
answers: args.answers ?? undefined,
|
|
186
|
+
lane: args.lane ?? undefined,
|
|
187
|
+
execModel: args.execModel,
|
|
188
|
+
evalModel: args.evalModel ?? undefined,
|
|
189
|
+
qaModel: args.qaModel ?? undefined,
|
|
190
|
+
maxRounds: args.maxRounds,
|
|
191
|
+
attemptBudget: args.attempts,
|
|
192
|
+
pluginRoot: args.pluginRoot,
|
|
193
|
+
startedAt: new Date().toISOString(),
|
|
194
|
+
noEval: args.noEval || undefined,
|
|
195
|
+
noQa: args.noQa || undefined,
|
|
196
|
+
adversarialVerify: args.adversarialVerify || undefined,
|
|
197
|
+
maxParallelScopes: args.parallelScopes ?? undefined,
|
|
198
|
+
});
|
|
199
|
+
|
|
200
|
+
mkdirSync(runRoot, { recursive: true });
|
|
201
|
+
writeFileSync(runArgsPath(cwd, args.slug), JSON.stringify(runArgsObj, null, 2) + "\n", "utf8");
|
|
202
|
+
|
|
203
|
+
// Printed, not just written — so the caller's `Workflow({args: ...})` reads this stdout instead
|
|
204
|
+
// of re-assembling the object from the flags it just typed.
|
|
205
|
+
console.log(JSON.stringify(runArgsObj, null, 2));
|
|
206
|
+
}
|
package/kernel/init/run.mjs
CHANGED
|
@@ -224,6 +224,16 @@ export function runFrontmatter({ slug, config, startedAt }) {
|
|
|
224
224
|
"deploy: ~",
|
|
225
225
|
`started_at: ${startedAt}`,
|
|
226
226
|
"closed_at: ~",
|
|
227
|
+
// The cause a terminal status ended on, written alongside `closed_at` by
|
|
228
|
+
// `probe resume --close` (kernel/probe/resume.mjs's closeRun) — the two land in one write, so a
|
|
229
|
+
// closed run's ledger never carries a timestamp with no reason beside it.
|
|
230
|
+
"close_cause: ~",
|
|
231
|
+
// The once-only guard's OWN record of what closed this run — deliberately separate from
|
|
232
|
+
// `status:` above, which every phase rewrites (`setRunStatus`) for the life of the run,
|
|
233
|
+
// including the product's own ship path immediately before `probe resume --close` runs. Only
|
|
234
|
+
// `closeRun` ever writes this line, so it is the one field an intervening `status:` rewrite
|
|
235
|
+
// cannot move (kernel/probe/resume.mjs's closeRun docblock has the measured scenario).
|
|
236
|
+
"closed_status: ~",
|
|
227
237
|
"---",
|
|
228
238
|
"",
|
|
229
239
|
`# Harness run — ${slug}`,
|
package/kernel/lib/paths.mjs
CHANGED
|
@@ -140,6 +140,16 @@ export const receipt = (cwd, slug) => join(localRoot(cwd, slug), RECEIPT_FILE);
|
|
|
140
140
|
export const intake = (cwd, slug) => join(localRoot(cwd, slug), "intake.md");
|
|
141
141
|
/** The breadboard the pitch was shaped with, verbatim, next to its digest in the receipt. */
|
|
142
142
|
export const breadboard = (cwd, slug) => join(localRoot(cwd, slug), "breadboard.md");
|
|
143
|
+
/**
|
|
144
|
+
* The launch record's filename, as a constant rather than a literal at each call site — the
|
|
145
|
+
* writer (`harness init run-args`) and every reader (`probe concurrency`'s `dialFrom()`) resolve
|
|
146
|
+
* the same path through this constant, the same way {@link runIdFromRoot} resolves `RECEIPT_FILE`
|
|
147
|
+
* against a bare run root rather than through {@link receipt}'s `(cwd, slug)` form: a caller that
|
|
148
|
+
* already holds the run root (an archived trace, `--run-root <dir>`) has no slug to reconstruct.
|
|
149
|
+
*/
|
|
150
|
+
export const RUN_ARGS_FILE = "run-args.json";
|
|
151
|
+
/** The launch record — the RunArgs object a run was launched with (GATE L0.9b), kernel-written. */
|
|
152
|
+
export const runArgsPath = (cwd, slug) => join(localRoot(cwd, slug), RUN_ARGS_FILE);
|
|
143
153
|
/** The run ledger — rounds, decisions, status frontmatter. */
|
|
144
154
|
export const harnessRun = (cwd, slug) => join(localRoot(cwd, slug), "harness-run.md");
|
|
145
155
|
/** File-derived mid-run digest, frozen by `reduce snapshot --write` as an audit anchor. */
|
|
@@ -30,13 +30,16 @@
|
|
|
30
30
|
//
|
|
31
31
|
// Usage: node kernel/harness.mjs probe concurrency --slug <slug> [--cwd <dir>] [--run-root <dir>]
|
|
32
32
|
// [--round N] [--gap-s N] [--format json|table]
|
|
33
|
+
// [--require-run-args]
|
|
33
34
|
// Exit: 0 = a report was produced with at least one usable leg · 1 = ran, and no leg in scope had
|
|
34
35
|
// a usable interval (the report still prints, and says why) · 2 = malformed argv.
|
|
36
|
+
// With --require-run-args the report is skipped: 0 = run-args.json exists at this run root ·
|
|
37
|
+
// 6 = it does not (mirrors `probe resume --require`'s "artifact absent" convention).
|
|
35
38
|
|
|
36
39
|
import { existsSync, readFileSync } from "node:fs";
|
|
37
40
|
import { join, resolve } from "node:path";
|
|
38
41
|
import { runArgs } from "../lib/argv.mjs";
|
|
39
|
-
import { localRoot, runIdFromRoot, RECEIPT_FILE } from "../lib/paths.mjs";
|
|
42
|
+
import { localRoot, runIdFromRoot, RECEIPT_FILE, RUN_ARGS_FILE } from "../lib/paths.mjs";
|
|
40
43
|
|
|
41
44
|
/** The round-addressed order id forms `<scope>-r<N>-a<M>` and `<phase>-r<N>`. */
|
|
42
45
|
const ROUND_SUFFIX = /^(.*?)-r(\d+)(?:-a(\d+))?$/;
|
|
@@ -301,16 +304,18 @@ export function summarise(index, legs) {
|
|
|
301
304
|
/**
|
|
302
305
|
* Which fan-out width the run was launched with — read, never assumed.
|
|
303
306
|
*
|
|
304
|
-
* The dial is written into `run-args.json` by
|
|
305
|
-
*
|
|
306
|
-
*
|
|
307
|
+
* The dial is written into `run-args.json` by `harness init run-args` (GATE L0.9b), the kernel
|
|
308
|
+
* writer tech-lead invokes right before the launch. A run opened before that writer existed, or one
|
|
309
|
+
* whose launcher never passed `--parallel-scopes`, still has no file or no key — so the honest
|
|
310
|
+
* answer stays the effective default WITH the fact that it is a default: reporting `4` unqualified
|
|
311
|
+
* would assert an operator choice nobody made.
|
|
307
312
|
*
|
|
308
313
|
* @param {string} runRoot - The run's LOCAL root.
|
|
309
314
|
* @returns {{max_parallel_scopes:number, source:string}} The value and where it came from.
|
|
310
315
|
*/
|
|
311
316
|
export function dialFrom(runRoot) {
|
|
312
317
|
try {
|
|
313
|
-
const a = JSON.parse(readFileSync(join(runRoot,
|
|
318
|
+
const a = JSON.parse(readFileSync(join(runRoot, RUN_ARGS_FILE), "utf8"));
|
|
314
319
|
const n = Number(a?.maxParallelScopes);
|
|
315
320
|
if (Number.isFinite(n) && n >= 1) return { max_parallel_scopes: n, source: "run-args" };
|
|
316
321
|
return { max_parallel_scopes: DEFAULT_MAX_PARALLEL_SCOPES, source: "default (run-args.json declares none)" };
|
|
@@ -474,7 +479,7 @@ export function table(r) {
|
|
|
474
479
|
/** The typed argv contract (see `./lib/argv.mjs`). */
|
|
475
480
|
export const ARGV_SPEC = {
|
|
476
481
|
usage: "harness.mjs probe concurrency (--slug <slug> | --run-root <dir>) [--cwd <dir>] [--round N] " +
|
|
477
|
-
"[--gap-s N] [--format json|table]",
|
|
482
|
+
"[--gap-s N] [--format json|table] [--require-run-args]",
|
|
478
483
|
_: { arity: 0, max: 0, name: "(no positional operands)" },
|
|
479
484
|
slug: { type: "str" },
|
|
480
485
|
// The archived-trace and post-export cases, and the same escape `verify t0 --out` has: a caller
|
|
@@ -484,6 +489,10 @@ export const ARGV_SPEC = {
|
|
|
484
489
|
round: { type: "int", min: 1 },
|
|
485
490
|
"gap-s": { type: "int", min: 1, default: DEFAULT_GAP_S },
|
|
486
491
|
format: { type: "enum", values: ["json", "table"], default: "json" },
|
|
492
|
+
// The positive enforcer for the launch record (see the banner below). Skips the concurrency
|
|
493
|
+
// report entirely — this is a presence check, not a measurement, and the two must not be
|
|
494
|
+
// confused by sharing an exit code.
|
|
495
|
+
"require-run-args": { type: "flag" },
|
|
487
496
|
};
|
|
488
497
|
|
|
489
498
|
/**
|
|
@@ -492,6 +501,14 @@ export const ARGV_SPEC = {
|
|
|
492
501
|
* @param {string[]} rawArgv - The subcommand's own arguments (harness.mjs strips the verb words).
|
|
493
502
|
* @returns {void} Exits 0 when at least one leg had a usable interval, 1 when none did — and the
|
|
494
503
|
* report prints either way, because "nothing was measurable" is the answer, not an error.
|
|
504
|
+
*
|
|
505
|
+
* `--require-run-args` is a different question with its own exit convention, mirroring
|
|
506
|
+
* `probe resume --require`'s 0 (satisfied) / 6 (artifact absent): it never reaches the report at
|
|
507
|
+
* all. The launch record had exactly one reader (`dialFrom()`, below) and zero enforcers — a
|
|
508
|
+
* run missing it proceeded green with a silently substituted default, "indistinguishable from an
|
|
509
|
+
* operator choice" (`gates.md` L0.9b). This flag is what `shapeup-run.js` calls at Preflight, before
|
|
510
|
+
* ORIENT, so that state stops being invisible: this module already owns run-root resolution and the
|
|
511
|
+
* file's path, so the check is a few lines here rather than a new kernel entry point.
|
|
495
512
|
*/
|
|
496
513
|
export function cli(rawArgv) {
|
|
497
514
|
const args = runArgs(ARGV_SPEC, rawArgv);
|
|
@@ -504,6 +521,14 @@ export function cli(rawArgv) {
|
|
|
504
521
|
? resolve(args.runRoot)
|
|
505
522
|
: localRoot(resolve(args.cwd || process.cwd()), args.slug);
|
|
506
523
|
|
|
524
|
+
if (args.requireRunArgs) {
|
|
525
|
+
const path = join(runRoot, RUN_ARGS_FILE);
|
|
526
|
+
const present = existsSync(path);
|
|
527
|
+
console.log(JSON.stringify({ ok: present, run_root: runRoot, run_args_path: path,
|
|
528
|
+
reason: present ? null : "no run-args.json at this run root — GATE L0.9b's launch record was never written before this launch" }));
|
|
529
|
+
process.exit(present ? 0 : 6);
|
|
530
|
+
}
|
|
531
|
+
|
|
507
532
|
const r = report(runRoot, { round: args.round ?? null, gapS: args.gapS });
|
|
508
533
|
console.log(args.format === "table" ? table(r) : JSON.stringify(r));
|
|
509
534
|
process.exit(r.launches.length ? 0 : 1);
|
package/kernel/probe/digest.mjs
CHANGED
|
@@ -6,7 +6,9 @@
|
|
|
6
6
|
// Script-first by design: regex over known log formats is free (no model tokens); an
|
|
7
7
|
// unrecognized line becomes a "raw" triple (file/line unknown) rather than being silently
|
|
8
8
|
// dropped, so a Sonnet fallback (or a human) still has something to look at — this module never
|
|
9
|
-
// invents a file:line it didn't find in the text.
|
|
9
|
+
// invents a file:line it didn't find in the text. `file` and `line` are independent: a
|
|
10
|
+
// diagnostic that names a file but no line number (a resource-compiler error, for example)
|
|
11
|
+
// still yields its file — `line` stays null rather than being guessed at.
|
|
10
12
|
//
|
|
11
13
|
// Zero dependencies, zero network — same discipline as oracles/*.
|
|
12
14
|
|
|
@@ -19,6 +21,18 @@ const PATTERNS = [
|
|
|
19
21
|
{ re: /^(?:✗|not ok\b.*?)[^()]*\((.+?):(\d+)\)\s*$/, kind: "test-failure" },
|
|
20
22
|
// ESLint/tsc style: "path/to/file.ts:12:34 - error TS2345: message"
|
|
21
23
|
{ re: /^(.+?):(\d+):\d+\s*[-–]\s*(?:error|warning)\b.*$/, kind: "compiler-diagnostic" },
|
|
24
|
+
// File-level diagnostic with NO line number: "resource.xml: error: message" or
|
|
25
|
+
// "resource.xml - fatal error: message" (resource compilers and linkers report this way —
|
|
26
|
+
// the failure is the whole file, so there is no line to cite). The file must look like a
|
|
27
|
+
// path (ends in a dotted extension, no embedded whitespace/colon) so this stays anchored to
|
|
28
|
+
// real diagnostics rather than matching arbitrary prose that happens to contain "error:".
|
|
29
|
+
{ re: /^([^\s:]+?\.[A-Za-z0-9]{1,10})\s*[:\-–—]\s*(?:fatal\s+error|error|warning)\b.*$/i, kind: "compiler-diagnostic" },
|
|
30
|
+
// Bundler style: "ERROR in ./src/components/Foo.tsx" (webpack et al.) — file, no line. The
|
|
31
|
+
// captured token must look like a path — leads with "./"/"../", or ends in a dotted extension
|
|
32
|
+
// of 1-10 alnum chars (same anchor the sibling pattern above uses) — so prose after "ERROR in"
|
|
33
|
+
// ("ERROR in the build pipeline", "ERROR in test suite failed to run") is left unmatched
|
|
34
|
+
// instead of handing back a fabricated file.
|
|
35
|
+
{ re: /^(?:ERROR|WARNING)\s+in\s+(\.{1,2}\/[^\s:]*|[^\s:]+\.[A-Za-z0-9]{1,10})\b/i, kind: "compiler-diagnostic" },
|
|
22
36
|
// Generic "Error: message" line followed later by a stack — capture the message alone.
|
|
23
37
|
{ re: /^\s*(?:Error|TypeError|ReferenceError|AssertionError)\s*:\s*(.+)$/, kind: "error-message" },
|
|
24
38
|
];
|
package/kernel/probe/owner.mjs
CHANGED
|
@@ -46,8 +46,11 @@ import { matchesAny } from "../../hooks/sandbox-guard.mjs";
|
|
|
46
46
|
*/
|
|
47
47
|
export function ownership(path, scopes, cwd = null) {
|
|
48
48
|
const rel = String(path).replace(/^\.\//, "");
|
|
49
|
-
const writers = (scopes || []).filter((s) => matchesAny(rel, s.allowed)).map((s) => s.scope_id).sort();
|
|
50
49
|
const shared = (scopes || []).filter((s) => matchesAny(rel, s.shared || [])).map((s) => s.scope_id).sort();
|
|
50
|
+
// `writers` is admits-the-path, the same union the sandbox fence composes (allowed ++ shared) —
|
|
51
|
+
// a path declared only in a contract's `shared` list is still a scope this path may write, and
|
|
52
|
+
// must read as owned rather than UNOWNED. `shared_with` (below) stays the narrower subset.
|
|
53
|
+
const writers = (scopes || []).filter((s) => matchesAny(rel, s.allowed) || matchesAny(rel, s.shared || [])).map((s) => s.scope_id).sort();
|
|
51
54
|
return { path: rel, owner: electOwner(rel, scopes), writers, shared_with: shared, exists: cwd ? existsSync(join(cwd, rel)) : null };
|
|
52
55
|
}
|
|
53
56
|
|