pi-crew 0.9.68 → 0.10.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +222 -0
- package/NOTICE.md +21 -0
- package/README.md +44 -2
- package/agents/analyst.md +1 -1
- package/agents/cold-verifier.md +3 -1
- package/agents/critic.md +1 -1
- package/agents/executor.md +1 -1
- package/agents/explorer.md +1 -1
- package/agents/planner.md +1 -1
- package/agents/reviewer.md +1 -1
- package/agents/security-reviewer.md +1 -1
- package/agents/test-engineer.md +1 -1
- package/agents/verifier.md +1 -1
- package/agents/writer.md +1 -1
- package/dist/index.mjs +68113 -60774
- package/docs/README.md +2 -0
- package/docs/actions-reference.md +31 -0
- package/docs/commands-reference.md +17 -6
- package/docs/resource-formats.md +13 -0
- package/package.json +4 -2
- package/schema.json +503 -91
- package/scripts/resource-sampler.mjs +36 -2
- package/skills/requirements-to-task-packet/SKILL.md +26 -0
- package/skills/widget-rendering/SKILL.md +7 -7
- package/src/agents/agent-config.ts +2 -1
- package/src/agents/discover-agents.ts +23 -14
- package/src/config/config-merge.ts +183 -0
- package/src/config/config-validation.ts +687 -0
- package/src/config/config.ts +22 -864
- package/src/config/defaults.ts +43 -2
- package/src/config/drift-detector.ts +1 -1
- package/src/config/env-vars.ts +691 -0
- package/src/config/role-tools.ts +11 -9
- package/src/config/sanitize-project-config.ts +172 -0
- package/src/config/types.ts +49 -1
- package/src/extension/async-notifier.ts +25 -2
- package/src/extension/crew-cleanup.ts +13 -0
- package/src/extension/crew-vibes/config.ts +2 -1
- package/src/extension/crew-vibes/footer.ts +19 -0
- package/src/extension/crew-vibes/index.ts +11 -1
- package/src/extension/plan-orchestrate.ts +132 -0
- package/src/extension/register.ts +8 -0
- package/src/extension/registration/command-registration.ts +1 -0
- package/src/extension/registration/commands/dashboard.ts +158 -0
- package/src/extension/registration/commands/index.ts +35 -0
- package/src/extension/registration/commands/manage.ts +303 -0
- package/src/extension/registration/commands/run.ts +228 -0
- package/src/extension/registration/commands/shared.ts +639 -0
- package/src/extension/registration/commands/status.ts +60 -0
- package/src/extension/registration/commands.ts +13 -1224
- package/src/extension/registration/foreground-run-controller.ts +10 -2
- package/src/extension/registration/lifecycle-handlers.ts +178 -17
- package/src/extension/registration/runtime-cleanup.ts +23 -5
- package/src/extension/registration/subagent-tools.ts +218 -9
- package/src/extension/registration/team-tool.ts +5 -1
- package/src/extension/registration/ui.ts +5 -4
- package/src/extension/rpc-hmac.ts +5 -3
- package/src/extension/team-tool/api/heartbeat.ts +47 -10
- package/src/extension/team-tool/api/plan-approval.ts +9 -0
- package/src/extension/team-tool/api/task-claims.ts +109 -40
- package/src/extension/team-tool/cancel.ts +84 -50
- package/src/extension/team-tool/dispatch/index.ts +1 -0
- package/src/extension/team-tool/dispatch/run.ts +4 -1
- package/src/extension/team-tool/doctor.ts +103 -1
- package/src/extension/team-tool/orchestrate.ts +66 -1
- package/src/extension/team-tool/plans.ts +192 -0
- package/src/extension/team-tool/respond.ts +197 -65
- package/src/extension/team-tool/run-deadline.ts +35 -3
- package/src/extension/team-tool/run-intent.ts +63 -0
- package/src/extension/team-tool/run.ts +74 -20
- package/src/extension/team-tool/status.ts +84 -26
- package/src/extension/team-tool.ts +11 -2
- package/src/hooks/registry.ts +1 -6
- package/src/i18n.ts +9 -0
- package/src/prompt/prompt-runtime.ts +521 -2
- package/src/prompt/worker-events-channel.ts +173 -0
- package/src/runtime/README.md +8 -8
- package/src/runtime/async-runner.ts +7 -3
- package/src/runtime/background-runner.ts +42 -14
- package/src/runtime/broker/broker-issuer.ts +9 -2
- package/src/runtime/broker/crew-broker-tokens.ts +43 -6
- package/src/runtime/broker/crew-broker.ts +838 -10
- package/src/runtime/broker/wait-status-cache.ts +157 -0
- package/src/runtime/budget-enforcement.ts +281 -0
- package/src/runtime/child-pi/child-pi-constants.ts +8 -0
- package/src/runtime/child-pi/child-pi-spawn.ts +60 -14
- package/src/runtime/child-pi/child-pi-streams.ts +21 -1
- package/src/runtime/child-pi/child-pi-timers.ts +324 -0
- package/src/runtime/child-pi/child-pi.ts +97 -201
- package/src/runtime/child-pi/mock-fixtures.ts +16 -2
- package/src/runtime/crew-agent-records.ts +259 -14
- package/src/runtime/delegate-spawn.ts +148 -0
- package/src/runtime/detached-run-results.ts +90 -0
- package/src/runtime/deterministic-ast.ts +2 -1
- package/src/runtime/dispatch-batch.ts +945 -0
- package/src/runtime/finalize-run.ts +557 -0
- package/src/runtime/goal-workflow/adaptive-plan.ts +116 -15
- package/src/runtime/goal-workflow/dynamic-workflow-runner.ts +8 -3
- package/src/runtime/goal-workflow/goal-state-store.ts +1 -1
- package/src/runtime/group-join.ts +11 -125
- package/src/runtime/live-session/live-session-runtime.ts +26 -1
- package/src/runtime/merge-gate.ts +32 -10
- package/src/runtime/merge-loop.ts +130 -0
- package/src/runtime/model/model-budget-summary.ts +53 -0
- package/src/runtime/model/model-fallback.ts +36 -2
- package/src/runtime/model/pi-args.ts +10 -0
- package/src/runtime/model/provider-extensions.ts +10 -0
- package/src/runtime/orphan-worker-registry.ts +1 -1
- package/src/runtime/output/output-validator.ts +45 -0
- package/src/runtime/parent-guard.ts +3 -1
- package/src/runtime/peer-dep.ts +2 -1
- package/src/runtime/per-write-validator.ts +0 -5
- package/src/runtime/pi-spawn.ts +61 -15
- package/src/runtime/plan-approval.ts +125 -0
- package/src/runtime/plan-replan.ts +151 -0
- package/src/runtime/process-status.ts +16 -1
- package/src/runtime/recovery/checkpoint.ts +0 -18
- package/src/runtime/recovery/crash-recovery.ts +111 -46
- package/src/runtime/run-tracker.ts +77 -10
- package/src/runtime/scheduler-context.ts +98 -0
- package/src/runtime/scheduling/coalesce-tasks.ts +5 -0
- package/src/runtime/scheduling/global-worker-cap.ts +2 -1
- package/src/runtime/scheduling/nested-slots.ts +70 -0
- package/src/runtime/scheduling/run-coalesced-task-group.ts +64 -13
- package/src/runtime/scheduling/task-graph-scheduler.ts +0 -10
- package/src/runtime/settings-store.ts +219 -0
- package/src/runtime/spawn-policy.ts +217 -0
- package/src/runtime/stale-reconciler.ts +87 -6
- package/src/runtime/subagent-manager.ts +25 -1
- package/src/runtime/task-output-context.ts +230 -9
- package/src/runtime/task-packet.ts +23 -1
- package/src/runtime/task-runner/child-executor.ts +106 -7
- package/src/runtime/task-runner/post-execution.ts +125 -1
- package/src/runtime/task-runner/pre-execution.ts +39 -1
- package/src/runtime/task-runner/prompt-builder.ts +51 -1
- package/src/runtime/task-runner/retrieval-orchestrator.ts +72 -18
- package/src/runtime/task-runner/spec-evidence.ts +403 -0
- package/src/runtime/task-runner/state-helpers.ts +26 -24
- package/src/runtime/task-runner.ts +11 -0
- package/src/runtime/team-runner.ts +132 -1673
- package/src/runtime/verification/spec-sandbox.ts +255 -0
- package/src/runtime/verification/verification-gates.ts +3 -2
- package/src/runtime/verification/verification-worktree.ts +2 -1
- package/src/runtime/workflow-phase-advance.ts +100 -0
- package/src/runtime/workspace-tree.ts +9 -0
- package/src/schema/config-schema.ts +66 -26
- package/src/schema/sensitive-config-paths.ts +64 -0
- package/src/schema/team-tool-schema.ts +13 -3
- package/src/state/README.md +4 -10
- package/src/state/atomic-write.ts +20 -3
- package/src/state/contracts.ts +38 -0
- package/src/state/coordination/mailbox.ts +12 -2
- package/src/state/event-log/cursor.ts +223 -0
- package/src/state/event-log/event-log-rotation.ts +12 -4
- package/src/state/event-log/event-log.ts +152 -369
- package/src/state/event-log/sequence-cache.ts +373 -0
- package/src/state/event-log/worker-atomic-writer.ts +2 -1
- package/src/state/stores/active-run-registry.ts +3 -2
- package/src/state/stores/manifest-io.ts +237 -0
- package/src/state/stores/ownership-map.ts +162 -0
- package/src/state/stores/plan-store.ts +241 -0
- package/src/state/stores/run-cache.ts +0 -90
- package/src/state/stores/spec-store.ts +189 -0
- package/src/state/stores/state-store.ts +139 -232
- package/src/state/types.ts +199 -0
- package/src/ui/dashboard-panes/plan-pane.ts +136 -0
- package/src/ui/dashboard-panes/progress-pane.ts +6 -0
- package/src/ui/dashboard-panes/transcript-pane.ts +31 -0
- package/src/ui/dock-footer.ts +49 -0
- package/src/ui/heartbeat-aggregator.ts +9 -1
- package/src/ui/inline-panel/agent-pane.ts +375 -0
- package/src/ui/inline-panel/agent-transcript.ts +338 -0
- package/src/ui/inline-panel/agent-view-overlay.ts +225 -0
- package/src/ui/inline-panel/crew-editor.ts +192 -0
- package/src/ui/inline-panel/index.ts +290 -0
- package/src/ui/inline-panel/panel-rows.ts +37 -0
- package/src/ui/inline-panel/panel-selection.ts +157 -0
- package/src/ui/inline-panel/panel-store.ts +111 -0
- package/src/ui/inline-panel/view-session-store.ts +36 -0
- package/src/ui/keybinding-map.ts +54 -13
- package/src/ui/pi-ui-compat.ts +9 -0
- package/src/ui/powerbar-publisher.ts +52 -1
- package/src/ui/run-dashboard.ts +31 -5
- package/src/ui/run-snapshot-cache.ts +57 -30
- package/src/ui/snapshot-types.ts +6 -1
- package/src/ui/widget/index.ts +176 -22
- package/src/ui/widget/task-list.ts +198 -0
- package/src/ui/widget/widget-formatters.ts +240 -4
- package/src/ui/widget/widget-renderer.ts +243 -38
- package/src/ui/widget/widget-types.ts +11 -0
- package/src/utils/child-process-shield.ts +106 -0
- package/src/utils/file-coalescer.ts +0 -4
- package/src/utils/fs-errno.ts +66 -0
- package/src/utils/fs-watch.ts +1 -1
- package/src/utils/internal-error.ts +3 -1
- package/src/utils/paths.ts +11 -3
- package/src/utils/redaction.ts +7 -0
- package/src/utils/safe-abort.ts +45 -0
- package/src/utils/task-name-generator.ts +1 -8
- package/src/workflows/discover-workflows.ts +20 -2
- package/src/workflows/validate-workflow.ts +7 -1
- package/src/workflows/workflow-config.ts +17 -0
- package/src/workflows/workflow-serializer.ts +3 -0
- package/src/worktree/worktree-manager.ts +22 -0
- package/workflows/default.workflow.md +36 -26
- package/workflows/strict-fast-fix.workflow.md +26 -0
- package/src/agents/agent-search.ts +0 -98
- package/src/benchmark/benchmark-runner.ts +0 -313
- package/src/benchmark/feedback-loop.ts +0 -73
- package/src/config/resilient-parser.ts +0 -117
- package/src/extension/crew-vibes/cat-frames.ts +0 -18
- package/src/extension/result-watcher.ts +0 -139
- package/src/observability/exporters/prometheus-exporter.ts +0 -54
- package/src/observability/metric-retention.ts +0 -64
- package/src/runtime/compaction/compaction-summary.ts +0 -278
- package/src/runtime/errors/crew-errors.ts +0 -162
- package/src/runtime/live-session/intercom-bridge.ts +0 -187
- package/src/runtime/loop-gates.ts +0 -128
- package/src/runtime/metric-parser.ts +0 -36
- package/src/runtime/output/stream-preview.ts +0 -184
- package/src/runtime/output/tool-progress.ts +0 -278
- package/src/runtime/phase-tracker.ts +0 -385
- package/src/runtime/pipeline-runner.ts +0 -523
- package/src/runtime/process/process-lifecycle.ts +0 -491
- package/src/runtime/recovery/retry-runner.ts +0 -330
- package/src/runtime/run-drift.ts +0 -219
- package/src/runtime/task-quality.ts +0 -199
- package/src/runtime/task-runner/run-projection.ts +0 -128
- package/src/runtime/verification/post-checks.ts +0 -142
- package/src/state/coordination/schedule.ts +0 -166
- package/src/state/event-log/jsonl-writer.ts +0 -115
- package/src/state/hook-instinct-bridge.ts +0 -94
- package/src/state/hook-integrations.ts +0 -51
- package/src/state/session-state-map.ts +0 -51
- package/src/state/stores/blob-store.ts +0 -308
- package/src/state/stores/instinct-store.ts +0 -275
- package/src/state/stores/observation-store.ts +0 -176
- package/src/state/tiered-eval.ts +0 -480
- package/src/state/types-eval.ts +0 -58
- package/src/tools/safe-bash-extension.ts +0 -54
- package/src/tools/safe-bash.ts +0 -505
- package/src/ui/agent-management-overlay.ts +0 -160
- package/src/ui/crew-footer.ts +0 -102
- package/src/ui/crew-select-list.ts +0 -114
- package/src/ui/dashboard-panes/capability-pane.ts +0 -77
- package/src/ui/transcript-entries.ts +0 -256
- package/src/utils/conflict-detect.ts +0 -721
- package/src/utils/fingerprint.ts +0 -180
- package/src/utils/gh-protocol.ts +0 -556
- package/src/utils/project-detector.ts +0 -160
- package/src/utils/sse-parser.ts +0 -131
- package/src/workflows/cost-estimator.ts +0 -34
- package/src/workflows/intermediate-store.ts +0 -166
|
@@ -0,0 +1,403 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* spec-evidence.ts — SPEC-EVIDENCE footer parser + coverage/strict gates
|
|
3
|
+
* (ADR-6 §2/§3/§4, WP-6 steps 3-4; round-1 review fixes).
|
|
4
|
+
*
|
|
5
|
+
* Footer contract (executor prompt, §2): the result must END with
|
|
6
|
+
*
|
|
7
|
+
* ```
|
|
8
|
+
* SPEC-EVIDENCE:
|
|
9
|
+
* <acceptanceId>: <one-line evidence>
|
|
10
|
+
* ```
|
|
11
|
+
*
|
|
12
|
+
* Parser semantics (round-1): the footer is the TRAILING region of the text —
|
|
13
|
+
* the contiguous run of marker lines, entry lines, and blank lines that
|
|
14
|
+
* reaches EOF. Earlier/quoted markers followed by prose are ignored (a quoted
|
|
15
|
+
* dependency footer cannot shadow or pollute the real one); repeated markers
|
|
16
|
+
* inside the trailing region are block separators, and citations UNION across
|
|
17
|
+
* blocks (a worker emitting one block per spec keeps ALL citations).
|
|
18
|
+
*
|
|
19
|
+
* Non-strict default (§3): mechanical coverage only — must-acceptance ids
|
|
20
|
+
* cited >= 1 time; gaps surface as an `unverified` badge, never a block.
|
|
21
|
+
* Strict (§4): coverage AND machine-check; trust is read from the snapshot's
|
|
22
|
+
* frozen `trustedAtFreeze` bit (provenance v2 — the live sidecar is never
|
|
23
|
+
* consulted at finalize, closing the TOCTOU window).
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import type { SpecGateResult, SpecSnapshot, TaskPacket } from "../../state/types.ts";
|
|
27
|
+
import { isSpecSandboxSupported, runSpecCheck, type SpecCheckOutcome } from "../verification/spec-sandbox.ts";
|
|
28
|
+
|
|
29
|
+
const FOOTER_MARKER = "SPEC-EVIDENCE:";
|
|
30
|
+
/** Acceptance ids are minted by the spec store; the parser stays mechanical —
|
|
31
|
+
* any `<token>:` line in the footer region is a citation, unknown tokens
|
|
32
|
+
* surface as unknownIds rather than being rejected here. */
|
|
33
|
+
const ENTRY_PATTERN = /^([A-Za-z0-9][A-Za-z0-9._-]{0,127}):\s?(.*)$/;
|
|
34
|
+
|
|
35
|
+
export interface SpecEvidenceFooter {
|
|
36
|
+
present: boolean;
|
|
37
|
+
/** acceptanceId → last one-line evidence text for that id. */
|
|
38
|
+
entries: Record<string, string>;
|
|
39
|
+
/** Cited ids in citation order (duplicates preserved for diagnostics). */
|
|
40
|
+
citedIds: string[];
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
const isBlank = (line: string): boolean => line.trim() === "";
|
|
44
|
+
const isMarker = (line: string): boolean => line.trim() === FOOTER_MARKER;
|
|
45
|
+
|
|
46
|
+
/** Parse the trailing SPEC-EVIDENCE footer region (see module header). */
|
|
47
|
+
export function parseSpecEvidenceFooter(text: string): SpecEvidenceFooter {
|
|
48
|
+
const lines = (text ?? "").split(/\r?\n/);
|
|
49
|
+
// Walk back over trailing blanks to the last content line.
|
|
50
|
+
let end = lines.length;
|
|
51
|
+
while (end > 0 && isBlank(lines[end - 1] ?? "")) end--;
|
|
52
|
+
// Extend the region backward over footer material (markers/entries/blanks).
|
|
53
|
+
let start = end;
|
|
54
|
+
while (start > 0) {
|
|
55
|
+
const line = lines[start - 1] ?? "";
|
|
56
|
+
if (isBlank(line) || isMarker(line) || ENTRY_PATTERN.test(line.trim())) {
|
|
57
|
+
start--;
|
|
58
|
+
continue;
|
|
59
|
+
}
|
|
60
|
+
break;
|
|
61
|
+
}
|
|
62
|
+
// Parse forward; require at least one marker INSIDE the region.
|
|
63
|
+
const entries: Record<string, string> = {};
|
|
64
|
+
const citedIds: string[] = [];
|
|
65
|
+
let sawMarker = false;
|
|
66
|
+
for (let i = start; i < end; i++) {
|
|
67
|
+
const line = lines[i] ?? "";
|
|
68
|
+
if (isMarker(line)) {
|
|
69
|
+
sawMarker = true;
|
|
70
|
+
continue; // block separator
|
|
71
|
+
}
|
|
72
|
+
if (isBlank(line)) continue;
|
|
73
|
+
const match = ENTRY_PATTERN.exec(line.trim());
|
|
74
|
+
if (!match) continue; // unreachable given the backward scan; defensive
|
|
75
|
+
citedIds.push(match[1]);
|
|
76
|
+
entries[match[1]] = match[2].trim();
|
|
77
|
+
}
|
|
78
|
+
if (!sawMarker) return { present: false, entries: {}, citedIds: [] };
|
|
79
|
+
return { present: true, entries, citedIds };
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/** Union footers parsed from several authoritative result sources (round-1:
|
|
83
|
+
* a footer may live in rawFinalText OR finalText OR finalStdout — child
|
|
84
|
+
* compaction can empty finalText while the footer survives elsewhere).
|
|
85
|
+
* First-seen evidence wins; citedIds keep first-seen order. */
|
|
86
|
+
export function mergeFooters(footers: SpecEvidenceFooter[]): SpecEvidenceFooter {
|
|
87
|
+
const entries: Record<string, string> = {};
|
|
88
|
+
const seen = new Set<string>();
|
|
89
|
+
const citedIds: string[] = [];
|
|
90
|
+
let present = false;
|
|
91
|
+
for (const f of footers) {
|
|
92
|
+
present = present || f.present;
|
|
93
|
+
for (const id of f.citedIds) {
|
|
94
|
+
if (!seen.has(id)) {
|
|
95
|
+
seen.add(id);
|
|
96
|
+
citedIds.push(id);
|
|
97
|
+
entries[id] = f.entries[id] ?? "";
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
return { present, entries, citedIds };
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/** Collect every acceptance id a snapshot pair exposes, split by the
|
|
105
|
+
* requirement's priority. Only `must` participates in coverage. */
|
|
106
|
+
function snapshotAcceptances(snapshots: SpecSnapshot[]): {
|
|
107
|
+
mustIds: string[];
|
|
108
|
+
knownIds: Set<string>;
|
|
109
|
+
} {
|
|
110
|
+
const mustIds: string[] = [];
|
|
111
|
+
const knownIds = new Set<string>();
|
|
112
|
+
for (const snap of snapshots) {
|
|
113
|
+
for (const item of snap.items) {
|
|
114
|
+
knownIds.add(item.acceptance.id);
|
|
115
|
+
if (item.requirement.priority === "must") mustIds.push(item.acceptance.id);
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
return { mustIds, knownIds };
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/** Coverage-only evaluation (§3). */
|
|
122
|
+
export function evaluateSpecCoverage(snapshots: SpecSnapshot[] | undefined, footer: SpecEvidenceFooter): SpecGateResult {
|
|
123
|
+
if (!snapshots || snapshots.length === 0) {
|
|
124
|
+
// Spec-less tasks are untouched (regression guard, B4-j).
|
|
125
|
+
return {
|
|
126
|
+
mode: "coverage",
|
|
127
|
+
applicable: false,
|
|
128
|
+
footerPresent: footer.present,
|
|
129
|
+
citedIds: footer.citedIds,
|
|
130
|
+
missingMustIds: [],
|
|
131
|
+
unknownIds: [],
|
|
132
|
+
evidence: footer.entries,
|
|
133
|
+
};
|
|
134
|
+
}
|
|
135
|
+
const { mustIds, knownIds } = snapshotAcceptances(snapshots);
|
|
136
|
+
const cited = new Set(footer.citedIds);
|
|
137
|
+
const missingMustIds = mustIds.filter((id) => !cited.has(id));
|
|
138
|
+
const unknownIds = footer.citedIds.filter((id) => !knownIds.has(id));
|
|
139
|
+
// §2: a missing footer is a gate event ONLY where the task has
|
|
140
|
+
// must-acceptances; should/could-only specs never produce a badge.
|
|
141
|
+
const complete = (mustIds.length === 0 || (footer.present && missingMustIds.length === 0)) && unknownIds.length === 0;
|
|
142
|
+
return {
|
|
143
|
+
mode: "coverage",
|
|
144
|
+
applicable: true,
|
|
145
|
+
footerPresent: footer.present,
|
|
146
|
+
citedIds: footer.citedIds,
|
|
147
|
+
missingMustIds,
|
|
148
|
+
unknownIds,
|
|
149
|
+
// Badge fires ONLY on mechanically-detectable gaps: missing footer,
|
|
150
|
+
// missing must-ids, unknown ids. Full-coverage fabrication passes
|
|
151
|
+
// WITHOUT a badge (§3 honesty rule).
|
|
152
|
+
...(complete ? {} : { badge: "unverified" as const }),
|
|
153
|
+
evidence: footer.entries,
|
|
154
|
+
};
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
// ── §4: strict mode — coverage AND machine-check ─────────────────────────
|
|
158
|
+
|
|
159
|
+
/** Per-acceptance strict outcome. Sandbox failures carry digest-only fields
|
|
160
|
+
* (leak discipline — raw output never persists). */
|
|
161
|
+
export type StrictCheckKind =
|
|
162
|
+
| "passed"
|
|
163
|
+
| "degraded-untrusted-spec"
|
|
164
|
+
| "degraded-non-idempotent"
|
|
165
|
+
| "degraded-no-command"
|
|
166
|
+
| "degraded-scaffold-mode"
|
|
167
|
+
| "degraded-already-failed"
|
|
168
|
+
| "failed";
|
|
169
|
+
|
|
170
|
+
export interface StrictCheckResult {
|
|
171
|
+
specId: string;
|
|
172
|
+
acceptanceId: string;
|
|
173
|
+
result: StrictCheckKind;
|
|
174
|
+
/** Sandbox outcome kind when result === "failed" (ADR §4 payload schema). */
|
|
175
|
+
outcome?: SpecCheckOutcome["outcome"];
|
|
176
|
+
expectedDigest?: string;
|
|
177
|
+
actualDigest?: string;
|
|
178
|
+
exitCode?: number | null;
|
|
179
|
+
signal?: string;
|
|
180
|
+
durationMs?: number;
|
|
181
|
+
stderrLength?: number;
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
export interface StrictGateReport {
|
|
185
|
+
checks: StrictCheckResult[];
|
|
186
|
+
/** Strict pass = coverage complete AND zero failed checks. Degradations
|
|
187
|
+
* badge the task but never fail it (the §4 compromise path). */
|
|
188
|
+
passed: boolean;
|
|
189
|
+
/** True when the platform cannot host the re-run sandbox (macOS/Windows) —
|
|
190
|
+
* every check fails closed; callers emit a loud platform warning. */
|
|
191
|
+
platformUnsupported: boolean;
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
export interface StrictEvalOptions {
|
|
195
|
+
/** Command cwd — the TASK's workspace (worktree-aware), not the run root. */
|
|
196
|
+
cwd: string;
|
|
197
|
+
/** "scaffold" (dry-run / executeWorkers=false) skips ALL machine-checks —
|
|
198
|
+
* the documented disable switch must reach the sandbox (round-1 P2). */
|
|
199
|
+
mode?: "run" | "scaffold";
|
|
200
|
+
/** Task already failed upstream (empty-result classifier / mutation guard):
|
|
201
|
+
* machine-checks are skipped — they cannot un-fail the task and re-running
|
|
202
|
+
* adds up to 60s × N to finalizing a dead task (round-1 P3). */
|
|
203
|
+
alreadyFailed?: boolean;
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
/** Strict-mode gate (§4): coverage (§3) AND machine-check of every
|
|
207
|
+
* machine-checkable must-acceptance. Executes ONLY snapshot-frozen commands
|
|
208
|
+
* — never the live state/specs/<id>.json (B4-k). Provenance v2: trust is the
|
|
209
|
+
* snapshot's `trustedAtFreeze` bit, minted from the user-store digest sidecar
|
|
210
|
+
* AT FREEZE — worker-authored specs, hand-forged workspace records, and
|
|
211
|
+
* post-freeze sidecar tampering all degrade to coverage-only. */
|
|
212
|
+
export async function evaluateSpecStrict(
|
|
213
|
+
snapshots: SpecSnapshot[] | undefined,
|
|
214
|
+
footer: SpecEvidenceFooter,
|
|
215
|
+
options: StrictEvalOptions,
|
|
216
|
+
): Promise<SpecGateResult & { strict: StrictGateReport }> {
|
|
217
|
+
const coverage = evaluateSpecCoverage(snapshots, footer);
|
|
218
|
+
if (!coverage.applicable) {
|
|
219
|
+
return {
|
|
220
|
+
...coverage,
|
|
221
|
+
mode: "strict",
|
|
222
|
+
strict: { checks: [], passed: true, platformUnsupported: !isSpecSandboxSupported() },
|
|
223
|
+
};
|
|
224
|
+
}
|
|
225
|
+
const platformUnsupported = !isSpecSandboxSupported();
|
|
226
|
+
const checks: StrictCheckResult[] = [];
|
|
227
|
+
let failed = false;
|
|
228
|
+
let degraded = false;
|
|
229
|
+
const skipMachineChecks = options.mode === "scaffold" || options.alreadyFailed === true;
|
|
230
|
+
for (const snap of snapshots ?? []) {
|
|
231
|
+
for (const item of snap.items) {
|
|
232
|
+
if (item.requirement.priority !== "must") continue; // should/could never block
|
|
233
|
+
if (snap.trustedAtFreeze !== true) {
|
|
234
|
+
// NEW-2: re-running a generated/unattested spec's commands would be a
|
|
235
|
+
// privilege-escalation vector — degrade to coverage-only.
|
|
236
|
+
degraded = true;
|
|
237
|
+
checks.push({ specId: snap.specId, acceptanceId: item.acceptance.id, result: "degraded-untrusted-spec" });
|
|
238
|
+
continue;
|
|
239
|
+
}
|
|
240
|
+
if (skipMachineChecks) {
|
|
241
|
+
degraded = true;
|
|
242
|
+
checks.push({
|
|
243
|
+
specId: snap.specId,
|
|
244
|
+
acceptanceId: item.acceptance.id,
|
|
245
|
+
result: options.mode === "scaffold" ? "degraded-scaffold-mode" : "degraded-already-failed",
|
|
246
|
+
});
|
|
247
|
+
continue;
|
|
248
|
+
}
|
|
249
|
+
if (!item.acceptance.command) {
|
|
250
|
+
degraded = true;
|
|
251
|
+
checks.push({ specId: snap.specId, acceptanceId: item.acceptance.id, result: "degraded-no-command" });
|
|
252
|
+
continue;
|
|
253
|
+
}
|
|
254
|
+
if (item.acceptance.idempotent !== true) {
|
|
255
|
+
// Non-idempotent musts cannot be machine-re-run (§4).
|
|
256
|
+
degraded = true;
|
|
257
|
+
checks.push({ specId: snap.specId, acceptanceId: item.acceptance.id, result: "degraded-non-idempotent" });
|
|
258
|
+
continue;
|
|
259
|
+
}
|
|
260
|
+
const outcome = await runSpecCheck(
|
|
261
|
+
{
|
|
262
|
+
command: item.acceptance.command,
|
|
263
|
+
expectedDigest: item.acceptance.expectedDigest,
|
|
264
|
+
expectedExitCode: item.acceptance.expectedExitCode,
|
|
265
|
+
},
|
|
266
|
+
{ cwd: options.cwd },
|
|
267
|
+
);
|
|
268
|
+
if (outcome.outcome === "passed") {
|
|
269
|
+
checks.push({
|
|
270
|
+
specId: snap.specId,
|
|
271
|
+
acceptanceId: item.acceptance.id,
|
|
272
|
+
result: "passed",
|
|
273
|
+
exitCode: outcome.exitCode,
|
|
274
|
+
durationMs: outcome.durationMs,
|
|
275
|
+
});
|
|
276
|
+
} else {
|
|
277
|
+
failed = true;
|
|
278
|
+
checks.push({
|
|
279
|
+
specId: snap.specId,
|
|
280
|
+
acceptanceId: item.acceptance.id,
|
|
281
|
+
result: "failed",
|
|
282
|
+
outcome: outcome.outcome,
|
|
283
|
+
expectedDigest: item.acceptance.expectedDigest,
|
|
284
|
+
actualDigest: outcome.actualDigest,
|
|
285
|
+
exitCode: outcome.exitCode,
|
|
286
|
+
signal: outcome.signal,
|
|
287
|
+
durationMs: outcome.durationMs,
|
|
288
|
+
stderrLength: outcome.stderrLength,
|
|
289
|
+
});
|
|
290
|
+
}
|
|
291
|
+
}
|
|
292
|
+
}
|
|
293
|
+
// Missing footer w/ musts ⇒ missingMustIds is non-empty (nothing cited),
|
|
294
|
+
// so strict coverage-completeness = no missing musts AND no unknown ids.
|
|
295
|
+
// (A should-only spec with no footer has no musts → nothing to fail.)
|
|
296
|
+
const coverageComplete = coverage.missingMustIds.length === 0 && coverage.unknownIds.length === 0;
|
|
297
|
+
const passed = !failed && coverageComplete;
|
|
298
|
+
const result: SpecGateResult & { strict: StrictGateReport } = {
|
|
299
|
+
...coverage,
|
|
300
|
+
mode: "strict",
|
|
301
|
+
// In strict mode a coverage gap FAILS the gate (B4-c strict column) —
|
|
302
|
+
// the badge still rides along for UI visibility.
|
|
303
|
+
...(coverageComplete && !degraded ? {} : { badge: "unverified" as const }),
|
|
304
|
+
strict: { checks, passed, platformUnsupported },
|
|
305
|
+
};
|
|
306
|
+
return result;
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
// ── finalize-time orchestration (extracted for testability, round-1 P3) ────
|
|
310
|
+
|
|
311
|
+
export interface SpecGateEventData {
|
|
312
|
+
type: "task.spec_gate" | "spec.strict_platform_warning" | "spec.check_failed" | "spec.freeze_failed";
|
|
313
|
+
data: Record<string, unknown>;
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
export interface SpecGateOutcome {
|
|
317
|
+
specGate: (SpecGateResult & { strict?: StrictGateReport }) | undefined;
|
|
318
|
+
/** Events the caller must append (payloads digest-only / ids only). */
|
|
319
|
+
events: SpecGateEventData[];
|
|
320
|
+
/** Set when the write-gate must fail (strict). The caller PREFIXES this to
|
|
321
|
+
* any pre-existing error instead of replacing it (round-1 P3). */
|
|
322
|
+
gateError?: string;
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
/** Compute the finalize-time spec gate for one task. Pure-ish orchestration:
|
|
326
|
+
* no manifest/event-log coupling so the wiring is unit-testable. Footer is
|
|
327
|
+
* the union of every authoritative result source (artifact-chain order). */
|
|
328
|
+
export async function computeSpecGate(args: {
|
|
329
|
+
packet: TaskPacket;
|
|
330
|
+
rawFinalText?: string;
|
|
331
|
+
finalText?: string;
|
|
332
|
+
finalStdout?: string;
|
|
333
|
+
/** Task workspace (worktree-aware) — sandbox cwd. */
|
|
334
|
+
sandboxCwd: string;
|
|
335
|
+
runtimeKind: string;
|
|
336
|
+
/** True when the task already failed upstream (classifier/mutation guard). */
|
|
337
|
+
alreadyFailed: boolean;
|
|
338
|
+
}): Promise<SpecGateOutcome> {
|
|
339
|
+
const packet = args.packet;
|
|
340
|
+
const hasSpecRefs =
|
|
341
|
+
(packet.specRefs?.length ?? 0) > 0 || (packet.specSnapshots?.length ?? 0) > 0 || (packet.unresolvedSpecRefs?.length ?? 0) > 0;
|
|
342
|
+
if (!hasSpecRefs) return { specGate: undefined, events: [] }; // spec-less: untouched (B4-j)
|
|
343
|
+
const footer = mergeFooters([
|
|
344
|
+
parseSpecEvidenceFooter(args.rawFinalText ?? ""),
|
|
345
|
+
parseSpecEvidenceFooter(args.finalText ?? ""),
|
|
346
|
+
parseSpecEvidenceFooter(args.finalStdout ?? ""),
|
|
347
|
+
]);
|
|
348
|
+
const events: SpecGateEventData[] = [];
|
|
349
|
+
const unresolved = packet.unresolvedSpecRefs ?? [];
|
|
350
|
+
if (unresolved.length > 0) {
|
|
351
|
+
events.push({ type: "spec.freeze_failed", data: { unresolvedSpecRefs: unresolved } });
|
|
352
|
+
}
|
|
353
|
+
if (packet.specStrict === true) {
|
|
354
|
+
const strictResult = await evaluateSpecStrict(packet.specSnapshots, footer, {
|
|
355
|
+
cwd: args.sandboxCwd,
|
|
356
|
+
mode: args.runtimeKind === "scaffold" ? "scaffold" : "run",
|
|
357
|
+
alreadyFailed: args.alreadyFailed,
|
|
358
|
+
});
|
|
359
|
+
const specGate =
|
|
360
|
+
unresolved.length > 0
|
|
361
|
+
? { ...strictResult, badge: "unverified" as const, missingMustIds: strictResult.missingMustIds }
|
|
362
|
+
: strictResult;
|
|
363
|
+
if (strictResult.strict.platformUnsupported) {
|
|
364
|
+
events.push({
|
|
365
|
+
type: "spec.strict_platform_warning",
|
|
366
|
+
data: { platform: process.platform, effect: "strict checks fail closed (no unshare)" },
|
|
367
|
+
});
|
|
368
|
+
}
|
|
369
|
+
for (const check of strictResult.strict.checks) {
|
|
370
|
+
if (check.result !== "failed") continue;
|
|
371
|
+
events.push({
|
|
372
|
+
type: "spec.check_failed",
|
|
373
|
+
data: {
|
|
374
|
+
specId: check.specId,
|
|
375
|
+
acceptanceId: check.acceptanceId,
|
|
376
|
+
outcome: check.outcome,
|
|
377
|
+
expectedDigest: check.expectedDigest,
|
|
378
|
+
actualDigest: check.actualDigest,
|
|
379
|
+
exitCode: check.exitCode,
|
|
380
|
+
signal: check.signal,
|
|
381
|
+
durationMs: check.durationMs,
|
|
382
|
+
},
|
|
383
|
+
});
|
|
384
|
+
}
|
|
385
|
+
const strictPassed = strictResult.strict.passed && unresolved.length === 0;
|
|
386
|
+
if (!strictPassed) {
|
|
387
|
+
const reasons = [
|
|
388
|
+
...(unresolved.length ? [`unresolved specRefs at freeze: ${unresolved.join(", ")}`] : []),
|
|
389
|
+
...(strictResult.missingMustIds.length
|
|
390
|
+
? [`missing must-acceptance evidence: ${strictResult.missingMustIds.join(", ")}`]
|
|
391
|
+
: []),
|
|
392
|
+
...(strictResult.unknownIds.length ? [`unknown acceptance ids cited: ${strictResult.unknownIds.join(", ")}`] : []),
|
|
393
|
+
...strictResult.strict.checks.filter((c) => c.result === "failed").map((c) => `${c.acceptanceId}: ${c.outcome}`),
|
|
394
|
+
].join("; ");
|
|
395
|
+
return { specGate, events, gateError: `Spec strict gate failed (${reasons || "machine-check failure"})` };
|
|
396
|
+
}
|
|
397
|
+
return { specGate, events };
|
|
398
|
+
}
|
|
399
|
+
// Non-strict: coverage only; unresolved refs badge but never block.
|
|
400
|
+
const coverage = evaluateSpecCoverage(packet.specSnapshots, footer);
|
|
401
|
+
const specGate = unresolved.length > 0 ? { ...coverage, applicable: true, badge: "unverified" as const } : coverage;
|
|
402
|
+
return { specGate, events };
|
|
403
|
+
}
|
|
@@ -43,12 +43,12 @@ export function persistSingleTaskUpdate(
|
|
|
43
43
|
): TeamTaskState[] {
|
|
44
44
|
// H5 (2026-08-10): lowered from 100 → 10. Each retry does
|
|
45
45
|
// flushPendingAtomicWrites (global) + loadRunManifestById (stat + parse)
|
|
46
|
-
// + statSync, ~5ms each.
|
|
47
|
-
//
|
|
48
|
-
//
|
|
49
|
-
//
|
|
50
|
-
//
|
|
51
|
-
//
|
|
46
|
+
// + statSync, ~5ms each. Every attempt now loads from disk (BUG-028);
|
|
47
|
+
// retries only fire under real contention from best-effort writers that
|
|
48
|
+
// don't hold the run lock (async-notifier, crash-recovery). If 10 retries
|
|
49
|
+
// cannot converge, the system is in a pathological state where 100 would
|
|
50
|
+
// not help either — the explicit error below surfaces it instead of
|
|
51
|
+
// blocking the event loop for 500ms.
|
|
52
52
|
const MAX_CAS_ATTEMPTS = 10;
|
|
53
53
|
let baseMtime = 0;
|
|
54
54
|
try {
|
|
@@ -83,24 +83,26 @@ export function persistSingleTaskUpdate(
|
|
|
83
83
|
// overwrite our buffered write between our load and our (async)
|
|
84
84
|
// fsync, silently losing the intermediate update.
|
|
85
85
|
flushPendingAtomicWrites();
|
|
86
|
-
//
|
|
87
|
-
//
|
|
88
|
-
//
|
|
89
|
-
// loadRunManifestById and handed them in as
|
|
90
|
-
//
|
|
91
|
-
//
|
|
92
|
-
//
|
|
93
|
-
//
|
|
94
|
-
//
|
|
95
|
-
//
|
|
96
|
-
//
|
|
97
|
-
//
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
86
|
+
// BUG-028 (2026-08-16): ALWAYS load the committed tasks from disk
|
|
87
|
+
// inside the lock — never trust fallbackTasks on attempt 0. The
|
|
88
|
+
// old F4 perf shortcut assumed "the caller already obtained the
|
|
89
|
+
// latest tasks via loadRunManifestById and handed them in as
|
|
90
|
+
// fallbackTasks", but that assumption is FALSE for fan-out
|
|
91
|
+
// workers: dispatch-batch.ts hands every unit the SAME dispatch-
|
|
92
|
+
// time snapshot (ctx.tasks), and the worker's array is only ever
|
|
93
|
+
// updated with its OWN task (updateTask). When the LAST unit of
|
|
94
|
+
// a batch terminal-persists, attempt 0 wrote the FULL stale array
|
|
95
|
+
// (siblings still "running") over disk where siblings were already
|
|
96
|
+
// terminal — resurrecting them and blocking finalize ("task is
|
|
97
|
+
// still running"). The mtime CAS below cannot catch this: it only
|
|
98
|
+
// detects writers between the entry stat and the in-lock stat,
|
|
99
|
+
// i.e. staleness acquired AFTER function entry — not fallback
|
|
100
|
+
// staleness that predates it. Loading disk here makes sibling
|
|
101
|
+
// state authoritative (matching mergeUnitResult / bug-027 policy)
|
|
102
|
+
// while `updated` still wins for THIS task via updateTask.
|
|
103
|
+
// fallbackTasks remains the fallback when the run state is absent
|
|
104
|
+
// (fresh run, manifest not yet on disk).
|
|
105
|
+
const latest = loadRunManifestById(manifest.cwd, manifest.runId)?.tasks ?? fallbackTasks;
|
|
104
106
|
merged = updateTask(latest, taskWithCheckpoint);
|
|
105
107
|
|
|
106
108
|
// F2: collapsed from 3 redundant statSync calls into 1. The previous
|
|
@@ -8,6 +8,7 @@ import { registerStreamBridge } from "./event-stream-bridge.ts";
|
|
|
8
8
|
import type { ModelAttemptSummary } from "./model/model-fallback.ts";
|
|
9
9
|
import { awaitRuntimeWarmup } from "./model/runtime-warmup.ts";
|
|
10
10
|
import type { ParsedPiJsonOutput } from "./output/pi-json-output.ts";
|
|
11
|
+
import type { ResultArtifactReadCache } from "./task-output-context.ts";
|
|
11
12
|
import { runChildProcessTask } from "./task-runner/child-executor.ts";
|
|
12
13
|
import { finalizeTaskResult, type TaskExecutionResult } from "./task-runner/post-execution.ts";
|
|
13
14
|
import { prepareTaskExecutionContext } from "./task-runner/pre-execution.ts";
|
|
@@ -74,6 +75,16 @@ export interface TaskRunnerInput {
|
|
|
74
75
|
* into every runTeamTask call).
|
|
75
76
|
*/
|
|
76
77
|
spawnBudget?: SpawnBudget;
|
|
78
|
+
/**
|
|
79
|
+
* R10-1 residual: per-run result-artifact read cache shared with the
|
|
80
|
+
* closeout aggregation. Threading it here lets collectDependencyOutputContext
|
|
81
|
+
* reuse reads the closeout already performed (fan-in graphs re-read the same
|
|
82
|
+
* `results/<taskId>.txt` once per downstream dispatch + retry). ONE instance
|
|
83
|
+
* per RUN (created in executeTeamRunCore, spread through baseInput so
|
|
84
|
+
* retries inherit it by reference) — never per call. Optional: undefined
|
|
85
|
+
* keeps the uncached behavior.
|
|
86
|
+
*/
|
|
87
|
+
resultReadCache?: ResultArtifactReadCache;
|
|
77
88
|
}
|
|
78
89
|
|
|
79
90
|
export async function runTeamTask(input: TaskRunnerInput): Promise<{ manifest: TeamRunManifest; tasks: TeamTaskState[] }> {
|