opencode-swarm 7.180.1 → 7.181.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -0
- package/binaries/win32-arm64/swarm-sandbox-runner.exe +0 -0
- package/binaries/win32-x64/swarm-sandbox-runner.exe +0 -0
- package/dist/cli/{coder-settlement-b57917a3.js → coder-settlement-ee8bmg83.js} +4 -4
- package/dist/cli/{config-doctor-ebat06eq.js → config-doctor-mnmxzg86.js} +3 -3
- package/dist/cli/{core-8k41qh0d.js → core-m9yymv84.js} +2 -2
- package/dist/cli/{curator-8yb3qkxs.js → curator-7nmwpcxy.js} +23 -23
- package/dist/cli/{curator-drift-3av9hffm.js → curator-drift-q42fr0c9.js} +5 -5
- package/dist/cli/{curator-llm-factory-htgqafq0.js → curator-llm-factory-s1b2p162.js} +23 -23
- package/dist/cli/{evidence-summary-service-s258ea5t.js → evidence-summary-service-m2b0mxmr.js} +9 -9
- package/dist/cli/{gate-evidence-fqxpf3xz.js → gate-evidence-4y2x85g4.js} +3 -3
- package/dist/cli/{guardrail-explain-r0qxa4zy.js → guardrail-explain-nvmf3y7j.js} +24 -24
- package/dist/cli/{guardrail-log-j2wdmvmy.js → guardrail-log-202rfw7b.js} +4 -4
- package/dist/cli/{guardrail-reset-wb2eqbvf.js → guardrail-reset-7nt7z18b.js} +23 -23
- package/dist/cli/{hive-promoter-kxpc3r17.js → hive-promoter-segsykns.js} +23 -23
- package/dist/cli/{index-z2zrcecq.js → index-3qhx9dx0.js} +2500 -1345
- package/dist/cli/{index-3bbjpbmn.js → index-4h54k8h3.js} +2 -2
- package/dist/cli/{index-v7fx8q3z.js → index-908yea8v.js} +1 -1
- package/dist/cli/{index-v0par978.js → index-9j5jjqks.js} +1 -1
- package/dist/cli/{index-n5carmk0.js → index-a7z6jygf.js} +2 -2
- package/dist/cli/{index-bhmrgbyv.js → index-b2zbak7v.js} +6 -2
- package/dist/cli/{index-xj5z92qz.js → index-basp4sxt.js} +1 -1
- package/dist/cli/{index-yf7tdhpz.js → index-cyyvc1cd.js} +1 -1
- package/dist/cli/{index-wx3wz7ya.js → index-edb3sgxg.js} +2 -2
- package/dist/cli/{index-2q5m4m2r.js → index-g5wjasjn.js} +3 -3
- package/dist/cli/{index-rp4akjrm.js → index-g9ax7jyf.js} +15 -4
- package/dist/cli/{index-dzm5nfh7.js → index-gqe4sm73.js} +1 -1
- package/dist/cli/{index-855qxtdm.js → index-gydvqmtg.js} +4 -4
- package/dist/cli/{index-3zvp7t5d.js → index-hxe22tx3.js} +5 -5
- package/dist/cli/{index-tnhk72rs.js → index-jqjdejqe.js} +2 -2
- package/dist/cli/{index-z2xf153n.js → index-kjstkfgh.js} +3 -3
- package/dist/cli/{index-rejpbgxp.js → index-m65tnv4j.js} +1 -1
- package/dist/cli/{index-600aqfvm.js → index-na3br902.js} +23 -23
- package/dist/cli/{index-cj64vp08.js → index-pee6ykkg.js} +1 -1
- package/dist/cli/{index-xd0e4amf.js → index-q1prchye.js} +5 -5
- package/dist/cli/{index-k5vwnnym.js → index-r6p6x0pb.js} +16 -7
- package/dist/cli/{index-k6f0nryj.js → index-rw0wy2y7.js} +3 -3
- package/dist/cli/{index-gw5egd8j.js → index-rxs0rrv1.js} +25 -25
- package/dist/cli/{index-g062wnw1.js → index-tczqm089.js} +1 -1
- package/dist/cli/{index-50v672wd.js → index-vs83acgx.js} +3 -3
- package/dist/cli/{index-w872yykk.js → index-y4zjbrqs.js} +4 -0
- package/dist/cli/{index-gy3tcmk4.js → index-yhx94xpr.js} +4 -4
- package/dist/cli/{index-as4xaams.js → index-ztph1v9m.js} +1 -1
- package/dist/cli/index.js +24 -24
- package/dist/cli/{knowledge-escalator-n30h6ran.js → knowledge-escalator-r9chkas8.js} +7 -7
- package/dist/cli/{knowledge-events-k2qqase2.js → knowledge-events-5j2n6age.js} +5 -5
- package/dist/cli/{knowledge-store-zy226v5a.js → knowledge-store-mybx9bmx.js} +1 -1
- package/dist/cli/{knowledge-validator-qgyxdpc9.js → knowledge-validator-v1sj9tv4.js} +2 -2
- package/dist/cli/{mcp-pj7acze3.js → mcp-7yx5230p.js} +1 -1
- package/dist/cli/{model-preflight-61xgqwcs.js → model-preflight-ckvv1arg.js} +1 -1
- package/dist/cli/{pending-delegations-sgnnyw55.js → pending-delegations-amjfmc7t.js} +2 -2
- package/dist/cli/{pr-subscriptions-7fb7xm3q.js → pr-subscriptions-rctqs0gk.js} +1 -1
- package/dist/cli/{pr-workflow-gate-5dqcrkz6.js → pr-workflow-gate-9he8ys4e.js} +23 -23
- package/dist/cli/{runner-p9hstpxn.js → runner-tzgwvvn1.js} +1 -1
- package/dist/cli/{scan-cursor-jc7wwb7e.js → scan-cursor-pxqkphrh.js} +2 -2
- package/dist/cli/{schema-5k68crjb.js → schema-n8pk3f80.js} +6 -2
- package/dist/cli/{scope-persistence-xeg4khwk.js → scope-persistence-p992x2yd.js} +6 -6
- package/dist/cli/{server-ej6tzd7t.js → server-ywrd2q6z.js} +23 -23
- package/dist/cli/{skill-generator-nspt9b3n.js → skill-generator-cdppktqx.js} +8 -8
- package/dist/cli/{snapshot-coordination-init-3rs3zpaa.js → snapshot-coordination-init-z42ev5g9.js} +23 -23
- package/dist/cli/{speckit-checkoff-dqf636ya.js → speckit-checkoff-6ynv46bn.js} +4 -4
- package/dist/cli/{worktree-collision-ownership-rbzgbvj4.js → worktree-collision-ownership-65971p3m.js} +2 -2
- package/dist/cli/{worktree-isolation-6zpnzr70.js → worktree-isolation-xf7dfc46.js} +23 -23
- package/dist/commands/harness-opt.d.ts +44 -0
- package/dist/commands/registry.d.ts +56 -0
- package/dist/config/schema.d.ts +20 -0
- package/dist/gate-evidence.d.ts +21 -0
- package/dist/harness/store.d.ts +9 -0
- package/dist/index.js +263 -251
- package/dist/services/harness-optimizer/comparative.d.ts +81 -0
- package/dist/services/harness-optimizer/controller.d.ts +126 -0
- package/dist/services/harness-optimizer/execution.d.ts +68 -0
- package/dist/services/harness-optimizer/index.d.ts +10 -0
- package/dist/services/harness-optimizer/lineage.d.ts +70 -0
- package/dist/services/harness-optimizer/manifest.d.ts +65 -0
- package/dist/services/harness-optimizer/oracle.d.ts +40 -0
- package/dist/tools/index.d.ts +1 -0
- package/dist/tools/manifest.d.ts +1 -0
- package/dist/tools/recover-rework-task.d.ts +5 -0
- package/dist/tools/tool-metadata.d.ts +4 -0
- package/dist/workflow/rework-recovery.d.ts +38 -0
- package/dist/workflow/stage-a-repair.d.ts +42 -0
- package/opencode-swarm.schema.json +42 -0
- package/package.json +1 -1
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
export type ComparativeArmKey = 'baseline' | 'ablation' | 'simple-agent';
|
|
2
|
+
export declare const COMPARATIVE_ARM_KEYS: readonly ComparativeArmKey[];
|
|
3
|
+
export interface ComparativeTaskDescriptor {
|
|
4
|
+
id: string;
|
|
5
|
+
instruction: string;
|
|
6
|
+
}
|
|
7
|
+
export type ComparativeExecutorStatus = 'completed' | 'transient_error' | 'failed';
|
|
8
|
+
export interface ComparativeExecutorResult {
|
|
9
|
+
status: ComparativeExecutorStatus;
|
|
10
|
+
text: string;
|
|
11
|
+
tokens?: {
|
|
12
|
+
input?: number;
|
|
13
|
+
cache?: number;
|
|
14
|
+
output?: number;
|
|
15
|
+
};
|
|
16
|
+
/** Host-reported USD cost of this invocation, when supplied. */
|
|
17
|
+
costUsd?: number;
|
|
18
|
+
}
|
|
19
|
+
export type ComparativeExecutor = (invocation: {
|
|
20
|
+
cwd: string;
|
|
21
|
+
/** Protocol arm, or the round-role labels baseline/candidate on substrate rounds. */
|
|
22
|
+
arm: ComparativeArmKey | 'candidate';
|
|
23
|
+
taskId: string;
|
|
24
|
+
/** Host context passthrough (optional; populated on substrate-driven rounds). */
|
|
25
|
+
instruction?: string;
|
|
26
|
+
payload?: string;
|
|
27
|
+
seed?: string;
|
|
28
|
+
abortSignal?: AbortSignal;
|
|
29
|
+
projectRoot?: string;
|
|
30
|
+
}) => Promise<ComparativeExecutorResult> | ComparativeExecutorResult;
|
|
31
|
+
export interface ComparativeArmResult {
|
|
32
|
+
arm: ComparativeArmKey;
|
|
33
|
+
taskPopulationHash: string;
|
|
34
|
+
/** Denominator: every task in the population was attempted. */
|
|
35
|
+
n: number;
|
|
36
|
+
completed: number;
|
|
37
|
+
failed: number;
|
|
38
|
+
transientFailures: number;
|
|
39
|
+
streamDigest: string;
|
|
40
|
+
outcomes: Array<{
|
|
41
|
+
taskId: string;
|
|
42
|
+
status: ComparativeExecutorStatus;
|
|
43
|
+
}>;
|
|
44
|
+
}
|
|
45
|
+
export interface ComparativeProtocolResult {
|
|
46
|
+
arms: Record<ComparativeArmKey, ComparativeArmResult>;
|
|
47
|
+
taskPopulationHash: string;
|
|
48
|
+
}
|
|
49
|
+
export declare class StreamSnapshotDuplicateError extends Error {
|
|
50
|
+
readonly code = "STREAM_SNAPSHOT_DUPLICATE";
|
|
51
|
+
constructor(digest: string);
|
|
52
|
+
}
|
|
53
|
+
export declare function computeTaskPopulationHash(tasks: readonly ComparativeTaskDescriptor[]): string;
|
|
54
|
+
export declare function computeArmStreamDigest(args: {
|
|
55
|
+
arm: ComparativeArmKey;
|
|
56
|
+
seed: string;
|
|
57
|
+
tasks: readonly ComparativeTaskDescriptor[];
|
|
58
|
+
}): string;
|
|
59
|
+
/**
|
|
60
|
+
* Run the three comparative arms. Each arm executes the full task
|
|
61
|
+
* population inside its own disposable git worktree (never the running
|
|
62
|
+
* checkout), with the working-tree fingerprint verified unchanged by
|
|
63
|
+
* withDisposableWorktree. Failed arms stay in the result with their
|
|
64
|
+
* outcomes — negative results are evidence, not noise.
|
|
65
|
+
*/
|
|
66
|
+
export declare function runComparativeProtocol(args: {
|
|
67
|
+
projectRoot: string;
|
|
68
|
+
tasks: ComparativeTaskDescriptor[];
|
|
69
|
+
seed: string;
|
|
70
|
+
executor: ComparativeExecutor;
|
|
71
|
+
/** Previously-recorded digests for this task population; a repeat rejects. */
|
|
72
|
+
previousStreamDigests?: string[];
|
|
73
|
+
/**
|
|
74
|
+
* Arm toggles (harness_opt.run_ablation_arm / run_simple_agent_arm).
|
|
75
|
+
* The baseline arm always runs; unlisted arms are skipped and absent
|
|
76
|
+
* from the result. Defaults to all three arms.
|
|
77
|
+
*/
|
|
78
|
+
runAblationArm?: boolean;
|
|
79
|
+
runSimpleAgentArm?: boolean;
|
|
80
|
+
abortSignal?: AbortSignal;
|
|
81
|
+
}): Promise<ComparativeProtocolResult>;
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
import type { ComparativeExecutor, ComparativeTaskDescriptor } from './comparative.js';
|
|
2
|
+
export type HarnessOptSplit = 'train' | 'validation' | 'test';
|
|
3
|
+
/**
|
|
4
|
+
* Stop reasons the round controller can actually produce. Every member has
|
|
5
|
+
* a producing code path (pinned by tests/unit/harness-opt/stop-reasons.test.ts
|
|
6
|
+
* and the controller suite); conditions that surface as typed ERRORS rather
|
|
7
|
+
* than stop results (held-out consumption, frozen-content mismatch, stream
|
|
8
|
+
* snapshot duplication, replay mismatch) are intentionally NOT members —
|
|
9
|
+
* they propagate as their substrate/module errors.
|
|
10
|
+
*/
|
|
11
|
+
export declare const HARNESS_OPT_STOP_REASONS: readonly ["completed", "inconclusive", "transient_retry_budget_exhausted", "round_budget_exhausted", "wall_clock_budget_exhausted", "spend_budget_exhausted", "stopped_by_operator"];
|
|
12
|
+
export type HarnessOptStopReason = (typeof HARNESS_OPT_STOP_REASONS)[number];
|
|
13
|
+
export interface HarnessOptRoundResult {
|
|
14
|
+
stopReason: HarnessOptStopReason;
|
|
15
|
+
transientRetries: number;
|
|
16
|
+
roundId: string;
|
|
17
|
+
roundCounter: number;
|
|
18
|
+
saltedSeed: string;
|
|
19
|
+
taskSetHash: string;
|
|
20
|
+
decision: {
|
|
21
|
+
status: string;
|
|
22
|
+
decisionId: string;
|
|
23
|
+
};
|
|
24
|
+
}
|
|
25
|
+
export declare class FrozenTaskSetMismatchError extends Error {
|
|
26
|
+
readonly code = "FROZEN_CONTENT_MISMATCH";
|
|
27
|
+
constructor(key: string);
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* Content-freeze the task set BEFORE any round. Re-freezing the same
|
|
31
|
+
* {split, seed} key with mutated task content is rejected (typed
|
|
32
|
+
* FROZEN_CONTENT_MISMATCH); identical content is idempotent. The frozen
|
|
33
|
+
* input lives content-addressed under .swarm/evolution/harness-opt/ and
|
|
34
|
+
* the substrate's immutable admission backs the freeze for non-train
|
|
35
|
+
* splits.
|
|
36
|
+
*/
|
|
37
|
+
export declare function freezeHarnessOptTaskSet(args: {
|
|
38
|
+
projectRoot: string;
|
|
39
|
+
tasks: ComparativeTaskDescriptor[];
|
|
40
|
+
split: HarnessOptSplit;
|
|
41
|
+
seed: string;
|
|
42
|
+
}): Promise<{
|
|
43
|
+
ok: true;
|
|
44
|
+
contentHash: string;
|
|
45
|
+
} | {
|
|
46
|
+
ok: false;
|
|
47
|
+
code: 'FROZEN_CONTENT_MISMATCH';
|
|
48
|
+
reason: string;
|
|
49
|
+
}>;
|
|
50
|
+
/**
|
|
51
|
+
* Execute ONE governed optimization round. The per-round seed salt
|
|
52
|
+
* `${seed}#roundNNNNNN` guarantees a fresh substrate runId per round, so
|
|
53
|
+
* the SUBSTRATE's claimHeldOutTest is what surfaces
|
|
54
|
+
* TestAlreadyConsumedError on a second held-out round over the same
|
|
55
|
+
* frozen task set — the controller never masks consumption with its own
|
|
56
|
+
* counter.
|
|
57
|
+
*/
|
|
58
|
+
export declare function runHarnessOptRound(args: {
|
|
59
|
+
projectRoot: string;
|
|
60
|
+
tasks: ComparativeTaskDescriptor[];
|
|
61
|
+
split: HarnessOptSplit;
|
|
62
|
+
seed: string;
|
|
63
|
+
maxTransientRetries?: number;
|
|
64
|
+
/** Hard wall-clock budget for the round, in milliseconds. */
|
|
65
|
+
maxWallClockMs?: number;
|
|
66
|
+
/** Soft spend budget for the round, in USD. */
|
|
67
|
+
maxSpendUsd?: number;
|
|
68
|
+
executor: ComparativeExecutor;
|
|
69
|
+
abortSignal?: AbortSignal;
|
|
70
|
+
}): Promise<HarnessOptRoundResult>;
|
|
71
|
+
export interface PilotGraduationRecord {
|
|
72
|
+
v: 1;
|
|
73
|
+
recordId: string;
|
|
74
|
+
eligible: boolean;
|
|
75
|
+
evaluatedAt: string;
|
|
76
|
+
criteria: {
|
|
77
|
+
lowerCiThreshold: number;
|
|
78
|
+
};
|
|
79
|
+
evidence: {
|
|
80
|
+
improvementLowerCi: number;
|
|
81
|
+
protectedRegressions: string[];
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
/**
|
|
85
|
+
* Evaluate pilot graduation against the predeclared criteria. The evidence
|
|
86
|
+
* — including failing (negative) evidence — is retained verbatim in the
|
|
87
|
+
* durable record. Default flipping stays owned by #2504; this surface only
|
|
88
|
+
* reports eligibility.
|
|
89
|
+
*/
|
|
90
|
+
export declare function evaluatePilotGraduation(args: {
|
|
91
|
+
projectRoot: string;
|
|
92
|
+
criteria: {
|
|
93
|
+
lowerCiThreshold: number;
|
|
94
|
+
};
|
|
95
|
+
evidence: {
|
|
96
|
+
improvementLowerCi: number;
|
|
97
|
+
protectedRegressions: string[];
|
|
98
|
+
};
|
|
99
|
+
}): Promise<{
|
|
100
|
+
eligible: boolean;
|
|
101
|
+
recordId: string;
|
|
102
|
+
}>;
|
|
103
|
+
export declare function loadPilotGraduationRecord(projectRoot: string, recordId: string): PilotGraduationRecord | null;
|
|
104
|
+
/** Human-only operator stop: halts further rounds until explicitly resumed
|
|
105
|
+
* via resumeHarnessOptLoop (harness-opt stop --resume). */
|
|
106
|
+
export declare function stopHarnessOptLoop(args: {
|
|
107
|
+
projectRoot: string;
|
|
108
|
+
reason: string;
|
|
109
|
+
}): Promise<{
|
|
110
|
+
stopped: true;
|
|
111
|
+
reason: string;
|
|
112
|
+
}>;
|
|
113
|
+
/** Human-only operator resume: clears an operator stop so rounds run again. */
|
|
114
|
+
export declare function resumeHarnessOptLoop(args: {
|
|
115
|
+
projectRoot: string;
|
|
116
|
+
}): Promise<{
|
|
117
|
+
resumed: boolean;
|
|
118
|
+
previousReason: string | null;
|
|
119
|
+
}>;
|
|
120
|
+
export interface HarnessOptStatus {
|
|
121
|
+
roundCounter: number;
|
|
122
|
+
stopped: boolean;
|
|
123
|
+
stopReason: string | null;
|
|
124
|
+
frozenTaskSets: number;
|
|
125
|
+
}
|
|
126
|
+
export declare function harnessOptStatus(projectRoot: string): HarnessOptStatus;
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import { type ComparativeExecutor, type ComparativeTaskDescriptor } from './comparative.js';
|
|
2
|
+
export interface HarnessOptTaskSetDescriptor {
|
|
3
|
+
split: 'train' | 'validation' | 'test';
|
|
4
|
+
seed: string;
|
|
5
|
+
tasks: ComparativeTaskDescriptor[];
|
|
6
|
+
}
|
|
7
|
+
export interface TokenUsage {
|
|
8
|
+
tokens_input: number | 'unknown';
|
|
9
|
+
tokens_cache: number | 'unknown';
|
|
10
|
+
tokens_output: number | 'unknown';
|
|
11
|
+
}
|
|
12
|
+
export declare function harnessOptTaskSetRoot(projectRoot: string, taskSetHash: string): string;
|
|
13
|
+
/**
|
|
14
|
+
* Materialize the task-set input root (content-addressed by the task
|
|
15
|
+
* population hash). Existing materialization is left untouched — the
|
|
16
|
+
* content hash pins it; a mutation is detected by re-hashing before use.
|
|
17
|
+
*/
|
|
18
|
+
export declare function materializeHarnessOptInput(args: {
|
|
19
|
+
projectRoot: string;
|
|
20
|
+
descriptor: HarnessOptTaskSetDescriptor;
|
|
21
|
+
}): {
|
|
22
|
+
inputRoot: string;
|
|
23
|
+
taskSetHash: string;
|
|
24
|
+
};
|
|
25
|
+
export declare class HarnessOptInputTamperedError extends Error {
|
|
26
|
+
readonly code = "HARNESS_OPT_INPUT_TAMPERED";
|
|
27
|
+
constructor(taskSetHash: string);
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* Verify the MATERIALIZED ON-DISK input still matches its content-addressed
|
|
31
|
+
* hash: re-reads every task instruction from the content-addressed input
|
|
32
|
+
* root and recomputes the task-population hash from the bytes on disk. A
|
|
33
|
+
* tampered materialization fails typed. (Issue #2503 integrity check.)
|
|
34
|
+
*/
|
|
35
|
+
export declare function verifyHarnessOptInput(args: {
|
|
36
|
+
inputRoot: string;
|
|
37
|
+
taskIds: readonly string[];
|
|
38
|
+
taskSetHash: string;
|
|
39
|
+
}): boolean;
|
|
40
|
+
export interface SubstrateExecutionResult {
|
|
41
|
+
runStatus: string;
|
|
42
|
+
decisionStatus: string;
|
|
43
|
+
decisionId: string;
|
|
44
|
+
/** Host-reported USD spend accumulated across executor invocations. */
|
|
45
|
+
reportedSpendUsd: number;
|
|
46
|
+
outcomes: Array<{
|
|
47
|
+
candidateId: string;
|
|
48
|
+
outcome: string;
|
|
49
|
+
}>;
|
|
50
|
+
tokens: TokenUsage;
|
|
51
|
+
}
|
|
52
|
+
/**
|
|
53
|
+
* Execute one governed evaluation round through the production substrate.
|
|
54
|
+
* The simplified executor runs inside the substrate's disposable worktree
|
|
55
|
+
* (isolatedRoot — never the running checkout) and token usage reported by
|
|
56
|
+
* the host executor is captured; missing host data stays 'unknown'.
|
|
57
|
+
*/
|
|
58
|
+
export declare function runSubstrateEvaluation(args: {
|
|
59
|
+
projectRoot: string;
|
|
60
|
+
inputRoot: string;
|
|
61
|
+
descriptor: HarnessOptTaskSetDescriptor;
|
|
62
|
+
seed: string;
|
|
63
|
+
decidedAt: string;
|
|
64
|
+
maxTransientRetries?: number;
|
|
65
|
+
maxSpendUsd?: number;
|
|
66
|
+
executor: ComparativeExecutor;
|
|
67
|
+
abortSignal?: AbortSignal;
|
|
68
|
+
}): Promise<SubstrateExecutionResult>;
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
export type { ComparativeArmKey, ComparativeArmResult, ComparativeExecutor, ComparativeExecutorResult, ComparativeProtocolResult, ComparativeTaskDescriptor, } from './comparative.js';
|
|
2
|
+
export { COMPARATIVE_ARM_KEYS, computeArmStreamDigest, computeTaskPopulationHash, runComparativeProtocol, StreamSnapshotDuplicateError, } from './comparative.js';
|
|
3
|
+
export type { HarnessOptRoundResult, HarnessOptSplit, HarnessOptStatus, PilotGraduationRecord, } from './controller.js';
|
|
4
|
+
export { evaluatePilotGraduation, FrozenTaskSetMismatchError, freezeHarnessOptTaskSet, harnessOptStatus, loadPilotGraduationRecord, runHarnessOptRound, stopHarnessOptLoop, } from './controller.js';
|
|
5
|
+
export type { HarnessOptLineageRecord } from './lineage.js';
|
|
6
|
+
export { computeCandidateConfigDigest, computePromptSelectionDigest, listHarnessOptLineage, ReplayDecisionMismatchError, recordHarnessOptRound, replayHarnessOptLineage, } from './lineage.js';
|
|
7
|
+
export type { ComparativeManifestV1, ManifestValidationFailureCode, ManifestValidationResult, } from './manifest.js';
|
|
8
|
+
export { validateComparativeManifest } from './manifest.js';
|
|
9
|
+
export type { OracleArmInput, OracleVerdict, TokenCount } from './oracle.js';
|
|
10
|
+
export { evaluateIndependentOracle } from './oracle.js';
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import type { ComparativeTaskDescriptor } from './comparative.js';
|
|
2
|
+
import type { TokenUsage } from './execution.js';
|
|
3
|
+
export interface HarnessOptLineageRecord {
|
|
4
|
+
v: 1;
|
|
5
|
+
roundId: string;
|
|
6
|
+
/** Identical to roundId: one identity string, no derived second id. */
|
|
7
|
+
replayLineageId: string;
|
|
8
|
+
recordedAt: string;
|
|
9
|
+
split: 'train' | 'validation' | 'test';
|
|
10
|
+
baseSeed: string;
|
|
11
|
+
saltedSeed: string;
|
|
12
|
+
roundCounter: number;
|
|
13
|
+
taskSetHash: string;
|
|
14
|
+
candidateConfigDigest: string;
|
|
15
|
+
promptSelectionDigest: string;
|
|
16
|
+
tokens_input: number | 'unknown';
|
|
17
|
+
tokens_cache: number | 'unknown';
|
|
18
|
+
tokens_output: number | 'unknown';
|
|
19
|
+
artifactOutcome: string;
|
|
20
|
+
oracle: {
|
|
21
|
+
verdict: string;
|
|
22
|
+
reasons: string[];
|
|
23
|
+
};
|
|
24
|
+
decision: {
|
|
25
|
+
status: string;
|
|
26
|
+
decisionId: string;
|
|
27
|
+
};
|
|
28
|
+
execution: {
|
|
29
|
+
decidedAt: string;
|
|
30
|
+
maxTransientRetries?: number;
|
|
31
|
+
tasks: ComparativeTaskDescriptor[];
|
|
32
|
+
};
|
|
33
|
+
}
|
|
34
|
+
export declare class ReplayDecisionMismatchError extends Error {
|
|
35
|
+
readonly code = "REPLAY_DECISION_MISMATCH";
|
|
36
|
+
constructor(recorded: string, replayed: string);
|
|
37
|
+
}
|
|
38
|
+
export declare function recordHarnessOptRound(args: {
|
|
39
|
+
projectRoot: string;
|
|
40
|
+
record: HarnessOptLineageRecord;
|
|
41
|
+
}): Promise<void>;
|
|
42
|
+
export declare function listHarnessOptLineage(projectRoot: string): HarnessOptLineageRecord[];
|
|
43
|
+
export declare function loadHarnessOptLineageRecord(args: {
|
|
44
|
+
projectRoot: string;
|
|
45
|
+
roundId: string;
|
|
46
|
+
}): HarnessOptLineageRecord | null;
|
|
47
|
+
export declare function computeCandidateConfigDigest(args: {
|
|
48
|
+
saltedSeed: string;
|
|
49
|
+
split: string;
|
|
50
|
+
maxTransientRetries?: number;
|
|
51
|
+
}): string;
|
|
52
|
+
export declare function computePromptSelectionDigest(tasks: readonly ComparativeTaskDescriptor[]): string;
|
|
53
|
+
/**
|
|
54
|
+
* Replay a recorded round. The recorded fully-salted seed and decidedAt are
|
|
55
|
+
* re-used WITHOUT reading or advancing the round counter, so the substrate
|
|
56
|
+
* derives the identical runId and returns the immutable cached run. The
|
|
57
|
+
* replay executor throws if ever invoked — a replay must hit the cached
|
|
58
|
+
* record, never re-spend execution. Any decision mismatch (a tampered or
|
|
59
|
+
* stale record) fails typed.
|
|
60
|
+
*/
|
|
61
|
+
export declare function replayHarnessOptLineage(args: {
|
|
62
|
+
projectRoot: string;
|
|
63
|
+
lineageId: string;
|
|
64
|
+
}): Promise<{
|
|
65
|
+
decision: {
|
|
66
|
+
status: string;
|
|
67
|
+
decisionId: string;
|
|
68
|
+
};
|
|
69
|
+
}>;
|
|
70
|
+
export type { TokenUsage };
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Frozen comparative-evaluation manifest (issue #2503).
|
|
3
|
+
*
|
|
4
|
+
* A comparative run MUST be governed by a manifest frozen BEFORE any
|
|
5
|
+
* optimization: a named previous stable release, the repaired unoptimized
|
|
6
|
+
* baseline reference, the frozen corpus/task population, exact PR revisions
|
|
7
|
+
* with defect/clean labels, excluded training cases, matched budgets,
|
|
8
|
+
* per-arm sample sizes, numeric acceptance thresholds, and metric
|
|
9
|
+
* definitions. The validator refuses empty required fields and improvement
|
|
10
|
+
* claims without a measured result — the issue's "no empty manifest fields
|
|
11
|
+
* or unmeasured improvement claims" contract.
|
|
12
|
+
*/
|
|
13
|
+
export type ManifestValidationFailureCode = 'MANIFEST_MALFORMED' | 'MANIFEST_FIELD_EMPTY' | 'CLAIM_UNMEASURED';
|
|
14
|
+
export interface ComparativeManifestV1 {
|
|
15
|
+
releaseName: string;
|
|
16
|
+
repairedBaselineRef?: string;
|
|
17
|
+
corpus?: {
|
|
18
|
+
taskPopulationPointer: string;
|
|
19
|
+
excludedTrainingCases: Array<{
|
|
20
|
+
caseId: string;
|
|
21
|
+
reason: string;
|
|
22
|
+
}>;
|
|
23
|
+
};
|
|
24
|
+
prRevisions?: Array<{
|
|
25
|
+
revision: string;
|
|
26
|
+
label: 'defect' | 'clean';
|
|
27
|
+
}>;
|
|
28
|
+
matchedBudgets?: {
|
|
29
|
+
model: string;
|
|
30
|
+
provider: string;
|
|
31
|
+
tool: string;
|
|
32
|
+
timeBudgetMs?: number;
|
|
33
|
+
};
|
|
34
|
+
sampleSizes: {
|
|
35
|
+
baseline: number;
|
|
36
|
+
ablation: number;
|
|
37
|
+
'simple-agent': number;
|
|
38
|
+
};
|
|
39
|
+
thresholds: Record<string, number>;
|
|
40
|
+
metricDefinitions?: Array<{
|
|
41
|
+
metric: string;
|
|
42
|
+
definition: string;
|
|
43
|
+
}>;
|
|
44
|
+
defectLabels: string[];
|
|
45
|
+
cleanLabels: string[];
|
|
46
|
+
claims: Array<{
|
|
47
|
+
kind: 'improvement' | 'negative';
|
|
48
|
+
metric: string;
|
|
49
|
+
measuredResult?: Record<string, number>;
|
|
50
|
+
}>;
|
|
51
|
+
}
|
|
52
|
+
export type ManifestValidationResult = {
|
|
53
|
+
ok: true;
|
|
54
|
+
manifest: ComparativeManifestV1;
|
|
55
|
+
} | {
|
|
56
|
+
ok: false;
|
|
57
|
+
code: ManifestValidationFailureCode;
|
|
58
|
+
reason: string;
|
|
59
|
+
};
|
|
60
|
+
/**
|
|
61
|
+
* Validate a frozen comparative manifest. Accepts only fully-populated
|
|
62
|
+
* manifests; every rejection names the offending field. Unknown/malformed
|
|
63
|
+
* input fails closed with MANIFEST_MALFORMED rather than being coerced.
|
|
64
|
+
*/
|
|
65
|
+
export declare function validateComparativeManifest(input: unknown): ManifestValidationResult;
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Independent oracle for the governed HarnessOpt capstone (issue #2503).
|
|
3
|
+
*
|
|
4
|
+
* Scores accepted task outcomes, artifact validity, verification evidence,
|
|
5
|
+
* and completion quality SEPARATELY from the optimizer and from the task's
|
|
6
|
+
* own scorer (src/evaluation/runner.ts scoreExecution). The oracle is the
|
|
7
|
+
* acceptance backstop the issue demands: a candidate whose token count
|
|
8
|
+
* improves while accepted artifact quality or verification evidence falls
|
|
9
|
+
* MUST be rejected, with reasons naming each drop.
|
|
10
|
+
*/
|
|
11
|
+
export type TokenCount = number | 'unknown';
|
|
12
|
+
export interface OracleArmInput {
|
|
13
|
+
/** Task-population denominator for the arm. */
|
|
14
|
+
n: number;
|
|
15
|
+
/** Count of accepted artifacts the arm produced. */
|
|
16
|
+
acceptedArtifacts: number;
|
|
17
|
+
/** Count of verification-evidence records backing those artifacts. */
|
|
18
|
+
verificationEvidence: number;
|
|
19
|
+
tokens: {
|
|
20
|
+
input: TokenCount;
|
|
21
|
+
cache: TokenCount;
|
|
22
|
+
output: TokenCount;
|
|
23
|
+
};
|
|
24
|
+
}
|
|
25
|
+
export interface OracleVerdict {
|
|
26
|
+
verdict: 'accept' | 'reject';
|
|
27
|
+
reasons: string[];
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* Compare a candidate arm against the baseline arm. Acceptance requires
|
|
31
|
+
* the candidate to be no worse on accepted artifacts, verification
|
|
32
|
+
* evidence, or (when both arms report tokens) token usage. Any quality or
|
|
33
|
+
* verification drop rejects the candidate regardless of token improvement.
|
|
34
|
+
* Token totals reported as 'unknown' by the host are treated as unknown,
|
|
35
|
+
* never zero, and therefore never counted as an improvement axis.
|
|
36
|
+
*/
|
|
37
|
+
export declare function evaluateIndependentOracle(args: {
|
|
38
|
+
baseline: OracleArmInput;
|
|
39
|
+
candidate: OracleArmInput;
|
|
40
|
+
}): Promise<OracleVerdict>;
|
package/dist/tools/index.d.ts
CHANGED
|
@@ -123,6 +123,7 @@ export { executeRecordImplementationReview, record_implementation_review, } from
|
|
|
123
123
|
export { executeRecordIssuePublication, record_issue_publication, } from './record-issue-publication';
|
|
124
124
|
export { executeRecordIssueReproduction, record_issue_reproduction, } from './record-issue-reproduction';
|
|
125
125
|
export { executeRecordRecurrenceSweep, record_recurrence_sweep, } from './record-recurrence-sweep';
|
|
126
|
+
export { executeRecoverReworkTask, recover_rework_task, } from './recover-rework-task';
|
|
126
127
|
export { executeRunPrFeedbackStageA, run_pr_feedback_stage_a, } from './run-pr-feedback-stage-a';
|
|
127
128
|
export { symbols } from './symbols';
|
|
128
129
|
export { type SyntaxCheckFileResult, type SyntaxCheckInput, type SyntaxCheckResult, syntax_check, syntaxCheck, } from './syntax-check';
|
package/dist/tools/manifest.d.ts
CHANGED
|
@@ -48,6 +48,7 @@ export declare const TOOL_MANIFEST: {
|
|
|
48
48
|
submit_pr_review_result: () => ToolDefinition;
|
|
49
49
|
approve_plan_critic: () => ToolDefinition;
|
|
50
50
|
approve_retry_sounding_board: () => ToolDefinition;
|
|
51
|
+
recover_rework_task: () => ToolDefinition;
|
|
51
52
|
prepare_pr_workflow_checkout: () => ToolDefinition;
|
|
52
53
|
record_implementation_review: () => ToolDefinition;
|
|
53
54
|
record_issue_publication: () => ToolDefinition;
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
import { createSwarmTool } from './create-tool.js';
|
|
2
|
+
export declare function executeRecoverReworkTask(args: unknown, directory: string, context?: {
|
|
3
|
+
sessionID?: string;
|
|
4
|
+
}): Promise<string>;
|
|
5
|
+
export declare const recover_rework_task: ReturnType<typeof createSwarmTool>;
|
|
@@ -177,6 +177,10 @@ export declare const TOOL_METADATA: {
|
|
|
177
177
|
description: string;
|
|
178
178
|
agents: "architect"[];
|
|
179
179
|
};
|
|
180
|
+
recover_rework_task: {
|
|
181
|
+
description: string;
|
|
182
|
+
agents: "architect"[];
|
|
183
|
+
};
|
|
180
184
|
prepare_pr_workflow_checkout: {
|
|
181
185
|
description: string;
|
|
182
186
|
agents: "architect"[];
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
export interface ReworkRecoverySummary {
|
|
2
|
+
taskId: string;
|
|
3
|
+
generation: number;
|
|
4
|
+
state: string;
|
|
5
|
+
transitionId: string;
|
|
6
|
+
recordedAt: string;
|
|
7
|
+
/**
|
|
8
|
+
* Whether the stage_a_repair audit event actually landed in
|
|
9
|
+
* `.swarm/events.jsonl`. The durable transition is authoritative regardless;
|
|
10
|
+
* this surfaces the best-effort append honestly instead of asserting an
|
|
11
|
+
* audit record that may not exist (issue #2755 review, FB-001).
|
|
12
|
+
*/
|
|
13
|
+
auditEventRecorded: boolean;
|
|
14
|
+
}
|
|
15
|
+
/**
|
|
16
|
+
* Architect-supervised recovery from `rework_required` (issue #2755).
|
|
17
|
+
*
|
|
18
|
+
* The mechanical `stage_a_passed` transition deliberately fails closed from
|
|
19
|
+
* `rework_required` (`TASK_WORKFLOW_CODER_MUTATION_REQUIRED`): a genuine code
|
|
20
|
+
* defect must go back through the coder. But a Stage B verdict that failed
|
|
21
|
+
* WITHOUT a code defect (e.g. a tool-argument mistake scored as a verdict —
|
|
22
|
+
* sibling issue #2756) strands a task whose code is correct, reviewed, and
|
|
23
|
+
* green on Stage A checks, with no agent-legal exit; the only other remedy
|
|
24
|
+
* ever offered is the human-only `/swarm recover`, whose repair leg skips
|
|
25
|
+
* this state entirely.
|
|
26
|
+
*
|
|
27
|
+
* This is the audited escape hatch the #2703 precedent
|
|
28
|
+
* (`approve_retry_sounding_board`) established for exactly this class: it
|
|
29
|
+
* writes the SAME durable transition the mechanical path would have written,
|
|
30
|
+
* admitted ONLY by the `supervisedRecovery` flag the reducer accepts from
|
|
31
|
+
* `rework_required`, and records a distinguishable `stage_a_repair` audit
|
|
32
|
+
* event (action `rework_recovered`). Every precondition fails closed; the
|
|
33
|
+
* mechanical path and the `/swarm recover` auto-scan are untouched.
|
|
34
|
+
*/
|
|
35
|
+
export declare function forceRecoverReworkTask(directory: string, sessionID: string, options: {
|
|
36
|
+
taskId: string;
|
|
37
|
+
reason?: string;
|
|
38
|
+
}): Promise<ReworkRecoverySummary>;
|
|
@@ -21,12 +21,54 @@ export interface StageARepairResult {
|
|
|
21
21
|
results: StageARepairOutcome[];
|
|
22
22
|
truncated: boolean;
|
|
23
23
|
}
|
|
24
|
+
/**
|
|
25
|
+
* Appends a stage_a_repair lifecycle event to `.swarm/events.jsonl`.
|
|
26
|
+
* Mirrors coder-settlement's appendSettlementEvent contract: best-effort,
|
|
27
|
+
* never throws, appended through the canonical `appendCoreEventSync` seam
|
|
28
|
+
* (which owns `.swarm` creation, lock retry, and torn-tail framing — its
|
|
29
|
+
* single atomic append replaces the former EBUSY/EPERM one-retry), final
|
|
30
|
+
* failure surfaced via criticalWarn so a silently missing audit line is
|
|
31
|
+
* visible. Returns whether the audit line actually landed, so callers on
|
|
32
|
+
* supervised-write paths can surface audit honesty instead of asserting it
|
|
33
|
+
* (issue #2755 review: the recover_rework_task tool must not claim an audit
|
|
34
|
+
* record exists when the append failed).
|
|
35
|
+
*/
|
|
36
|
+
export declare function appendStageARepairEvent(directory: string, payload: Record<string, unknown>): Promise<boolean>;
|
|
24
37
|
export type PreCheckGreennessResult = {
|
|
25
38
|
green: true;
|
|
26
39
|
} | {
|
|
27
40
|
green: false;
|
|
28
41
|
reason: 'no_pre_check_bundles' | 'pre_check_failed_or_stale';
|
|
29
42
|
};
|
|
43
|
+
/**
|
|
44
|
+
* Decides whether durable post-settlement Stage A proof exists for a wedged
|
|
45
|
+
* task. "Green" REQUIRES BOTH a secretscan evidence bundle whose latest entry
|
|
46
|
+
* is pass/approved/info with full coverage and zero findings, AND a sast_scan
|
|
47
|
+
* evidence bundle whose latest entry is pass/approved/info. Also requires
|
|
48
|
+
* every considered entry to be newer than the settlement commit when
|
|
49
|
+
* `settledAfterMs` is supplied (a scan taken before the coder's mutation
|
|
50
|
+
* proves nothing about it).
|
|
51
|
+
*
|
|
52
|
+
* Deliberately more conservative than `pre_check_batch`'s own default bar in
|
|
53
|
+
* two edge cases it cannot reproduce from persisted evidence alone: (1) a
|
|
54
|
+
* degraded-but-tolerated SAST run (Semgrep process failure with zero
|
|
55
|
+
* findings) persists an ordinary `verdict: 'fail'` entry structurally
|
|
56
|
+
* indistinguishable from a genuine failure — `failure_kind` is only present
|
|
57
|
+
* on the tool's transient return value, not the persisted bundle — so this
|
|
58
|
+
* function treats it as failing rather than risk silently waving through a
|
|
59
|
+
* scan that never actually completed; (2) a project running with SAST
|
|
60
|
+
* disabled (`sast_enabled: false`) never persists a `sast_scan` bundle at
|
|
61
|
+
* all, so a wedged task from such a project cannot be auto-repaired via this
|
|
62
|
+
* path and needs manual attention. Both are intentional fail-closed
|
|
63
|
+
* trade-offs for a security-relevant repair tool, not bugs — see PR #2316
|
|
64
|
+
* review finding ST-001/UIB-004 for why an absent-SAST-is-fine policy was
|
|
65
|
+
* removed in the first place.
|
|
66
|
+
*
|
|
67
|
+
* Evidence bucket names vs. entry type tags differ for SAST: the scanner
|
|
68
|
+
* persists its bundle under bucket `sast_scan` (see `src/tools/sast-scan.ts`)
|
|
69
|
+
* with individual entries tagged `type: 'sast'`.
|
|
70
|
+
*/
|
|
71
|
+
export declare function hasGreenPostSettlementPreCheck(directory: string, settledAfterMs: number | null): Promise<PreCheckGreennessResult>;
|
|
30
72
|
export interface StageAScanResult {
|
|
31
73
|
/** One entry per enumerated task, classified in the shared #2665 vocabulary. */
|
|
32
74
|
results: TaskRecoveryStatus[];
|
|
@@ -3769,6 +3769,48 @@
|
|
|
3769
3769
|
},
|
|
3770
3770
|
"additionalProperties": false
|
|
3771
3771
|
},
|
|
3772
|
+
"harness_opt": {
|
|
3773
|
+
"description": "Governed HarnessOpt optimization capstone (issue #2503). Disabled by default; /swarm harness-opt run requires enabled: true plus --confirm.",
|
|
3774
|
+
"type": "object",
|
|
3775
|
+
"properties": {
|
|
3776
|
+
"enabled": {
|
|
3777
|
+
"default": false,
|
|
3778
|
+
"type": "boolean"
|
|
3779
|
+
},
|
|
3780
|
+
"max_rounds": {
|
|
3781
|
+
"default": 5,
|
|
3782
|
+
"type": "integer",
|
|
3783
|
+
"minimum": 1,
|
|
3784
|
+
"maximum": 50
|
|
3785
|
+
},
|
|
3786
|
+
"max_transient_retries": {
|
|
3787
|
+
"default": 2,
|
|
3788
|
+
"type": "integer",
|
|
3789
|
+
"minimum": 0,
|
|
3790
|
+
"maximum": 10
|
|
3791
|
+
},
|
|
3792
|
+
"max_wall_clock_ms": {
|
|
3793
|
+
"default": 3600000,
|
|
3794
|
+
"type": "integer",
|
|
3795
|
+
"minimum": 10000,
|
|
3796
|
+
"maximum": 86400000
|
|
3797
|
+
},
|
|
3798
|
+
"max_spend_usd": {
|
|
3799
|
+
"type": "number",
|
|
3800
|
+
"minimum": 0,
|
|
3801
|
+
"maximum": 1000
|
|
3802
|
+
},
|
|
3803
|
+
"run_ablation_arm": {
|
|
3804
|
+
"default": true,
|
|
3805
|
+
"type": "boolean"
|
|
3806
|
+
},
|
|
3807
|
+
"run_simple_agent_arm": {
|
|
3808
|
+
"default": true,
|
|
3809
|
+
"type": "boolean"
|
|
3810
|
+
}
|
|
3811
|
+
},
|
|
3812
|
+
"additionalProperties": false
|
|
3813
|
+
},
|
|
3772
3814
|
"spec_writer": {
|
|
3773
3815
|
"description": "Spec writer agent (v2) — independent model for .swarm/spec.md authorship.",
|
|
3774
3816
|
"type": "object",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "opencode-swarm",
|
|
3
|
-
"version": "7.
|
|
3
|
+
"version": "7.181.1",
|
|
4
4
|
"description": "Architect-centric agentic swarm plugin for OpenCode - hub-and-spoke orchestration with SME consultation, code generation, and QA review",
|
|
5
5
|
"main": "dist/index.js",
|
|
6
6
|
"types": "dist/index.d.ts",
|