opencode-swarm 7.180.1 → 7.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/README.md +14 -0
  2. package/binaries/win32-arm64/swarm-sandbox-runner.exe +0 -0
  3. package/binaries/win32-x64/swarm-sandbox-runner.exe +0 -0
  4. package/dist/cli/{coder-settlement-b57917a3.js → coder-settlement-pqrfkm9d.js} +3 -3
  5. package/dist/cli/{config-doctor-ebat06eq.js → config-doctor-bjqg57jr.js} +2 -2
  6. package/dist/cli/{core-8k41qh0d.js → core-3t3dzkrc.js} +1 -1
  7. package/dist/cli/{curator-drift-3av9hffm.js → curator-drift-gay4fx5j.js} +4 -4
  8. package/dist/cli/{curator-llm-factory-htgqafq0.js → curator-llm-factory-1wx0bd8k.js} +22 -22
  9. package/dist/cli/{curator-8yb3qkxs.js → curator-y8qr4xgh.js} +22 -22
  10. package/dist/cli/{evidence-summary-service-s258ea5t.js → evidence-summary-service-jeewwfpg.js} +8 -8
  11. package/dist/cli/{gate-evidence-fqxpf3xz.js → gate-evidence-ncnfxz7t.js} +2 -2
  12. package/dist/cli/{guardrail-explain-r0qxa4zy.js → guardrail-explain-cg9e3rr8.js} +23 -23
  13. package/dist/cli/{guardrail-log-j2wdmvmy.js → guardrail-log-tjcnpwhv.js} +3 -3
  14. package/dist/cli/{guardrail-reset-wb2eqbvf.js → guardrail-reset-ey81kcdq.js} +22 -22
  15. package/dist/cli/{hive-promoter-kxpc3r17.js → hive-promoter-ghcynvfv.js} +22 -22
  16. package/dist/cli/{index-v7fx8q3z.js → index-06f6h0fv.js} +1 -1
  17. package/dist/cli/{index-xj5z92qz.js → index-0vvpe98m.js} +1 -1
  18. package/dist/cli/{index-gw5egd8j.js → index-1rcwsfwz.js} +24 -24
  19. package/dist/cli/{index-cj64vp08.js → index-6aj0ej18.js} +1 -1
  20. package/dist/cli/{index-yf7tdhpz.js → index-886rpw6w.js} +1 -1
  21. package/dist/cli/{index-xd0e4amf.js → index-dn0wqvv2.js} +5 -5
  22. package/dist/cli/{index-wx3wz7ya.js → index-ge1w25n8.js} +2 -2
  23. package/dist/cli/{index-bhmrgbyv.js → index-gez80bfr.js} +5 -1
  24. package/dist/cli/{index-2q5m4m2r.js → index-gm67qky7.js} +3 -3
  25. package/dist/cli/{index-3zvp7t5d.js → index-jysjhvpv.js} +5 -5
  26. package/dist/cli/{index-855qxtdm.js → index-jzg3wrx6.js} +3 -3
  27. package/dist/cli/{index-z2zrcecq.js → index-m31a2q35.js} +2492 -1342
  28. package/dist/cli/{index-rejpbgxp.js → index-m65tnv4j.js} +1 -1
  29. package/dist/cli/{index-g062wnw1.js → index-m7havzy4.js} +1 -1
  30. package/dist/cli/{index-600aqfvm.js → index-mkpczxy7.js} +22 -22
  31. package/dist/cli/{index-k5vwnnym.js → index-n7syegpa.js} +1 -1
  32. package/dist/cli/{index-3bbjpbmn.js → index-pb3vrqsf.js} +2 -2
  33. package/dist/cli/{index-k6f0nryj.js → index-pc3cw1xm.js} +2 -2
  34. package/dist/cli/{index-gy3tcmk4.js → index-r1cnmq4j.js} +4 -4
  35. package/dist/cli/{index-rp4akjrm.js → index-rqm075jf.js} +14 -3
  36. package/dist/cli/{index-dzm5nfh7.js → index-s5ghg4kg.js} +1 -1
  37. package/dist/cli/{index-tnhk72rs.js → index-t5gqdkqx.js} +1 -1
  38. package/dist/cli/{index-v0par978.js → index-t8hv63wy.js} +1 -1
  39. package/dist/cli/{index-as4xaams.js → index-wxjvb4vq.js} +1 -1
  40. package/dist/cli/{index-z2xf153n.js → index-xq9gsfsc.js} +3 -3
  41. package/dist/cli/{index-50v672wd.js → index-yet4r7kp.js} +3 -3
  42. package/dist/cli/{index-n5carmk0.js → index-ytxgr424.js} +1 -1
  43. package/dist/cli/index.js +23 -23
  44. package/dist/cli/{knowledge-escalator-n30h6ran.js → knowledge-escalator-jrkx2b6j.js} +6 -6
  45. package/dist/cli/{knowledge-events-k2qqase2.js → knowledge-events-bha1zpbk.js} +4 -4
  46. package/dist/cli/{knowledge-store-zy226v5a.js → knowledge-store-q438tkyd.js} +1 -1
  47. package/dist/cli/{knowledge-validator-qgyxdpc9.js → knowledge-validator-3nx18fst.js} +2 -2
  48. package/dist/cli/{mcp-pj7acze3.js → mcp-98693syb.js} +1 -1
  49. package/dist/cli/{pending-delegations-sgnnyw55.js → pending-delegations-nkpvtkz4.js} +2 -2
  50. package/dist/cli/{pr-subscriptions-7fb7xm3q.js → pr-subscriptions-qrvhr6jw.js} +1 -1
  51. package/dist/cli/{pr-workflow-gate-5dqcrkz6.js → pr-workflow-gate-k44gdrqd.js} +22 -22
  52. package/dist/cli/{runner-p9hstpxn.js → runner-tzgwvvn1.js} +1 -1
  53. package/dist/cli/{scan-cursor-jc7wwb7e.js → scan-cursor-qjdn8kv6.js} +2 -2
  54. package/dist/cli/{schema-5k68crjb.js → schema-pyax2tq7.js} +5 -1
  55. package/dist/cli/{scope-persistence-xeg4khwk.js → scope-persistence-dyc17kze.js} +5 -5
  56. package/dist/cli/{server-ej6tzd7t.js → server-bmzj5vt2.js} +22 -22
  57. package/dist/cli/{skill-generator-nspt9b3n.js → skill-generator-mp29bht6.js} +7 -7
  58. package/dist/cli/{snapshot-coordination-init-3rs3zpaa.js → snapshot-coordination-init-azkzphpy.js} +22 -22
  59. package/dist/cli/{speckit-checkoff-dqf636ya.js → speckit-checkoff-1cqfakw2.js} +3 -3
  60. package/dist/cli/{worktree-collision-ownership-rbzgbvj4.js → worktree-collision-ownership-58sab2m6.js} +2 -2
  61. package/dist/cli/{worktree-isolation-6zpnzr70.js → worktree-isolation-9egt1ccv.js} +22 -22
  62. package/dist/commands/harness-opt.d.ts +44 -0
  63. package/dist/commands/registry.d.ts +56 -0
  64. package/dist/config/schema.d.ts +20 -0
  65. package/dist/harness/store.d.ts +9 -0
  66. package/dist/index.js +259 -247
  67. package/dist/services/harness-optimizer/comparative.d.ts +81 -0
  68. package/dist/services/harness-optimizer/controller.d.ts +126 -0
  69. package/dist/services/harness-optimizer/execution.d.ts +68 -0
  70. package/dist/services/harness-optimizer/index.d.ts +10 -0
  71. package/dist/services/harness-optimizer/lineage.d.ts +70 -0
  72. package/dist/services/harness-optimizer/manifest.d.ts +65 -0
  73. package/dist/services/harness-optimizer/oracle.d.ts +40 -0
  74. package/opencode-swarm.schema.json +42 -0
  75. package/package.json +1 -1
@@ -0,0 +1,81 @@
1
+ export type ComparativeArmKey = 'baseline' | 'ablation' | 'simple-agent';
2
+ export declare const COMPARATIVE_ARM_KEYS: readonly ComparativeArmKey[];
3
+ export interface ComparativeTaskDescriptor {
4
+ id: string;
5
+ instruction: string;
6
+ }
7
+ export type ComparativeExecutorStatus = 'completed' | 'transient_error' | 'failed';
8
+ export interface ComparativeExecutorResult {
9
+ status: ComparativeExecutorStatus;
10
+ text: string;
11
+ tokens?: {
12
+ input?: number;
13
+ cache?: number;
14
+ output?: number;
15
+ };
16
+ /** Host-reported USD cost of this invocation, when supplied. */
17
+ costUsd?: number;
18
+ }
19
+ export type ComparativeExecutor = (invocation: {
20
+ cwd: string;
21
+ /** Protocol arm, or the round-role labels baseline/candidate on substrate rounds. */
22
+ arm: ComparativeArmKey | 'candidate';
23
+ taskId: string;
24
+ /** Host context passthrough (optional; populated on substrate-driven rounds). */
25
+ instruction?: string;
26
+ payload?: string;
27
+ seed?: string;
28
+ abortSignal?: AbortSignal;
29
+ projectRoot?: string;
30
+ }) => Promise<ComparativeExecutorResult> | ComparativeExecutorResult;
31
+ export interface ComparativeArmResult {
32
+ arm: ComparativeArmKey;
33
+ taskPopulationHash: string;
34
+ /** Denominator: every task in the population was attempted. */
35
+ n: number;
36
+ completed: number;
37
+ failed: number;
38
+ transientFailures: number;
39
+ streamDigest: string;
40
+ outcomes: Array<{
41
+ taskId: string;
42
+ status: ComparativeExecutorStatus;
43
+ }>;
44
+ }
45
+ export interface ComparativeProtocolResult {
46
+ arms: Record<ComparativeArmKey, ComparativeArmResult>;
47
+ taskPopulationHash: string;
48
+ }
49
+ export declare class StreamSnapshotDuplicateError extends Error {
50
+ readonly code = "STREAM_SNAPSHOT_DUPLICATE";
51
+ constructor(digest: string);
52
+ }
53
+ export declare function computeTaskPopulationHash(tasks: readonly ComparativeTaskDescriptor[]): string;
54
+ export declare function computeArmStreamDigest(args: {
55
+ arm: ComparativeArmKey;
56
+ seed: string;
57
+ tasks: readonly ComparativeTaskDescriptor[];
58
+ }): string;
59
+ /**
60
+ * Run the three comparative arms. Each arm executes the full task
61
+ * population inside its own disposable git worktree (never the running
62
+ * checkout), with the working-tree fingerprint verified unchanged by
63
+ * withDisposableWorktree. Failed arms stay in the result with their
64
+ * outcomes — negative results are evidence, not noise.
65
+ */
66
+ export declare function runComparativeProtocol(args: {
67
+ projectRoot: string;
68
+ tasks: ComparativeTaskDescriptor[];
69
+ seed: string;
70
+ executor: ComparativeExecutor;
71
+ /** Previously-recorded digests for this task population; a repeat rejects. */
72
+ previousStreamDigests?: string[];
73
+ /**
74
+ * Arm toggles (harness_opt.run_ablation_arm / run_simple_agent_arm).
75
+ * The baseline arm always runs; unlisted arms are skipped and absent
76
+ * from the result. Defaults to all three arms.
77
+ */
78
+ runAblationArm?: boolean;
79
+ runSimpleAgentArm?: boolean;
80
+ abortSignal?: AbortSignal;
81
+ }): Promise<ComparativeProtocolResult>;
@@ -0,0 +1,126 @@
1
+ import type { ComparativeExecutor, ComparativeTaskDescriptor } from './comparative.js';
2
+ export type HarnessOptSplit = 'train' | 'validation' | 'test';
3
+ /**
4
+ * Stop reasons the round controller can actually produce. Every member has
5
+ * a producing code path (pinned by tests/unit/harness-opt/stop-reasons.test.ts
6
+ * and the controller suite); conditions that surface as typed ERRORS rather
7
+ * than stop results (held-out consumption, frozen-content mismatch, stream
8
+ * snapshot duplication, replay mismatch) are intentionally NOT members —
9
+ * they propagate as their substrate/module errors.
10
+ */
11
+ export declare const HARNESS_OPT_STOP_REASONS: readonly ["completed", "inconclusive", "transient_retry_budget_exhausted", "round_budget_exhausted", "wall_clock_budget_exhausted", "spend_budget_exhausted", "stopped_by_operator"];
12
+ export type HarnessOptStopReason = (typeof HARNESS_OPT_STOP_REASONS)[number];
13
+ export interface HarnessOptRoundResult {
14
+ stopReason: HarnessOptStopReason;
15
+ transientRetries: number;
16
+ roundId: string;
17
+ roundCounter: number;
18
+ saltedSeed: string;
19
+ taskSetHash: string;
20
+ decision: {
21
+ status: string;
22
+ decisionId: string;
23
+ };
24
+ }
25
+ export declare class FrozenTaskSetMismatchError extends Error {
26
+ readonly code = "FROZEN_CONTENT_MISMATCH";
27
+ constructor(key: string);
28
+ }
29
+ /**
30
+ * Content-freeze the task set BEFORE any round. Re-freezing the same
31
+ * {split, seed} key with mutated task content is rejected (typed
32
+ * FROZEN_CONTENT_MISMATCH); identical content is idempotent. The frozen
33
+ * input lives content-addressed under .swarm/evolution/harness-opt/ and
34
+ * the substrate's immutable admission backs the freeze for non-train
35
+ * splits.
36
+ */
37
+ export declare function freezeHarnessOptTaskSet(args: {
38
+ projectRoot: string;
39
+ tasks: ComparativeTaskDescriptor[];
40
+ split: HarnessOptSplit;
41
+ seed: string;
42
+ }): Promise<{
43
+ ok: true;
44
+ contentHash: string;
45
+ } | {
46
+ ok: false;
47
+ code: 'FROZEN_CONTENT_MISMATCH';
48
+ reason: string;
49
+ }>;
50
+ /**
51
+ * Execute ONE governed optimization round. The per-round seed salt
52
+ * `${seed}#roundNNNNNN` guarantees a fresh substrate runId per round, so
53
+ * the SUBSTRATE's claimHeldOutTest is what surfaces
54
+ * TestAlreadyConsumedError on a second held-out round over the same
55
+ * frozen task set — the controller never masks consumption with its own
56
+ * counter.
57
+ */
58
+ export declare function runHarnessOptRound(args: {
59
+ projectRoot: string;
60
+ tasks: ComparativeTaskDescriptor[];
61
+ split: HarnessOptSplit;
62
+ seed: string;
63
+ maxTransientRetries?: number;
64
+ /** Hard wall-clock budget for the round, in milliseconds. */
65
+ maxWallClockMs?: number;
66
+ /** Soft spend budget for the round, in USD. */
67
+ maxSpendUsd?: number;
68
+ executor: ComparativeExecutor;
69
+ abortSignal?: AbortSignal;
70
+ }): Promise<HarnessOptRoundResult>;
71
+ export interface PilotGraduationRecord {
72
+ v: 1;
73
+ recordId: string;
74
+ eligible: boolean;
75
+ evaluatedAt: string;
76
+ criteria: {
77
+ lowerCiThreshold: number;
78
+ };
79
+ evidence: {
80
+ improvementLowerCi: number;
81
+ protectedRegressions: string[];
82
+ };
83
+ }
84
+ /**
85
+ * Evaluate pilot graduation against the predeclared criteria. The evidence
86
+ * — including failing (negative) evidence — is retained verbatim in the
87
+ * durable record. Default flipping stays owned by #2504; this surface only
88
+ * reports eligibility.
89
+ */
90
+ export declare function evaluatePilotGraduation(args: {
91
+ projectRoot: string;
92
+ criteria: {
93
+ lowerCiThreshold: number;
94
+ };
95
+ evidence: {
96
+ improvementLowerCi: number;
97
+ protectedRegressions: string[];
98
+ };
99
+ }): Promise<{
100
+ eligible: boolean;
101
+ recordId: string;
102
+ }>;
103
+ export declare function loadPilotGraduationRecord(projectRoot: string, recordId: string): PilotGraduationRecord | null;
104
+ /** Human-only operator stop: halts further rounds until explicitly resumed
105
+ * via resumeHarnessOptLoop (harness-opt stop --resume). */
106
+ export declare function stopHarnessOptLoop(args: {
107
+ projectRoot: string;
108
+ reason: string;
109
+ }): Promise<{
110
+ stopped: true;
111
+ reason: string;
112
+ }>;
113
+ /** Human-only operator resume: clears an operator stop so rounds run again. */
114
+ export declare function resumeHarnessOptLoop(args: {
115
+ projectRoot: string;
116
+ }): Promise<{
117
+ resumed: boolean;
118
+ previousReason: string | null;
119
+ }>;
120
+ export interface HarnessOptStatus {
121
+ roundCounter: number;
122
+ stopped: boolean;
123
+ stopReason: string | null;
124
+ frozenTaskSets: number;
125
+ }
126
+ export declare function harnessOptStatus(projectRoot: string): HarnessOptStatus;
@@ -0,0 +1,68 @@
1
+ import { type ComparativeExecutor, type ComparativeTaskDescriptor } from './comparative.js';
2
+ export interface HarnessOptTaskSetDescriptor {
3
+ split: 'train' | 'validation' | 'test';
4
+ seed: string;
5
+ tasks: ComparativeTaskDescriptor[];
6
+ }
7
+ export interface TokenUsage {
8
+ tokens_input: number | 'unknown';
9
+ tokens_cache: number | 'unknown';
10
+ tokens_output: number | 'unknown';
11
+ }
12
+ export declare function harnessOptTaskSetRoot(projectRoot: string, taskSetHash: string): string;
13
+ /**
14
+ * Materialize the task-set input root (content-addressed by the task
15
+ * population hash). Existing materialization is left untouched — the
16
+ * content hash pins it; a mutation is detected by re-hashing before use.
17
+ */
18
+ export declare function materializeHarnessOptInput(args: {
19
+ projectRoot: string;
20
+ descriptor: HarnessOptTaskSetDescriptor;
21
+ }): {
22
+ inputRoot: string;
23
+ taskSetHash: string;
24
+ };
25
+ export declare class HarnessOptInputTamperedError extends Error {
26
+ readonly code = "HARNESS_OPT_INPUT_TAMPERED";
27
+ constructor(taskSetHash: string);
28
+ }
29
+ /**
30
+ * Verify the MATERIALIZED ON-DISK input still matches its content-addressed
31
+ * hash: re-reads every task instruction from the content-addressed input
32
+ * root and recomputes the task-population hash from the bytes on disk. A
33
+ * tampered materialization fails typed. (Issue #2503 integrity check.)
34
+ */
35
+ export declare function verifyHarnessOptInput(args: {
36
+ inputRoot: string;
37
+ taskIds: readonly string[];
38
+ taskSetHash: string;
39
+ }): boolean;
40
+ export interface SubstrateExecutionResult {
41
+ runStatus: string;
42
+ decisionStatus: string;
43
+ decisionId: string;
44
+ /** Host-reported USD spend accumulated across executor invocations. */
45
+ reportedSpendUsd: number;
46
+ outcomes: Array<{
47
+ candidateId: string;
48
+ outcome: string;
49
+ }>;
50
+ tokens: TokenUsage;
51
+ }
52
+ /**
53
+ * Execute one governed evaluation round through the production substrate.
54
+ * The simplified executor runs inside the substrate's disposable worktree
55
+ * (isolatedRoot — never the running checkout) and token usage reported by
56
+ * the host executor is captured; missing host data stays 'unknown'.
57
+ */
58
+ export declare function runSubstrateEvaluation(args: {
59
+ projectRoot: string;
60
+ inputRoot: string;
61
+ descriptor: HarnessOptTaskSetDescriptor;
62
+ seed: string;
63
+ decidedAt: string;
64
+ maxTransientRetries?: number;
65
+ maxSpendUsd?: number;
66
+ executor: ComparativeExecutor;
67
+ abortSignal?: AbortSignal;
68
+ }): Promise<SubstrateExecutionResult>;
@@ -0,0 +1,10 @@
1
+ export type { ComparativeArmKey, ComparativeArmResult, ComparativeExecutor, ComparativeExecutorResult, ComparativeProtocolResult, ComparativeTaskDescriptor, } from './comparative.js';
2
+ export { COMPARATIVE_ARM_KEYS, computeArmStreamDigest, computeTaskPopulationHash, runComparativeProtocol, StreamSnapshotDuplicateError, } from './comparative.js';
3
+ export type { HarnessOptRoundResult, HarnessOptSplit, HarnessOptStatus, PilotGraduationRecord, } from './controller.js';
4
+ export { evaluatePilotGraduation, FrozenTaskSetMismatchError, freezeHarnessOptTaskSet, harnessOptStatus, loadPilotGraduationRecord, runHarnessOptRound, stopHarnessOptLoop, } from './controller.js';
5
+ export type { HarnessOptLineageRecord } from './lineage.js';
6
+ export { computeCandidateConfigDigest, computePromptSelectionDigest, listHarnessOptLineage, ReplayDecisionMismatchError, recordHarnessOptRound, replayHarnessOptLineage, } from './lineage.js';
7
+ export type { ComparativeManifestV1, ManifestValidationFailureCode, ManifestValidationResult, } from './manifest.js';
8
+ export { validateComparativeManifest } from './manifest.js';
9
+ export type { OracleArmInput, OracleVerdict, TokenCount } from './oracle.js';
10
+ export { evaluateIndependentOracle } from './oracle.js';
@@ -0,0 +1,70 @@
1
+ import type { ComparativeTaskDescriptor } from './comparative.js';
2
+ import type { TokenUsage } from './execution.js';
3
+ export interface HarnessOptLineageRecord {
4
+ v: 1;
5
+ roundId: string;
6
+ /** Identical to roundId: one identity string, no derived second id. */
7
+ replayLineageId: string;
8
+ recordedAt: string;
9
+ split: 'train' | 'validation' | 'test';
10
+ baseSeed: string;
11
+ saltedSeed: string;
12
+ roundCounter: number;
13
+ taskSetHash: string;
14
+ candidateConfigDigest: string;
15
+ promptSelectionDigest: string;
16
+ tokens_input: number | 'unknown';
17
+ tokens_cache: number | 'unknown';
18
+ tokens_output: number | 'unknown';
19
+ artifactOutcome: string;
20
+ oracle: {
21
+ verdict: string;
22
+ reasons: string[];
23
+ };
24
+ decision: {
25
+ status: string;
26
+ decisionId: string;
27
+ };
28
+ execution: {
29
+ decidedAt: string;
30
+ maxTransientRetries?: number;
31
+ tasks: ComparativeTaskDescriptor[];
32
+ };
33
+ }
34
+ export declare class ReplayDecisionMismatchError extends Error {
35
+ readonly code = "REPLAY_DECISION_MISMATCH";
36
+ constructor(recorded: string, replayed: string);
37
+ }
38
+ export declare function recordHarnessOptRound(args: {
39
+ projectRoot: string;
40
+ record: HarnessOptLineageRecord;
41
+ }): Promise<void>;
42
+ export declare function listHarnessOptLineage(projectRoot: string): HarnessOptLineageRecord[];
43
+ export declare function loadHarnessOptLineageRecord(args: {
44
+ projectRoot: string;
45
+ roundId: string;
46
+ }): HarnessOptLineageRecord | null;
47
+ export declare function computeCandidateConfigDigest(args: {
48
+ saltedSeed: string;
49
+ split: string;
50
+ maxTransientRetries?: number;
51
+ }): string;
52
+ export declare function computePromptSelectionDigest(tasks: readonly ComparativeTaskDescriptor[]): string;
53
+ /**
54
+ * Replay a recorded round. The recorded fully-salted seed and decidedAt are
55
+ * re-used WITHOUT reading or advancing the round counter, so the substrate
56
+ * derives the identical runId and returns the immutable cached run. The
57
+ * replay executor throws if ever invoked — a replay must hit the cached
58
+ * record, never re-spend execution. Any decision mismatch (a tampered or
59
+ * stale record) fails typed.
60
+ */
61
+ export declare function replayHarnessOptLineage(args: {
62
+ projectRoot: string;
63
+ lineageId: string;
64
+ }): Promise<{
65
+ decision: {
66
+ status: string;
67
+ decisionId: string;
68
+ };
69
+ }>;
70
+ export type { TokenUsage };
@@ -0,0 +1,65 @@
1
+ /**
2
+ * Frozen comparative-evaluation manifest (issue #2503).
3
+ *
4
+ * A comparative run MUST be governed by a manifest frozen BEFORE any
5
+ * optimization: a named previous stable release, the repaired unoptimized
6
+ * baseline reference, the frozen corpus/task population, exact PR revisions
7
+ * with defect/clean labels, excluded training cases, matched budgets,
8
+ * per-arm sample sizes, numeric acceptance thresholds, and metric
9
+ * definitions. The validator refuses empty required fields and improvement
10
+ * claims without a measured result — the issue's "no empty manifest fields
11
+ * or unmeasured improvement claims" contract.
12
+ */
13
+ export type ManifestValidationFailureCode = 'MANIFEST_MALFORMED' | 'MANIFEST_FIELD_EMPTY' | 'CLAIM_UNMEASURED';
14
+ export interface ComparativeManifestV1 {
15
+ releaseName: string;
16
+ repairedBaselineRef?: string;
17
+ corpus?: {
18
+ taskPopulationPointer: string;
19
+ excludedTrainingCases: Array<{
20
+ caseId: string;
21
+ reason: string;
22
+ }>;
23
+ };
24
+ prRevisions?: Array<{
25
+ revision: string;
26
+ label: 'defect' | 'clean';
27
+ }>;
28
+ matchedBudgets?: {
29
+ model: string;
30
+ provider: string;
31
+ tool: string;
32
+ timeBudgetMs?: number;
33
+ };
34
+ sampleSizes: {
35
+ baseline: number;
36
+ ablation: number;
37
+ 'simple-agent': number;
38
+ };
39
+ thresholds: Record<string, number>;
40
+ metricDefinitions?: Array<{
41
+ metric: string;
42
+ definition: string;
43
+ }>;
44
+ defectLabels: string[];
45
+ cleanLabels: string[];
46
+ claims: Array<{
47
+ kind: 'improvement' | 'negative';
48
+ metric: string;
49
+ measuredResult?: Record<string, number>;
50
+ }>;
51
+ }
52
+ export type ManifestValidationResult = {
53
+ ok: true;
54
+ manifest: ComparativeManifestV1;
55
+ } | {
56
+ ok: false;
57
+ code: ManifestValidationFailureCode;
58
+ reason: string;
59
+ };
60
+ /**
61
+ * Validate a frozen comparative manifest. Accepts only fully-populated
62
+ * manifests; every rejection names the offending field. Unknown/malformed
63
+ * input fails closed with MANIFEST_MALFORMED rather than being coerced.
64
+ */
65
+ export declare function validateComparativeManifest(input: unknown): ManifestValidationResult;
@@ -0,0 +1,40 @@
1
+ /**
2
+ * Independent oracle for the governed HarnessOpt capstone (issue #2503).
3
+ *
4
+ * Scores accepted task outcomes, artifact validity, verification evidence,
5
+ * and completion quality SEPARATELY from the optimizer and from the task's
6
+ * own scorer (src/evaluation/runner.ts scoreExecution). The oracle is the
7
+ * acceptance backstop the issue demands: a candidate whose token count
8
+ * improves while accepted artifact quality or verification evidence falls
9
+ * MUST be rejected, with reasons naming each drop.
10
+ */
11
+ export type TokenCount = number | 'unknown';
12
+ export interface OracleArmInput {
13
+ /** Task-population denominator for the arm. */
14
+ n: number;
15
+ /** Count of accepted artifacts the arm produced. */
16
+ acceptedArtifacts: number;
17
+ /** Count of verification-evidence records backing those artifacts. */
18
+ verificationEvidence: number;
19
+ tokens: {
20
+ input: TokenCount;
21
+ cache: TokenCount;
22
+ output: TokenCount;
23
+ };
24
+ }
25
+ export interface OracleVerdict {
26
+ verdict: 'accept' | 'reject';
27
+ reasons: string[];
28
+ }
29
+ /**
30
+ * Compare a candidate arm against the baseline arm. Acceptance requires
31
+ * the candidate to be no worse on accepted artifacts, verification
32
+ * evidence, or (when both arms report tokens) token usage. Any quality or
33
+ * verification drop rejects the candidate regardless of token improvement.
34
+ * Token totals reported as 'unknown' by the host are treated as unknown,
35
+ * never zero, and therefore never counted as an improvement axis.
36
+ */
37
+ export declare function evaluateIndependentOracle(args: {
38
+ baseline: OracleArmInput;
39
+ candidate: OracleArmInput;
40
+ }): Promise<OracleVerdict>;
@@ -3769,6 +3769,48 @@
3769
3769
  },
3770
3770
  "additionalProperties": false
3771
3771
  },
3772
+ "harness_opt": {
3773
+ "description": "Governed HarnessOpt optimization capstone (issue #2503). Disabled by default; /swarm harness-opt run requires enabled: true plus --confirm.",
3774
+ "type": "object",
3775
+ "properties": {
3776
+ "enabled": {
3777
+ "default": false,
3778
+ "type": "boolean"
3779
+ },
3780
+ "max_rounds": {
3781
+ "default": 5,
3782
+ "type": "integer",
3783
+ "minimum": 1,
3784
+ "maximum": 50
3785
+ },
3786
+ "max_transient_retries": {
3787
+ "default": 2,
3788
+ "type": "integer",
3789
+ "minimum": 0,
3790
+ "maximum": 10
3791
+ },
3792
+ "max_wall_clock_ms": {
3793
+ "default": 3600000,
3794
+ "type": "integer",
3795
+ "minimum": 10000,
3796
+ "maximum": 86400000
3797
+ },
3798
+ "max_spend_usd": {
3799
+ "type": "number",
3800
+ "minimum": 0,
3801
+ "maximum": 1000
3802
+ },
3803
+ "run_ablation_arm": {
3804
+ "default": true,
3805
+ "type": "boolean"
3806
+ },
3807
+ "run_simple_agent_arm": {
3808
+ "default": true,
3809
+ "type": "boolean"
3810
+ }
3811
+ },
3812
+ "additionalProperties": false
3813
+ },
3772
3814
  "spec_writer": {
3773
3815
  "description": "Spec writer agent (v2) — independent model for .swarm/spec.md authorship.",
3774
3816
  "type": "object",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "opencode-swarm",
3
- "version": "7.180.1",
3
+ "version": "7.181.0",
4
4
  "description": "Architect-centric agentic swarm plugin for OpenCode - hub-and-spoke orchestration with SME consultation, code generation, and QA review",
5
5
  "main": "dist/index.js",
6
6
  "types": "dist/index.d.ts",