opencode-swarm 7.136.1 → 7.136.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.opencode/skills/brainstorm/SKILL.md +1 -1
- package/.opencode/skills/clarify/SKILL.md +3 -3
- package/.opencode/skills/clarify-spec/SKILL.md +2 -2
- package/.opencode/skills/consult/SKILL.md +1 -1
- package/.opencode/skills/council/SKILL.md +1 -1
- package/.opencode/skills/critic-gate/SKILL.md +8 -3
- package/.opencode/skills/discover/SKILL.md +1 -1
- package/.opencode/skills/execute/SKILL.md +1 -1
- package/.opencode/skills/gate-attribution/SKILL.md +1 -1
- package/.opencode/skills/issue-ingest/SKILL.md +3 -3
- package/.opencode/skills/phase-wrap/SKILL.md +1 -1
- package/.opencode/skills/plan/SKILL.md +3 -3
- package/.opencode/skills/pre-phase-briefing/SKILL.md +1 -1
- package/.opencode/skills/resume/SKILL.md +1 -1
- package/.opencode/skills/specify/SKILL.md +1 -1
- package/.opencode/skills/swarm-pr-feedback/SKILL.md +41 -1
- package/.opencode/skills/swarm-pr-review/SKILL.md +72 -24
- package/README.md +18 -0
- package/dist/background/workspace-snapshot.d.ts +54 -0
- package/dist/cli/{config-doctor-6fphg5xm.js → config-doctor-g71mz50j.js} +2 -2
- package/dist/cli/{core-4z1s2ak1.js → core-jpjk2qvt.js} +1 -1
- package/dist/cli/{curation-policy-m24jbyag.js → curation-policy-zhyttjpa.js} +6 -6
- package/dist/cli/{curator-zrt5sqj2.js → curator-36dxw38r.js} +27 -24
- package/dist/cli/{curator-llm-factory-w5arq7tr.js → curator-llm-factory-k2a7cv41.js} +27 -24
- package/dist/cli/{evidence-summary-service-w006jnpg.js → evidence-summary-service-nse06dqr.js} +8 -6
- package/dist/cli/{gate-evidence-aenyz6vt.js → gate-evidence-kdygjr56.js} +4 -4
- package/dist/cli/{guardrail-explain-462jwhbh.js → guardrail-explain-1k337z5d.js} +28 -25
- package/dist/cli/{guardrail-log-ax21jd9e.js → guardrail-log-xbc9bvdt.js} +3 -3
- package/dist/cli/hashing-mjwn6j4y.js +34 -0
- package/dist/cli/{hive-promoter-2vacnbws.js → hive-promoter-daxrn62p.js} +27 -24
- package/dist/cli/{index-r9v98d0y.js → index-0vq9gnqd.js} +46 -5
- package/dist/cli/{index-jwrydqjp.js → index-176zwqcq.js} +2 -2
- package/dist/cli/{index-cggqh2dz.js → index-45y0y3xh.js} +57 -1000
- package/dist/cli/{index-gf7hqbmz.js → index-68tjkaqk.js} +5 -1
- package/dist/cli/{index-dyy3hvk3.js → index-79pgzj9a.js} +4 -4
- package/dist/cli/{index-45t7w06b.js → index-7t3vjw5e.js} +1 -1
- package/dist/cli/{index-ey29aap6.js → index-b3jfcptk.js} +1 -1
- package/dist/cli/{index-prnjt2jr.js → index-brtg922w.js} +4576 -2316
- package/dist/cli/{index-7a2hm51h.js → index-cze4bq1x.js} +41 -0
- package/dist/cli/{index-mrtms113.js → index-d2sf9an1.js} +2 -2
- package/dist/cli/{index-wxyxf0bd.js → index-d61h3f3h.js} +2 -2
- package/dist/cli/{index-rhgctfn3.js → index-g2kpqpvh.js} +29 -26
- package/dist/cli/{index-g7g4hqh3.js → index-g5rtcb1n.js} +1 -1
- package/dist/cli/{index-1sq47n6t.js → index-gj5jerzz.js} +12 -6
- package/dist/cli/{index-c9ddxv4k.js → index-hh34tyv7.js} +1 -1
- package/dist/cli/{index-h3bweb88.js → index-kp5h245k.js} +5 -5
- package/dist/cli/{index-tcn457d5.js → index-kym0cctr.js} +1 -1
- package/dist/cli/{index-kzc2ygea.js → index-mgbpbs1f.js} +3 -3
- package/dist/cli/{index-tfa40hwb.js → index-n89fdxwg.js} +19 -2
- package/dist/cli/index-ne28wyyc.js +783 -0
- package/dist/cli/index-nfm9f10v.js +1100 -0
- package/dist/cli/{index-zkddc5ye.js → index-pff46kfv.js} +2 -2
- package/dist/cli/{index-f4cmtd89.js → index-rtgqy0yg.js} +2 -2
- package/dist/cli/{index-09b9zncg.js → index-tncy55bp.js} +1 -1
- package/dist/cli/{index-vg3yx648.js → index-w6j1n5az.js} +1 -1
- package/dist/cli/{index-hzkvbycn.js → index-wbdxmf1a.js} +6 -6
- package/dist/cli/{index-y6a7gjtj.js → index-wsdnttf4.js} +228 -81
- package/dist/cli/index-y8552snf.js +264 -0
- package/dist/cli/{index-9gxp450h.js → index-zcrvn579.js} +2 -2
- package/dist/cli/{index-p1pqqwgp.js → index-zwh5dewz.js} +1 -1
- package/dist/cli/{index-dzyjb33e.js → index-zzhyws9g.js} +1 -1
- package/dist/cli/index.js +27 -24
- package/dist/cli/{knowledge-escalator-1zt8qy1c.js → knowledge-escalator-h7fspgph.js} +7 -7
- package/dist/cli/{knowledge-events-ej3s9tsm.js → knowledge-events-vkf7an5n.js} +5 -5
- package/dist/cli/{knowledge-link-mm1w967j.js → knowledge-link-zr40rnwr.js} +4 -4
- package/dist/cli/{knowledge-store-ey4cbkp4.js → knowledge-store-s4976v9c.js} +5 -5
- package/dist/cli/{knowledge-validator-zwmq7s2c.js → knowledge-validator-64ppqy4y.js} +8 -8
- package/dist/cli/{pending-delegations-qajsxct0.js → pending-delegations-0h5b18p7.js} +3 -3
- package/dist/cli/{pr-subscriptions-qhr41epq.js → pr-subscriptions-jn0h047q.js} +3 -3
- package/dist/cli/runner-deeswadt.js +21 -0
- package/dist/cli/{scan-cursor-809hf2n1.js → scan-cursor-xbkae12h.js} +6 -6
- package/dist/cli/{schema-mhd7xqwr.js → schema-7jm70cab.js} +5 -1
- package/dist/cli/{scope-persistence-h2fpgxww.js → scope-persistence-5xc9ntdh.js} +4 -4
- package/dist/cli/{skill-generator-yyc8zd9j.js → skill-generator-8gtq1ajr.js} +9 -9
- package/dist/cli/{telemetry-859khp82.js → telemetry-6678gya0.js} +1 -1
- package/dist/cli/{workspace-snapshot-jmyamqnv.js → workspace-snapshot-h5rzw37b.js} +5 -1
- package/dist/cli/{worktree-collision-ownership-13btcj9g.js → worktree-collision-ownership-wt7cc850.js} +3 -3
- package/dist/commands/registry.d.ts +72 -0
- package/dist/commands/skill-opt.d.ts +41 -0
- package/dist/config/schema.d.ts +57 -0
- package/dist/hooks/delegation-gate.d.ts +12 -1
- package/dist/hooks/gate-denial-tracker.d.ts +175 -0
- package/dist/hooks/guardrails/execution-episode.d.ts +41 -0
- package/dist/hooks/guardrails/execution-stall.d.ts +285 -0
- package/dist/hooks/guardrails/file-authority.d.ts +11 -2
- package/dist/hooks/guardrails/internals-guard.d.ts +117 -0
- package/dist/hooks/guardrails/messages-transform.d.ts +85 -0
- package/dist/hooks/pr-workflow-gate.d.ts +261 -4
- package/dist/hooks/pr-workflow-response-gate.d.ts +27 -9
- package/dist/hooks/trajectory-logger.d.ts +76 -0
- package/dist/hooks/write-target-resolver.d.ts +19 -0
- package/dist/index.js +511 -462
- package/dist/memory/schema.d.ts +3 -3
- package/dist/prm/index.d.ts +2 -0
- package/dist/services/skill-evaluator.d.ts +17 -0
- package/dist/services/skill-optimizer/activation.d.ts +58 -0
- package/dist/services/skill-optimizer/candidates.d.ts +88 -0
- package/dist/services/skill-optimizer/controller.d.ts +146 -0
- package/dist/services/skill-optimizer/deterministic-seed.d.ts +29 -0
- package/dist/services/skill-optimizer/lifecycle.d.ts +70 -0
- package/dist/services/skill-optimizer/promoted-external-staleness.d.ts +98 -0
- package/dist/services/skill-optimizer/retirement.d.ts +54 -0
- package/dist/services/skill-optimizer/skill-eval-tasks.d.ts +47 -0
- package/dist/services/skill-optimizer/smoke.d.ts +46 -0
- package/dist/services/skill-optimizer/store.d.ts +118 -0
- package/dist/state.d.ts +32 -0
- package/dist/telemetry.d.ts +66 -1
- package/dist/tools/write-pr-review-trigger-eval.d.ts +2 -1
- package/dist/types/events.d.ts +14 -1
- package/dist/utils/stable-stringify.d.ts +46 -0
- package/evaluation-fixtures/skill-eval/scoring/score-skill-eval.cjs +97 -0
- package/package.json +2 -1
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Skill smoke validator — the `smoke_validated` transition (issue #1822).
|
|
3
|
+
*
|
|
4
|
+
* Composes existing primitives rather than duplicating them:
|
|
5
|
+
* - `validateSkillPath` (path containment — knowledge-validator.ts);
|
|
6
|
+
* - symlink/reparse denial (deny escape; mirrors bundled-skills.ts pattern);
|
|
7
|
+
* - frontmatter schema check (YAML parse + required keys);
|
|
8
|
+
* - the phrase-eval gate (`evaluateSkillChange` / `isRejectedSkillContent`
|
|
9
|
+
* from skill-evaluator.ts) — refuses content the rejection ledger already
|
|
10
|
+
* rejected;
|
|
11
|
+
* - a bounded subprocess check via `spawnAsync` if the skill declares a
|
|
12
|
+
* check command (cwd explicit, stdin ignored, timeout, kill, 512KB cap).
|
|
13
|
+
*/
|
|
14
|
+
import { validateSkillPath } from '../../hooks/knowledge-validator.js';
|
|
15
|
+
import { spawnAsync } from '../../hooks/spawn-helper.js';
|
|
16
|
+
import { evaluateSkillChange, isRejectedSkillContent } from '../skill-evaluator.js';
|
|
17
|
+
export interface SmokeInput {
|
|
18
|
+
directory: string;
|
|
19
|
+
skillSlug: string;
|
|
20
|
+
/** Candidate SKILL.md content (not yet written to the skill root). */
|
|
21
|
+
candidateContent: string;
|
|
22
|
+
/** Incumbent SKILL.md content (current). Empty string if no incumbent. */
|
|
23
|
+
incumbentContent: string;
|
|
24
|
+
/** Optional check command the skill declares (e.g. `["bun", "test"]`). */
|
|
25
|
+
checkCommand?: string[];
|
|
26
|
+
/** Timeout for the optional check command. */
|
|
27
|
+
checkTimeoutMs?: number;
|
|
28
|
+
}
|
|
29
|
+
export interface SmokeResult {
|
|
30
|
+
ok: boolean;
|
|
31
|
+
verdict: 'COMPLIANT' | 'VIOLATED';
|
|
32
|
+
notes: string[];
|
|
33
|
+
}
|
|
34
|
+
export declare const _internals: {
|
|
35
|
+
validateSkillPath: typeof validateSkillPath;
|
|
36
|
+
isSymbolicLink: (p: string) => boolean;
|
|
37
|
+
escapedRoot: (root: string, target: string) => boolean;
|
|
38
|
+
realpath: (p: string) => string;
|
|
39
|
+
evaluateSkillChange: typeof evaluateSkillChange;
|
|
40
|
+
isRejectedSkillContent: typeof isRejectedSkillContent;
|
|
41
|
+
spawnAsync: typeof spawnAsync;
|
|
42
|
+
};
|
|
43
|
+
/** Validate a candidate skill before the validation-running transition. */
|
|
44
|
+
export declare function validateSkillSmoke(input: SmokeInput): Promise<SmokeResult>;
|
|
45
|
+
/** Read the incumbent SKILL.md content, or return empty string if absent. */
|
|
46
|
+
export declare function readIncumbentContent(directory: string, skillSlug: string): string;
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Append-only lifecycle store for the governed skill optimizer (issue #1822).
|
|
3
|
+
*
|
|
4
|
+
* Storage layout (all under `.swarm/`, AGENTS.md invariant #4):
|
|
5
|
+
* .swarm/evolution/skills/<skillSlug>/<candidateId>/lifecycle.jsonl (authoritative)
|
|
6
|
+
* .swarm/evolution/skills/<skillSlug>/<candidateId>/state.json (derived projection)
|
|
7
|
+
* .swarm/evolution/skills/<skillSlug>/<candidateId>/baseline.md (frozen baseline snapshot)
|
|
8
|
+
* .swarm/evolution/skills/<skillSlug>/<candidateId>/candidate.md (drafted candidate)
|
|
9
|
+
* .swarm/evolution/skills/<skillSlug>/<candidateId>/diff.patch (computed diff)
|
|
10
|
+
* .swarm/evolution/skills/<skillSlug>/<candidateId>/rollback.md (pre-activation snapshot)
|
|
11
|
+
* .swarm/evolution/skills/lifecycle-quarantine.<ts>.<hash> (corrupt-tail salvage)
|
|
12
|
+
*
|
|
13
|
+
* Integrity model mirrors `src/plan/ledger.ts`:
|
|
14
|
+
* - fsync+rename atomic append, gated by an evidence lock;
|
|
15
|
+
* - replay stops at the first unparseable line, sets `truncated`, captures
|
|
16
|
+
* the bad suffix, and quarantines it WITHOUT rewriting the canonical ledger;
|
|
17
|
+
* - hash-before/hash-after chain (reuses the plan-ledger field semantics —
|
|
18
|
+
* no parallel "previousStateHash");
|
|
19
|
+
* - partial/corrupt writes never count as acceptance (replay-after-write
|
|
20
|
+
* verification in `recordTransition`, lifecycle.ts).
|
|
21
|
+
*
|
|
22
|
+
* IDs are collision-resistant and filesystem-safe (`crypto.randomUUID()`).
|
|
23
|
+
*/
|
|
24
|
+
import { existsSync, readFileSync, renameSync, unlinkSync, writeFileSync } from 'node:fs';
|
|
25
|
+
import { withEvidenceLock } from '../../evidence/lock.js';
|
|
26
|
+
/** Lifecycle states (issue #1822 durable lifecycle). */
|
|
27
|
+
export type SkillOptState = 'discovered' | 'drafted' | 'smoke_validated' | 'validation_running' | 'accepted_pending_approval' | 'rejected' | 'inconclusive' | 'activated' | 'expired' | 'rolled_back';
|
|
28
|
+
/** A single append-only lifecycle event. Hash chain reuses plan-ledger semantics. */
|
|
29
|
+
export interface SkillOptEvent {
|
|
30
|
+
seq: number;
|
|
31
|
+
timestamp: string;
|
|
32
|
+
candidateId: string;
|
|
33
|
+
skillSlug: string;
|
|
34
|
+
eventType: string;
|
|
35
|
+
fromState: SkillOptState | null;
|
|
36
|
+
toState: SkillOptState;
|
|
37
|
+
actor: string;
|
|
38
|
+
origin: string;
|
|
39
|
+
contentHashBefore: string | null;
|
|
40
|
+
contentHashAfter: string | null;
|
|
41
|
+
hashBefore: string;
|
|
42
|
+
hashAfter: string;
|
|
43
|
+
reason: string;
|
|
44
|
+
evidenceRefs: string[];
|
|
45
|
+
payload?: Record<string, unknown>;
|
|
46
|
+
}
|
|
47
|
+
export interface ReplayResult {
|
|
48
|
+
events: SkillOptEvent[];
|
|
49
|
+
state: SkillOptState | null;
|
|
50
|
+
truncated: boolean;
|
|
51
|
+
badSuffix: string | null;
|
|
52
|
+
/** Sequence number of the last complete (hash-verified) event. */
|
|
53
|
+
lastCompleteSeq: number;
|
|
54
|
+
}
|
|
55
|
+
/** SHA-256 over canonical JSON of an arbitrary value. */
|
|
56
|
+
export declare function computeStateHash(value: unknown): string;
|
|
57
|
+
/** SHA-256 of raw text content (a SKILL.md body). */
|
|
58
|
+
export declare function computeContentHash(content: string): string;
|
|
59
|
+
export declare function isValidSkillSlug(slug: string): boolean;
|
|
60
|
+
/** Validate a candidate ID is filesystem-safe (uuid or equivalent). */
|
|
61
|
+
export declare function isValidCandidateId(id: string): boolean;
|
|
62
|
+
/** Mint a fresh collision-resistant candidate ID. */
|
|
63
|
+
export declare function mintCandidateId(): string;
|
|
64
|
+
/**
|
|
65
|
+
* Fsync-then-rename atomic write (mirrors `writeFileFsyncedThenRename` in
|
|
66
|
+
* `src/plan/ledger.ts:169`). fsync guarantees the bytes hit durable storage
|
|
67
|
+
* before the rename makes the file visible.
|
|
68
|
+
*/
|
|
69
|
+
declare function writeFileFsyncedThenRename(tempPath: string, targetPath: string, data: string): void;
|
|
70
|
+
/** DI seam for test injection (AGENTS.md invariant #7 — preferred over mock.module). */
|
|
71
|
+
export declare const _internals: {
|
|
72
|
+
withEvidenceLock: typeof withEvidenceLock;
|
|
73
|
+
writeFileFsyncedThenRename: typeof writeFileFsyncedThenRename;
|
|
74
|
+
writeFileSync: typeof writeFileSync;
|
|
75
|
+
readFileSync: typeof readFileSync;
|
|
76
|
+
renameSync: typeof renameSync;
|
|
77
|
+
unlinkSync: typeof unlinkSync;
|
|
78
|
+
existsSync: typeof existsSync;
|
|
79
|
+
statSync: import("node:fs").StatSyncFn;
|
|
80
|
+
};
|
|
81
|
+
/**
|
|
82
|
+
* Append a lifecycle event atomically. Computes the hash chain from the
|
|
83
|
+
* current replayed state, fsync+rename under an evidence lock, then verifies
|
|
84
|
+
* the append by re-reading. Returns the persisted event.
|
|
85
|
+
*
|
|
86
|
+
* Throws if the post-write replay does not contain the appended event at the
|
|
87
|
+
* expected seq — a partial/corrupt write never counts as a successful
|
|
88
|
+
* transition.
|
|
89
|
+
*/
|
|
90
|
+
export declare function appendEvent(directory: string, eventInput: Omit<SkillOptEvent, 'seq' | 'timestamp' | 'hashBefore' | 'hashAfter'> & {
|
|
91
|
+
timestamp?: string;
|
|
92
|
+
}): Promise<SkillOptEvent>;
|
|
93
|
+
/**
|
|
94
|
+
* Replay a candidate's lifecycle ledger. Mirrors `readLedgerEventsWithIntegrity`
|
|
95
|
+
* in `src/plan/ledger.ts:1105`: stops at the first unparseable line, marks the
|
|
96
|
+
* replay `truncated`, and surfaces the bad suffix for quarantine. The canonical
|
|
97
|
+
* ledger is NEVER rewritten or truncated by this read.
|
|
98
|
+
*/
|
|
99
|
+
export declare function replayCandidate(directory: string, skillSlug: string, candidateId: string): ReplayResult;
|
|
100
|
+
/**
|
|
101
|
+
* Quarantine a corrupt ledger suffix. Writes the bad suffix to a unique side
|
|
102
|
+
* file under `.swarm/evolution/skills/`. NEVER rewrites or truncates the
|
|
103
|
+
* canonical `lifecycle.jsonl`. Mirrors `quarantineLedgerSuffix` in
|
|
104
|
+
* `src/plan/ledger.ts:1188`.
|
|
105
|
+
*/
|
|
106
|
+
export declare function quarantineSuffix(directory: string, skillSlug: string, badSuffix: string): string;
|
|
107
|
+
/**
|
|
108
|
+
* Derive the projection `state.json` from the ledger replay. Derived, not
|
|
109
|
+
* authoritative — callers must always re-derive from the ledger rather than
|
|
110
|
+
* trust a stale projection. Writes atomically (temp+rename).
|
|
111
|
+
*/
|
|
112
|
+
export declare function writeStateProjection(directory: string, skillSlug: string, candidateId: string, replay: ReplayResult): void;
|
|
113
|
+
/** Snapshot a text file (baseline/candidate/rollback) atomically. */
|
|
114
|
+
export declare function writeArtifact(directory: string, skillSlug: string, candidateId: string, fileName: string, content: string): string;
|
|
115
|
+
/** Read a snapshot artifact, returning null if absent. */
|
|
116
|
+
export declare function readArtifact(directory: string, skillSlug: string, candidateId: string, fileName: string): string | null;
|
|
117
|
+
export declare const SKILL_OPT_STORE_SCHEMA_VERSION = "1";
|
|
118
|
+
export {};
|
package/dist/state.d.ts
CHANGED
|
@@ -397,6 +397,22 @@ export interface AgentSessionState {
|
|
|
397
397
|
prmTrajectoryStep: number;
|
|
398
398
|
/** Whether a hard stop has been triggered */
|
|
399
399
|
prmHardStopPending: boolean;
|
|
400
|
+
/**
|
|
401
|
+
* Issue #2063 C2 — second, independent one-shot token for the PRM hard stop.
|
|
402
|
+
*
|
|
403
|
+
* `prmHardStopPending` is the DENY token: guardrails `toolBefore` consumes it
|
|
404
|
+
* by throwing the HARD STOP denial once. `prmHardStopInjectPending` is the
|
|
405
|
+
* INJECT token: `messagesTransform` consumes it by prepending the
|
|
406
|
+
* `[HARD STOP]` block into the next completion. They are deliberately
|
|
407
|
+
* separate because either consumer can run first, and a single shared flag
|
|
408
|
+
* meant whichever ran first disarmed the other — so the escalation was either
|
|
409
|
+
* denied without ever being explained, or explained without ever being
|
|
410
|
+
* denied.
|
|
411
|
+
*
|
|
412
|
+
* Optional: ~25 existing test/session literals enumerate the required PRM
|
|
413
|
+
* fields, and this one is additive.
|
|
414
|
+
*/
|
|
415
|
+
prmHardStopInjectPending?: boolean;
|
|
400
416
|
/** Per-session escalation tracker instance (set lazily by PRM hook) */
|
|
401
417
|
prmEscalationTracker?: EscalationTracker;
|
|
402
418
|
/** Cross-turn set of already-injected PRM advisory dedupe keys
|
|
@@ -406,6 +422,22 @@ export interface AgentSessionState {
|
|
|
406
422
|
* pattern's count advances escalation. Bounded by distinct (pattern, level)
|
|
407
423
|
* pairs — at most (numPatterns × 3 levels). */
|
|
408
424
|
prmInjectedAdvisoryKeys: Set<string>;
|
|
425
|
+
/**
|
|
426
|
+
* Issue #2063 B3/B5 — whether an "execution episode" is currently armed for
|
|
427
|
+
* this session.
|
|
428
|
+
*
|
|
429
|
+
* An episode arms when the session actually attempts execution work (a `Task`
|
|
430
|
+
* dispatch to a mutating/verifying role, or an `update_task_status(...,
|
|
431
|
+
* in_progress)` that succeeds) and disarms on episode lapse. Consumers read
|
|
432
|
+
* it through {@link isExecutionEpisodeArmed} in
|
|
433
|
+
* `src/hooks/guardrails/execution-episode.ts` rather than touching the field,
|
|
434
|
+
* so the arming policy has exactly one owner.
|
|
435
|
+
*
|
|
436
|
+
* Deliberately reset on rehydrate (see `src/session/snapshot-reader.ts`): a
|
|
437
|
+
* stale `in_progress` task left over from a previous session must NOT arm a
|
|
438
|
+
* fresh one.
|
|
439
|
+
*/
|
|
440
|
+
executionEpisodeArmed?: boolean;
|
|
409
441
|
/** Active PR subscriptions for the background poller, keyed by `${repoFullName}::${prNumber}` */
|
|
410
442
|
prSubscriptions: Map<string, PrSubscriptionState>;
|
|
411
443
|
/**
|
package/dist/telemetry.d.ts
CHANGED
|
@@ -1,5 +1,31 @@
|
|
|
1
1
|
import type { DelegationCostFields } from './services/cost-accounting.js';
|
|
2
|
-
export type TelemetryEvent = 'session_started' | 'session_ended' | 'agent_activated' | 'delegation_begin' | 'delegation_end' | 'task_state_changed' | 'gate_passed' | 'gate_failed' | 'gate_parse_error' | 'reviewer_gate_decision' | 'phase_changed' | 'budget_updated' | 'model_fallback' | 'hard_limit_hit' | 'revision_limit_hit' | 'loop_detected' | 'scope_violation' | 'qa_skip_violation' | 'heartbeat' | 'turbo_mode_changed' | 'auto_oversight_escalation' | 'environment_detected' | 'evidence_lock_acquired' | 'evidence_lock_contended' | 'evidence_lock_stale_recovered' | 'plan_ledger_cas_retry' | 'plan_md_write_failed' | 'snapshot_failed' | '
|
|
2
|
+
export type TelemetryEvent = 'session_started' | 'session_ended' | 'agent_activated' | 'delegation_begin' | 'delegation_end' | 'task_state_changed' | 'gate_passed' | 'gate_failed' | 'gate_parse_error' | 'reviewer_gate_decision' | 'phase_changed' | 'budget_updated' | 'model_fallback' | 'hard_limit_hit' | 'revision_limit_hit' | 'loop_detected' | 'no_op_strong_warning' | 'scope_violation' | 'qa_skip_violation' | 'heartbeat' | 'turbo_mode_changed' | 'auto_oversight_escalation' | 'environment_detected' | 'evidence_lock_acquired' | 'evidence_lock_contended' | 'evidence_lock_stale_recovered' | 'plan_ledger_cas_retry' | 'plan_md_write_failed' | 'snapshot_failed' | 'gate_denial_loop'
|
|
3
|
+
/**
|
|
4
|
+
* Issue #2063 B5 — an ARMED execution episode reached the advisory rung:
|
|
5
|
+
* `execution_stall_warn_calls` tool calls with no delegation completion, no
|
|
6
|
+
* file write, no `update_task_status`, and no workspace change. Emitted once
|
|
7
|
+
* per non-progress streak.
|
|
8
|
+
*/
|
|
9
|
+
| 'execution_stall_warning'
|
|
10
|
+
/**
|
|
11
|
+
* Issue #2063 B5 — the same episode reached `execution_stall_stop_calls` and
|
|
12
|
+
* a non-productive tool (read/glob/grep/bash/shell) was hard-denied. Emitted
|
|
13
|
+
* once per non-progress streak; the denial itself repeats until a progress
|
|
14
|
+
* event clears the rung.
|
|
15
|
+
*/
|
|
16
|
+
| 'execution_stall_denied'
|
|
17
|
+
/**
|
|
18
|
+
* Issue #2063 B4 — a read/glob/grep/bash call resolved to a path inside the
|
|
19
|
+
* INSTALLED opencode-swarm package and was denied. `target` is relative to
|
|
20
|
+
* the package root so no user home path is written to the ledger.
|
|
21
|
+
*/
|
|
22
|
+
| 'swarm_internals_read_denied' | 'prm_pattern_detected' | 'prm_course_correction_injected' | 'prm_escalation_triggered' | 'prm_hard_stop'
|
|
23
|
+
/**
|
|
24
|
+
* Issue #2063 C2 — DELIVERY of a PRM hard stop, as distinct from the
|
|
25
|
+
* `prm_hard_stop` TRIGGER emitted by `src/prm/escalation.ts`. A trigger with
|
|
26
|
+
* no matching delivery means the containment never reached the agent.
|
|
27
|
+
*/
|
|
28
|
+
| 'prm_hard_stop_delivered';
|
|
3
29
|
/** Stable classification for how a reviewer-gate decision was established. */
|
|
4
30
|
export type ReviewerGateEvidenceKind = 'genuine' | 'fallback' | 'data_quality' | 'block';
|
|
5
31
|
/**
|
|
@@ -68,6 +94,37 @@ export declare const telemetry: {
|
|
|
68
94
|
hardLimitHit(sessionId: string, agentName: string, limitType: string, value: number): void;
|
|
69
95
|
revisionLimitHit(sessionId: string, agentName: string): void;
|
|
70
96
|
loopDetected(sessionId: string, agentName: string, loopType: string): void;
|
|
97
|
+
/**
|
|
98
|
+
* Issue #2063 B2 — stage 2 of the no-op ladder fired: the session has made
|
|
99
|
+
* `count` consecutive tool calls with no file write and no subagent dispatch,
|
|
100
|
+
* at or beyond 2× `no_op_warning_threshold`.
|
|
101
|
+
*/
|
|
102
|
+
noOpStrongWarning(sessionId: string, agentName: string, count: number, threshold: number): void;
|
|
103
|
+
/**
|
|
104
|
+
* Issue #2063 B1 — a fail-closed `tool.execute.before` denial streak reached
|
|
105
|
+
* the hard rung: `count` consecutive denials with the same classification
|
|
106
|
+
* (`code`) for the same tool in the same session. Emitted from
|
|
107
|
+
* `src/hooks/gate-denial-tracker.ts` at every denial at or past the rung, so
|
|
108
|
+
* the ledger shows how long the model kept retrying after being told to stop.
|
|
109
|
+
*/
|
|
110
|
+
gateDenialLoop(sessionId: string, tool: string, code: string, count: number): void;
|
|
111
|
+
/**
|
|
112
|
+
* Issue #2063 B5 — advisory rung of the execution-stall ladder. Emitted from
|
|
113
|
+
* `src/hooks/guardrails/execution-stall.ts` once per non-progress streak, so
|
|
114
|
+
* a warning with no matching `execution_stall_denied` means the agent
|
|
115
|
+
* recovered on its own.
|
|
116
|
+
*/
|
|
117
|
+
executionStallWarning(sessionId: string, count: number, threshold: number): void;
|
|
118
|
+
/**
|
|
119
|
+
* Issue #2063 B5 — hard rung of the execution-stall ladder. `tool` is the
|
|
120
|
+
* normalized name of the denied non-productive tool.
|
|
121
|
+
*/
|
|
122
|
+
executionStallDenied(sessionId: string, tool: string, count: number, threshold: number): void;
|
|
123
|
+
/**
|
|
124
|
+
* Issue #2063 B4 — a call targeting the installed plugin package was denied.
|
|
125
|
+
* `target` is package-root-relative (never an absolute user path).
|
|
126
|
+
*/
|
|
127
|
+
swarmInternalsReadDenied(sessionId: string, tool: string, target: string): void;
|
|
71
128
|
scopeViolation(sessionId: string, agentName: string, file: string, reason: string): void;
|
|
72
129
|
qaSkipViolation(sessionId: string, agentName: string, skipCount: number): void;
|
|
73
130
|
heartbeat(sessionId: string): void;
|
|
@@ -78,6 +135,14 @@ export declare const telemetry: {
|
|
|
78
135
|
prmCourseCorrectionInjected(sessionId: string, pattern: string, level: number): void;
|
|
79
136
|
prmEscalationTriggered(sessionId: string, pattern: string, level: number, occurrenceCount: number): void;
|
|
80
137
|
prmHardStop(sessionId: string, pattern: string, level: number, occurrenceCount: number): void;
|
|
138
|
+
/**
|
|
139
|
+
* Issue #2063 C2 — the PRM hard-stop DENIAL was actually delivered to the
|
|
140
|
+
* agent (thrown by the guardrails `toolBefore` consumer). `prm_hard_stop`
|
|
141
|
+
* above records the TRIGGER and is emitted solely by
|
|
142
|
+
* `src/prm/escalation.ts`; a trigger without a matching delivery means the
|
|
143
|
+
* containment armed but never reached the model.
|
|
144
|
+
*/
|
|
145
|
+
prmHardStopDelivered(sessionId: string, pattern: string, level: number, occurrenceCount: number): void;
|
|
81
146
|
};
|
|
82
147
|
/**
|
|
83
148
|
* Test-only dependency-injection seam. Production code calls
|
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
import { z } from 'zod';
|
|
2
|
-
import { resolveExactMergeBase, resolvePrWorkflowRevisionDigest } from '../background/workspace-snapshot.js';
|
|
2
|
+
import { resolveExactMergeBase, resolvePrWorkflowRevisionDigest, resolvePrWorkflowRevisionDigestAsync } from '../background/workspace-snapshot.js';
|
|
3
3
|
import { createSwarmTool } from './create-tool';
|
|
4
4
|
export { PR_REVIEW_TRIGGER_DEFINITIONS } from '../background/pr-review-trigger-contract.js';
|
|
5
5
|
export declare const _internals: {
|
|
6
6
|
resolvePrWorkflowRevisionDigest: typeof resolvePrWorkflowRevisionDigest;
|
|
7
|
+
resolvePrWorkflowRevisionDigestAsync: typeof resolvePrWorkflowRevisionDigestAsync;
|
|
7
8
|
resolveMergeBase: typeof resolveExactMergeBase;
|
|
8
9
|
};
|
|
9
10
|
declare const WritePrReviewTriggerEvalArgsSchema: z.ZodObject<{
|
package/dist/types/events.d.ts
CHANGED
|
@@ -142,4 +142,17 @@ export interface PrmHardStopEvent {
|
|
|
142
142
|
level: number;
|
|
143
143
|
occurrenceCount: number;
|
|
144
144
|
}
|
|
145
|
-
|
|
145
|
+
/**
|
|
146
|
+
* Issue #2063 C2 — the hard stop was DELIVERED (denial thrown at the agent),
|
|
147
|
+
* as opposed to {@link PrmHardStopEvent} which records that it was TRIGGERED.
|
|
148
|
+
* The pair makes "armed but never reached the model" observable.
|
|
149
|
+
*/
|
|
150
|
+
export interface PrmHardStopDeliveredEvent {
|
|
151
|
+
type: 'prm_hard_stop_delivered';
|
|
152
|
+
timestamp: string;
|
|
153
|
+
sessionId: string;
|
|
154
|
+
pattern: string;
|
|
155
|
+
level: number;
|
|
156
|
+
occurrenceCount: number;
|
|
157
|
+
}
|
|
158
|
+
export type V619Event = SoundingBoardConsultedEvent | ArchitectLoopDetectedEvent | PrecedentManipulationDetectedEvent | CoderSelfAuditEvent | CoderRetryCircuitBreakerEvent | AgentConflictDetectedEvent | AuthorityHandoffResolvedEvent | SpecStaleDetectedEvent | SpecDriftAcknowledgedEvent | TaskRemovedEvent | PrmPatternDetectedEvent | PrmCourseCorrectionInjectedEvent | PrmEscalationTriggeredEvent | PrmHardStopEvent | PrmHardStopDeliveredEvent;
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Stable JSON serialization helpers.
|
|
3
|
+
*
|
|
4
|
+
* # Why this exists
|
|
5
|
+
*
|
|
6
|
+
* Several subsystems hash tool-call arguments for repetition / spiral
|
|
7
|
+
* detection. Two requirements make naive `JSON.stringify(value)` incorrect:
|
|
8
|
+
*
|
|
9
|
+
* 1. **Key-order independence.** Two semantically identical objects whose
|
|
10
|
+
* keys were inserted in different order (`{a:1,b:2}` vs `{b:2,a:1}`) must
|
|
11
|
+
* hash equally, otherwise a genuine repetition loop whose args happen to
|
|
12
|
+
* be built with reordered keys is missed.
|
|
13
|
+
*
|
|
14
|
+
* 2. **No nested-key loss.** A `JSON.stringify(value, sortedKeysArray)`
|
|
15
|
+
* property-list replacer looks like it sorts keys — and it does, but only
|
|
16
|
+
* for the top level. At every deeper object it acts as a *filter*,
|
|
17
|
+
* dropping any key not present in the (top-level-derived) list. For nested
|
|
18
|
+
* args like `{todos:[{content,status}]}` every todo collapses to `{}`,
|
|
19
|
+
* re-introducing the exact false-collision class this is meant to prevent.
|
|
20
|
+
*
|
|
21
|
+
* `stableCanonicalStringify` rebuilds each object with sorted keys at every
|
|
22
|
+
* depth (no filtering) and serializes arrays element-wise, producing a stable
|
|
23
|
+
* canonical string suitable for hashing.
|
|
24
|
+
*
|
|
25
|
+
* Originally introduced for the adversarial-detector spiral hash
|
|
26
|
+
* (issue #2060) and shared with `file-authority.hashArgs` so both code paths
|
|
27
|
+
* use one correct implementation.
|
|
28
|
+
*/
|
|
29
|
+
/**
|
|
30
|
+
* Recursively produces a canonical JSON string with object keys sorted at
|
|
31
|
+
* EVERY depth (not just the top level). Arrays are serialized element-wise in
|
|
32
|
+
* index order.
|
|
33
|
+
*
|
|
34
|
+
* Why not `JSON.stringify(value, sortedKeysArray)`: a property-list replacer
|
|
35
|
+
* array acts as a KEY FILTER at every object depth, not just the top level, so
|
|
36
|
+
* any key not in the (top-level-derived) list is silently dropped from nested
|
|
37
|
+
* objects. For tool args like `{todos:[{content,status}]}`, that collapses
|
|
38
|
+
* every todo to `{}`, re-introducing the exact false-collision class this
|
|
39
|
+
* function exists to eliminate. Sorting must be done by rebuilding each object
|
|
40
|
+
* with sorted keys before serialization.
|
|
41
|
+
*
|
|
42
|
+
* Throws on cyclic structures (infinite recursion) and on values that
|
|
43
|
+
* `JSON.stringify` cannot represent (BigInt); callers should wrap in try/catch
|
|
44
|
+
* and fall back to a stable coarse hash.
|
|
45
|
+
*/
|
|
46
|
+
export declare function stableCanonicalStringify(value: unknown): string;
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/*
|
|
3
|
+
* Skill-eval project scorer wrapper (issue #1822 — D1, critic C1).
|
|
4
|
+
*
|
|
5
|
+
* The evaluation substrate invokes `kind:'project'` scorers as isolated
|
|
6
|
+
* subprocesses (runner.ts:175-228), passing:
|
|
7
|
+
* - SWARM_EVAL_TASK_ID, SWARM_EVAL_CANDIDATE_ID, SWARM_EVAL_SEED
|
|
8
|
+
* - SWARM_EVAL_ARTIFACT_DIR (contains model-output.json with the candidate's
|
|
9
|
+
* generated text under { v: 1, text: "..." })
|
|
10
|
+
* - argv: [<this-script>, <phrase-spec-path>]
|
|
11
|
+
*
|
|
12
|
+
* The scorer reads the candidate text from model-output.json, reads the phrase
|
|
13
|
+
* spec (required_phrases / forbidden_phrases), and emits a ScorerOutputV1 line:
|
|
14
|
+
* { "v": 1, "score": <0..1>, "cost": { "source": "unavailable" } }
|
|
15
|
+
*
|
|
16
|
+
* The scoring arithmetic is the SAME as `scoreSkillPhrases` in
|
|
17
|
+
* src/services/skill-evaluator.ts (the source of truth). A parity test
|
|
18
|
+
* (tests/unit/services/skill-evaluator-refactor.test.ts) proves the two agree,
|
|
19
|
+
* so there is no duplicate scorer — one authoritative function, one thin
|
|
20
|
+
* subprocess mirror that must match.
|
|
21
|
+
*
|
|
22
|
+
* Score = requiredHits / max(1, required.length), minus a 1-point penalty if
|
|
23
|
+
* any forbidden phrase is present, clamped to >= 0. Phrase matching is
|
|
24
|
+
* case-insensitive substring.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
'use strict';
|
|
28
|
+
|
|
29
|
+
const fs = require('node:fs');
|
|
30
|
+
const path = require('node:path');
|
|
31
|
+
|
|
32
|
+
function readArtifact() {
|
|
33
|
+
const dir = process.env.SWARM_EVAL_ARTIFACT_DIR;
|
|
34
|
+
if (!dir) throw new Error('missing SWARM_EVAL_ARTIFACT_DIR');
|
|
35
|
+
const file = path.join(dir, 'model-output.json');
|
|
36
|
+
const raw = fs.readFileSync(file, 'utf8');
|
|
37
|
+
const parsed = JSON.parse(raw);
|
|
38
|
+
if (!parsed || typeof parsed.text !== 'string') {
|
|
39
|
+
throw new Error('model-output.json missing text field');
|
|
40
|
+
}
|
|
41
|
+
return parsed.text;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
function readPhraseSpec(specPath) {
|
|
45
|
+
if (!specPath || !fs.existsSync(specPath)) {
|
|
46
|
+
return { required: [], forbidden: [] };
|
|
47
|
+
}
|
|
48
|
+
const parsed = JSON.parse(fs.readFileSync(specPath, 'utf8'));
|
|
49
|
+
return {
|
|
50
|
+
required: Array.isArray(parsed.required_phrases) ? parsed.required_phrases : [],
|
|
51
|
+
forbidden: Array.isArray(parsed.forbidden_phrases) ? parsed.forbidden_phrases : [],
|
|
52
|
+
};
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
function includesPhrase(content, phrase) {
|
|
56
|
+
return content.toLowerCase().includes(String(phrase).toLowerCase());
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function scoreSkillPhrases(content, spec) {
|
|
60
|
+
const required = spec.required;
|
|
61
|
+
const forbidden = spec.forbidden;
|
|
62
|
+
let requiredHits = 0;
|
|
63
|
+
for (const phrase of required) {
|
|
64
|
+
if (includesPhrase(content, phrase)) requiredHits++;
|
|
65
|
+
}
|
|
66
|
+
const requiredScore = required.length === 0 ? 1 : requiredHits / Math.max(1, required.length);
|
|
67
|
+
const forbiddenPenalty = forbidden.some((p) => includesPhrase(content, p)) ? 1 : 0;
|
|
68
|
+
return Math.max(0, requiredScore - forbiddenPenalty);
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
function main() {
|
|
72
|
+
const specPath = process.argv[2];
|
|
73
|
+
const content = readArtifact();
|
|
74
|
+
const spec = readPhraseSpec(specPath);
|
|
75
|
+
const score = scoreSkillPhrases(content, spec);
|
|
76
|
+
process.stdout.write(
|
|
77
|
+
JSON.stringify({ v: 1, score, cost: { source: 'unavailable' } }) + '\n',
|
|
78
|
+
);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
try {
|
|
82
|
+
main();
|
|
83
|
+
} catch (err) {
|
|
84
|
+
// ScorerFailure 'malformed' — the runner treats non-zero exit / bad JSON as a
|
|
85
|
+
// task failure rather than a candidate score. Emit a zero-score with a
|
|
86
|
+
// metadata reason so the run records the failure deterministically.
|
|
87
|
+
process.stderr.write(`score-skill-eval failed: ${err && err.message ? err.message : String(err)}\n`);
|
|
88
|
+
process.stdout.write(
|
|
89
|
+
JSON.stringify({
|
|
90
|
+
v: 1,
|
|
91
|
+
score: 0,
|
|
92
|
+
cost: { source: 'unavailable' },
|
|
93
|
+
metadata: { failure: 'scorer-error', reason: String(err && err.message ? err.message : err).slice(0, 200) },
|
|
94
|
+
}) + '\n',
|
|
95
|
+
);
|
|
96
|
+
process.exitCode = 0;
|
|
97
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "opencode-swarm",
|
|
3
|
-
"version": "7.136.
|
|
3
|
+
"version": "7.136.3",
|
|
4
4
|
"description": "Architect-centric agentic swarm plugin for OpenCode - hub-and-spoke orchestration with SME consultation, code generation, and QA review",
|
|
5
5
|
"main": "dist/index.js",
|
|
6
6
|
"types": "dist/index.d.ts",
|
|
@@ -94,6 +94,7 @@
|
|
|
94
94
|
"lint:ci": "biome ci .",
|
|
95
95
|
"test:unit:ci": "bun scripts/ci/run-unit-tests-local.ts",
|
|
96
96
|
"drift:check": "bun run scripts/drift-check.ts",
|
|
97
|
+
"check:runtime-src-refs": "bun run scripts/check-runtime-src-refs.ts",
|
|
97
98
|
"drift:fix": "bun run scripts/drift-check.ts --fix --confirm",
|
|
98
99
|
"skills:sync": "bun run scripts/sync-qa-gate-skills.ts",
|
|
99
100
|
"format": "biome format . --write",
|