jules-orchestrator-kit 0.33.0 → 0.35.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -98,7 +98,9 @@ Autonomous coding agents can write software at 100× human speed—but unconstra
98
98
 
99
99
  * **⚡ AST Blast-Radius Selective Testing:** Traverses file import dependency graphs to execute only affected downstream test suites, cutting monorepo test latency from minutes to milliseconds.
100
100
 
101
- * **🚨 Asynchronous HITL Escalation Bridge (`agentctl escalate`):** Dispatches Slack & Discord webhook alerts when Jules needs feedback, allowing engineers to unblock agents asynchronously via `agentctl resume <id> --response "<reply>"`.
101
+ * **🔕 Type III Silence Governor & Interruption Budgeting (`v0.35.0`):** `agentctl escalate` manages Slack and Discord webhook alerts with configurable digest modes and hourly interruption budgets, buffering non-critical notifications while guaranteeing zero-latency delivery for critical escalations.
102
+
103
+ * **🩹 Automated Flaky Test Healing Swarm (`v0.35.0`):** `agentctl flaky heal` automatically consumes Wilson-quarantined tests (Exit Code 8) and dispatches specialized anti-flakiness repair tasks to eliminate race conditions, async timing leaks, and resource collisions without weakening test assertions.
102
104
 
103
105
  * **🔒 Zero Runtime Dependencies:** Built exclusively on Node.js 20+ built-ins (`node:fs`, `node:child_process`, `node:crypto`, `node:path`, `node:http`, `node:tty`, `node:test`). Zero third-party npm packages mean zero supply-chain CVE risk.
104
106
 
@@ -120,7 +122,7 @@ Autonomous coding agents can write software at 100× human speed—but unconstra
120
122
 
121
123
  * **🚀 Zero-Test Bootstrapping (`agentctl bootstrap`):** Synthesizes deterministic syntax-check and smoke-test verification oracles for untested legacy repositories so agents always operate against a falsifiable feedback loop.
122
124
 
123
- * **📈 Proven Scale & Reliability:** Empirically tested with **498 unit tests across 75 suites passing in < 10.0s**. An adversarial red-team suite (`test/adversarial-claims.test.mjs`) continuously attempts to falsify the safety guarantees documented above — including cross-platform probes for the case-insensitive filesystems on macOS and Windows — and a documentation-sync gate (`scripts/doc-sync-check.mjs`) blocks any release whose docs have drifted from the code.
125
+ * **📈 Proven Scale & Reliability:** Empirically tested with **526 unit tests across 79 suites passing in < 10.0s**. An adversarial red-team suite (`test/adversarial-claims.test.mjs`) continuously attempts to falsify the safety guarantees documented above — including cross-platform probes for the case-insensitive filesystems on macOS and Windows — and a documentation-sync gate (`scripts/doc-sync-check.mjs`) blocks any release whose docs have drifted from the code.
124
126
 
125
127
  <br/>
126
128
 
@@ -297,9 +299,9 @@ scope:
297
299
  limits:
298
300
  diffKb: 75 # 75 KB Diff Payload Governor limit
299
301
  promptKb: 50 # Maximum prompt payload size
300
- dailyTasks: 300 # Daily task session quota limit
302
+ dailyTasks: 300 # Task quota per rolling 24h window (not per calendar day)
301
303
  repairAttempts: 3 # Maximum OODA repair iterations
302
- concurrency: 1 # Worker slot concurrency limit
304
+ concurrency: 15 # Worker slots; the ultra plan allows up to 60
303
305
 
304
306
  # Dynamic Complexity & Cost Router — opt-in, disabled by default.
305
307
  # Provider-agnostic: "fast"/"complex" accept any provider key ("jules" |
@@ -404,9 +406,11 @@ Native stdio server exposing task dispatch, gate verification, and risk auditing
404
406
  | `lock` | `agentctl lock <acquire\|release\|status>`| Manages VFS mutex locks for multi-agent non-overlapping file ownership. | `0` (Locked/Released), `1` (Conflict) |
405
407
  | `clean` | `agentctl clean` | Prunes stale git worktrees, lockfiles, and temporary ledgers. | `0` (Clean) |
406
408
  | `evidence` | `agentctl evidence <generate\|verify\|show> [--manifest <path>] [--json]` | Generates, verifies, or prints a SHA-256 cryptographic evidence manifest (changed-file hashes + test-file tamper lock) for audit trails. | `0` (Verified/Generated), `1` (Tamper detected / Verification failed) |
409
+ | `escalate` | `agentctl escalate [<sessionId>] [--status] [--flush] [--clear]` | Dispatches or manages webhook escalation incidents across Slack and Discord with Type III Silence Governor and interruption budgeting. | `0` (Dispatched/Buffered) |
410
+ | `flaky` | `agentctl flaky <status\|heal\|reset> [--dispatch] [--dry-run] [--role <name>]` | Manages Wilson-quarantined tests (Exit Code 8) and dispatches automated anti-flakiness healing swarm without assertion weakening. | `0` (Healed/Queued/Listed) |
407
411
  | `mcp` | `agentctl mcp` | Starts stdio Model Context Protocol (MCP) server for tool integration. | `0` / Stdio stream |
408
412
  | `mcp init` | `agentctl mcp init [--target cursor\|vscode\|claude\|all]` | 1-click scaffolding for Cursor (`.cursor/mcp.json`), VS Code tasks (`tasks.json`), and Claude Desktop. | `0` (Scaffolded) |
409
- | `version` | `agentctl version` | Outputs orchestrator kit semantic version (`v0.32.5`). | `0` |
413
+ | `version` | `agentctl version` | Outputs orchestrator kit semantic version. | `0` |
410
414
 
411
415
  <br/>
412
416
 
package/bin/agentctl.mjs CHANGED
@@ -9,7 +9,7 @@ import { acquireLock, releaseLock, lockStatus, getQueueDir } from "../src/state.
9
9
  import { worktreePrune } from "../src/git.mjs";
10
10
  import { reapOrphanedIntents, reapStaleMutexDirs } from "../src/journal.mjs";
11
11
  import { KIT_VERSION } from "../src/version.mjs";
12
- import { budgetStatus, listOpenReservations, releaseOpenReservations } from "../src/budget.mjs";
12
+ import { budgetStatus, listOpenReservations, releaseOpenReservations, resolveConcurrency } from "../src/budget.mjs";
13
13
 
14
14
  const args = process.argv.slice(2);
15
15
  const command = args[0];
@@ -21,11 +21,14 @@ export const VERSION = KIT_VERSION;
21
21
  * counter. The ledger counts what *this checkout* dispatched; sessions started
22
22
  * from the web UI or another machine spend the same quota unseen, so a bare
23
23
  * "N / M used" invites the reader to trust a figure that cannot be complete.
24
+ * "last 24h", not "today": the provider's allowance resets on a rolling
25
+ * window, so a figure labelled by the calendar day would be a different number
26
+ * from the one being enforced.
24
27
  * @param {{ used: number, limit: number, source: string, certain: boolean }} b
25
28
  */
26
29
  export function formatBudgetLine(b) {
27
- const scope = `${b.used} / ${b.limit} used (this repo)`;
28
- if (b.source === "learned") return `${scope} — provider refused further work today`;
30
+ const scope = `${b.used} / ${b.limit} used in the last 24h (this repo)`;
31
+ if (b.source === "learned") return `${scope} — provider refused further work`;
29
32
  if (b.certain) return `${scope} — limit from ${b.source === "env" ? "JULES_DAILY_BUDGET" : "config"}`;
30
33
  return `${scope} — limit estimated from tier "${b.tier || "?"}", not enforced`;
31
34
  }
@@ -56,8 +59,10 @@ Commands:
56
59
  mcp init Scaffold IDE integration config (cursor | vscode | claude | all)
57
60
  rollback Restore git state & working tree to atomic pre-flight checkpoint
58
61
  resume Resume warm session with human response (--response "<text>")
62
+ escalate Dispatch or manage webhook escalation incidents (--flush, --status, --clear)
63
+ flaky Manage Wilson-quarantined tests and dispatch healing swarm (status | heal | reset)
59
64
  status Display queue and system status summary
60
- budget Show today's task budget and its provenance (reset --yes to reconcile)
65
+ budget Show the 24h task budget, worker slots and their provenance (reset --yes)
61
66
  scan Scan codebase for TODO/FIXME task candidates
62
67
  hydrate [prompt] Prepend active system learnings and baton-pass state to a prompt
63
68
  harvest Harvest failure traces and record/quarantine resolution rules
@@ -333,7 +338,7 @@ async function main() {
333
338
  const confirmed = args.includes("--yes") || args.includes("-y");
334
339
  if (!dryRun && !confirmed) {
335
340
  const open = listOpenReservations(root);
336
- console.log(`Would release ${open.length} open reservation(s) from today's ledger.`);
341
+ console.log(`Would release ${open.length} open reservation(s) from the last 24 hours.`);
337
342
  console.log("This rewrites nothing — it appends `budget_released` entries.");
338
343
  console.log("Re-run with --yes to confirm, or --dry-run for detail.");
339
344
  process.exit(0);
@@ -353,9 +358,15 @@ async function main() {
353
358
  process.exit(0);
354
359
  }
355
360
 
356
- console.log(`Daily Budget : ${formatBudgetLine(b)}`);
361
+ const slots = resolveConcurrency(config);
362
+ console.log(`Task Budget : ${formatBudgetLine(b)}`);
357
363
  console.log(` ${b.note}`);
358
- console.log(` Open reservations today: ${listOpenReservations(root).length}`);
364
+ console.log(` Window opened at ${b.windowStart} — the quota resets ${b.windowHours}h after each task,`);
365
+ console.log(" not at midnight, so this count spans yesterday's ledger too.");
366
+ console.log(` Open reservations in the window: ${listOpenReservations(root).length}`);
367
+ console.log("");
368
+ console.log(`Worker Slots : ${slots.concurrency} concurrent`);
369
+ console.log(` ${slots.note}`);
359
370
  console.log("");
360
371
  console.log("The ledger counts this checkout only — sessions started from the Jules");
361
372
  console.log("web UI or another machine spend the same quota without appearing here.");
@@ -770,6 +781,244 @@ async function main() {
770
781
  break;
771
782
  }
772
783
 
784
+ case "escalate": {
785
+ const {
786
+ dispatchEscalation,
787
+ flushEscalationDigest,
788
+ getEscalationDigestStatus,
789
+ clearEscalationDigest,
790
+ } = await import("../src/webhook.mjs");
791
+
792
+ const { values, positionals } = parseArgs({
793
+ args: args.slice(1),
794
+ options: {
795
+ reason: { type: "string", short: "r", default: "AWAITING_USER_FEEDBACK" },
796
+ branch: { type: "string", short: "b", default: config.baseBranch || "main" },
797
+ logs: { type: "string", short: "l" },
798
+ "log-file": { type: "string" },
799
+ critical: { type: "boolean" },
800
+ flush: { type: "boolean" },
801
+ status: { type: "boolean" },
802
+ clear: { type: "boolean" },
803
+ "dry-run": { type: "boolean", short: "d" },
804
+ json: { type: "boolean", short: "j" },
805
+ },
806
+ allowPositionals: true,
807
+ });
808
+
809
+ if (values.status) {
810
+ const st = getEscalationDigestStatus(root, config);
811
+ if (values.json) {
812
+ console.log(JSON.stringify({ ok: true, status: st }, null, 2));
813
+ } else {
814
+ console.log(`\n🔇 Type III Silence Governor Status (v${VERSION})`);
815
+ console.log(`--------------------------------------------------`);
816
+ console.log(` Notification Mode : ${st.mode.toUpperCase()}`);
817
+ console.log(` Pending Digest Count : ${st.pendingCount} / ${st.threshold}`);
818
+ console.log(` Interruption Budget : ${st.recentInterruptions} / ${st.budgetPerHour} per hour (Available: ${st.budgetAvailable})`);
819
+ if (st.createdAt) {
820
+ console.log(` Oldest Buffered Item : ${st.createdAt}`);
821
+ }
822
+ if (st.incidents.length > 0) {
823
+ console.log(`\n Buffered Incidents (${st.incidents.length}):`);
824
+ st.incidents.forEach((inc) => {
825
+ console.log(` - [${inc.reason}] Session ${inc.sessionId} (${inc.branch})`);
826
+ });
827
+ }
828
+ console.log(`--------------------------------------------------\n`);
829
+ }
830
+ process.exit(0);
831
+ }
832
+
833
+ if (values.clear) {
834
+ const res = clearEscalationDigest(root);
835
+ if (values.json) {
836
+ console.log(JSON.stringify(res, null, 2));
837
+ } else {
838
+ console.log(`✅ Cleared pending escalation digest buffer.`);
839
+ }
840
+ process.exit(0);
841
+ }
842
+
843
+ if (values.flush) {
844
+ const res = await flushEscalationDigest(config, { root, dryRun: values["dry-run"] });
845
+ if (values.json) {
846
+ console.log(JSON.stringify(res, null, 2));
847
+ } else {
848
+ if (res.flushed) {
849
+ console.log(`\n📢 Flushed Escalation Digest (${res.count} incidents)`);
850
+ console.log(` Slack : ${res.slack ? "✅ Delivered" : "❌ Skipped/Failed"}`);
851
+ console.log(` Discord : ${res.discord ? "✅ Delivered" : "❌ Skipped/Failed"}\n`);
852
+ } else {
853
+ console.log(`ℹ️ Nothing to flush: ${res.reason}`);
854
+ }
855
+ }
856
+ process.exit(0);
857
+ }
858
+
859
+ const sessionId = positionals[0] || values.session;
860
+ if (!sessionId) {
861
+ console.error("Error: Session ID is required for agentctl escalate <sessionId> (or use --flush / --status).");
862
+ process.exit(1);
863
+ }
864
+
865
+ let logContent = values.logs || "";
866
+ if (values["log-file"] && existsSync(values["log-file"])) {
867
+ logContent = readFileSync(values["log-file"], "utf-8");
868
+ }
869
+
870
+ const incident = {
871
+ sessionId,
872
+ branch: values.branch,
873
+ reason: values.reason,
874
+ logs: logContent,
875
+ critical: values.critical,
876
+ };
877
+
878
+ const res = await dispatchEscalation(incident, {
879
+ root,
880
+ config,
881
+ dryRun: values["dry-run"],
882
+ });
883
+
884
+ if (values.json) {
885
+ console.log(JSON.stringify({ ok: true, result: res }, null, 2));
886
+ } else {
887
+ if (res.buffered) {
888
+ console.log(`\n🔇 Incident buffered by Silence Governor (${res.reason})`);
889
+ console.log(` Session ID : ${sessionId}`);
890
+ console.log(` Digest Count : ${res.digestCount || 1}`);
891
+ console.log(` (Use 'agentctl escalate --flush' to deliver immediately)\n`);
892
+ } else if (res.dispatched) {
893
+ console.log(`\n🚨 Incident Escalation Dispatched!`);
894
+ console.log(` Session ID : ${sessionId}`);
895
+ console.log(` Reason : ${values.reason}`);
896
+ console.log(` Slack : ${res.slack ? "✅ Sent" : "N/A"}`);
897
+ console.log(` Discord : ${res.discord ? "✅ Sent" : "N/A"}\n`);
898
+ } else {
899
+ console.log(`⚠️ Escalation not dispatched: ${res.reason}`);
900
+ }
901
+ }
902
+ process.exit(0);
903
+ break;
904
+ }
905
+
906
+ case "flaky": {
907
+ const {
908
+ listQuarantinedTests,
909
+ clearFlakyLedger,
910
+ runFlakyHealingSwarm,
911
+ synthesizeFlakyHealingTask,
912
+ } = await import("../src/flaky-ledger.mjs");
913
+
914
+ const subAction = args[1] || "status";
915
+ const { values, positionals } = parseArgs({
916
+ args: args.slice(2),
917
+ options: {
918
+ dispatch: { type: "boolean" },
919
+ role: { type: "string", short: "r", default: "janitor" },
920
+ "test-cmd": { type: "string", short: "t" },
921
+ "dry-run": { type: "boolean", short: "d" },
922
+ json: { type: "boolean", short: "j" },
923
+ },
924
+ allowPositionals: true,
925
+ });
926
+
927
+ if (subAction === "status" || subAction === "list") {
928
+ const quarantined = listQuarantinedTests(root);
929
+ if (values.json) {
930
+ console.log(JSON.stringify({ ok: true, quarantined, count: quarantined.length }, null, 2));
931
+ } else {
932
+ console.log(`\n🧪 Statistical Flaky Test Quarantine (v${VERSION})`);
933
+ console.log(`--------------------------------------------------`);
934
+ if (quarantined.length === 0) {
935
+ console.log(` ✅ Zero quarantined tests. All suites stable.`);
936
+ } else {
937
+ console.log(` Found ${quarantined.length} test suite(s) quarantined (Exit Code 8):\n`);
938
+ quarantined.forEach((q, idx) => {
939
+ const oscPct = Math.round(q.oscillation * 100);
940
+ console.log(` ${idx + 1}. \`${q.testCmd}\``);
941
+ console.log(` Oscillation: ${oscPct}% (${q.fails} fails / ${q.passes} passes in last ${q.n} runs)`);
942
+ console.log(` Wilson CI : [${q.wilson.lower.toFixed(2)}, ${q.wilson.upper.toFixed(2)}]`);
943
+ console.log(` Last Run : ${q.lastRunTimestamp}\n`);
944
+ });
945
+ console.log(` 👉 To dispatch auto-healing swarm, run: agentctl flaky heal`);
946
+ }
947
+ console.log(`--------------------------------------------------\n`);
948
+ }
949
+ process.exit(0);
950
+ }
951
+
952
+ if (subAction === "heal") {
953
+ const targetCmd = positionals[0] || values["test-cmd"];
954
+ let res;
955
+ if (targetCmd) {
956
+ const taskPlan = synthesizeFlakyHealingTask({ testCmd: targetCmd }, { role: values.role });
957
+ if (values.dispatch && !values["dry-run"]) {
958
+ const { dispatch } = await import("../src/engine.mjs");
959
+ try {
960
+ const session = await dispatch(
961
+ { title: taskPlan.title, prompt: taskPlan.prompt, role: taskPlan.role },
962
+ { root, config }
963
+ );
964
+ taskPlan.session = session;
965
+ taskPlan.dispatched = true;
966
+ } catch (err) {
967
+ taskPlan.dispatchError = err.message;
968
+ taskPlan.dispatched = false;
969
+ }
970
+ } else if (!values["dry-run"]) {
971
+ const { writeFileSync } = await import("node:fs");
972
+ const queueDir = getQueueDir(root);
973
+ const filePath = join(queueDir, `${taskPlan.taskId}.md`);
974
+ writeFileSync(filePath, taskPlan.fullEnvelope, "utf-8");
975
+ taskPlan.taskFile = filePath;
976
+ taskPlan.queued = true;
977
+ }
978
+ res = { count: 1, tasks: [taskPlan], dryRun: values["dry-run"] };
979
+ } else {
980
+ res = await runFlakyHealingSwarm(root, {
981
+ dispatch: values.dispatch,
982
+ dryRun: values["dry-run"],
983
+ role: values.role,
984
+ });
985
+ }
986
+
987
+ if (values.json) {
988
+ console.log(JSON.stringify({ ok: true, ...res }, null, 2));
989
+ } else {
990
+ if (res.count === 0) {
991
+ console.log(`ℹ️ ${res.message || "No quarantined tests to heal."}`);
992
+ } else {
993
+ console.log(`\n🩹 Flaky Test Healing Swarm (${res.count} task${res.count > 1 ? "s" : ""})`);
994
+ console.log(`--------------------------------------------------`);
995
+ res.tasks.forEach((t) => {
996
+ console.log(` • ${t.title}`);
997
+ if (t.taskFile) console.log(` Queued: ${t.taskFile}`);
998
+ if (t.session) console.log(` Dispatched Session: ${t.session.id}`);
999
+ });
1000
+ console.log(`--------------------------------------------------\n`);
1001
+ }
1002
+ }
1003
+ process.exit(0);
1004
+ }
1005
+
1006
+ if (subAction === "reset" || subAction === "clear") {
1007
+ const targetCmd = positionals[0] || values["test-cmd"] || null;
1008
+ const res = clearFlakyLedger(root, targetCmd);
1009
+ if (values.json) {
1010
+ console.log(JSON.stringify(res, null, 2));
1011
+ } else {
1012
+ console.log(`✅ Flaky test ledger reset (${targetCmd ? `command: ${targetCmd}` : "all tests"}).`);
1013
+ }
1014
+ process.exit(0);
1015
+ }
1016
+
1017
+ console.error(`Unknown flaky subaction: '${subAction}'. Supported: agentctl flaky status | heal | reset`);
1018
+ process.exit(1);
1019
+ break;
1020
+ }
1021
+
773
1022
  case "test-gen": {
774
1023
  const { scaffoldTddTest, runTddCycle } = await import("../src/ops/tdd-generator.mjs");
775
1024
  const { values } = parseArgs({
package/index.mjs CHANGED
@@ -49,6 +49,9 @@ export {
49
49
  rollbackBudgetReservation,
50
50
  withBudget,
51
51
  checkDailyBudget,
52
+ scanBudgetWindow,
53
+ getLedgerPathsInWindow,
54
+ ROLLING_WINDOW_MS,
52
55
  acquireLock,
53
56
  releaseLock,
54
57
  lockStatus,
@@ -100,8 +103,19 @@ export { scorePromptFalsifiability, optimizeTaskPrompt, levenshteinDistance, ext
100
103
  // Atomic Git Checkpoints & Rollback
101
104
  export { createCheckpoint, restoreCheckpoint, listCheckpoints, pruneCheckpoints, CheckpointError } from "./src/ops/checkpoint.mjs";
102
105
 
103
- // Webhook & HITL Escalation Bridge
104
- export { dispatchEscalation, verifySignature, parseWebhookPayload, routeWebhookEvent, createWebhookServer } from "./src/webhook.mjs";
106
+ // Webhook, Silence Governor & HITL Escalation Bridge
107
+ export {
108
+ dispatchEscalation,
109
+ verifySignature,
110
+ parseWebhookPayload,
111
+ routeWebhookEvent,
112
+ createWebhookServer,
113
+ flushEscalationDigest,
114
+ getEscalationDigestStatus,
115
+ clearEscalationDigest,
116
+ bufferEscalationIncident,
117
+ DEFAULT_CRITICAL_REASONS,
118
+ } from "./src/webhook.mjs";
105
119
 
106
120
  // PR Review Evidence Bundler & Dev Server Probe
107
121
  export { synthesizePrDescription, probeDevServer } from "./src/engine.mjs";
@@ -140,8 +154,22 @@ export {
140
154
  isDailyQuotaRejection,
141
155
  listOpenReservations,
142
156
  releaseOpenReservations,
157
+ resolveConcurrency,
143
158
  CEILING_FILE,
144
159
  } from "./src/budget.mjs";
145
160
  export { KIT_VERSION } from "./src/version.mjs";
146
161
  export { VENDOR_TIERS, FALLBACK_TIER } from "./src/config.mjs";
147
162
  export { tierOptions } from "./src/wizard-init.mjs";
163
+
164
+ // Statistical Flaky Ledger & Automated Healing Swarm
165
+ export {
166
+ wilsonScoreInterval,
167
+ computeOscillation,
168
+ recordVerifyRun,
169
+ readVerifyRuns,
170
+ flakyVerdict,
171
+ listQuarantinedTests,
172
+ clearFlakyLedger,
173
+ synthesizeFlakyHealingTask,
174
+ runFlakyHealingSwarm,
175
+ } from "./src/flaky-ledger.mjs";
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "jules-orchestrator-kit",
3
- "version": "0.33.0",
3
+ "version": "0.35.0",
4
4
  "description": "Orchestration kit for running Google Jules autonomous agents.",
5
5
  "repository": {
6
6
  "type": "git",