jules-orchestrator-kit 0.41.0 → 0.41.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -51,16 +51,28 @@
51
51
  Get running in any repository in 3 commands (zero configuration required):
52
52
 
53
53
  ```bash
54
- # 1. Initialize orchestrator in your project (auto-detects Python, Rust, Go, Node, PHP, etc.)
54
+ # 1. Scaffold config, AGENTS.md, role prompts and guardrails
55
+ # (auto-detects Python, Rust, Go, Node, PHP, etc.)
55
56
  npx jules-orchestrator-kit init
57
+ ```
56
58
 
57
- # 2. Author a scoped, verified task envelope with guardrails & secret scrubbing
58
- npx jules-orchestrator-kit task create
59
+ ```bash
60
+ # 2. Commit what init wrote — .agent/config.yml is on the gate's deny list by
61
+ # design, so leaving it uncommitted makes the first gate reject your tree
62
+ git add .agent AGENTS.md .gitignore && git commit -m "chore: add agent config"
63
+ ```
59
64
 
60
- # 3. Inspect repository health & diagnostic status
61
- npx jules-orchestrator-kit doctor
65
+ ```bash
66
+ # 3. Author a scoped, verified task envelope with guardrails & secret scrubbing
67
+ npx jules-orchestrator-kit task create
62
68
  ```
63
69
 
70
+ > [!TIP]
71
+ > **Not sure what to run next?**
72
+ > `agentctl` with no arguments reads the repository state and prints the single
73
+ > next step — missing git repo, missing API key, empty queue, tasks ready to
74
+ > dispatch — instead of a wall of commands.
75
+
64
76
  > [!TIP]
65
77
  > **Prefer a global CLI?**
66
78
  > Install globally to access `agentctl` directly:
@@ -131,7 +143,7 @@ To maximize PR merge rates, dispatch tasks according to deterministic boundaries
131
143
  * **Fail-Closed Security & Secret Redaction:** Evaluates explicit Deny rules before Allow rules against canonicalized, case-folded paths. Redacts high-entropy keys and base64-encoded credentials (such as Kubernetes `Secret` manifests).
132
144
  * **Complexity & Cost Router:** Zero-dependency heuristic classifier (`src/router.mjs`) routing mechanical tasks to lightweight models while reserving primary models for complex refactors, with a `node --check` syntax-verification gate that transparently escalates a FAST-tier result to the primary provider if it left broken JS on disk.
133
145
  * **Terminal UI & Diagnostic Matrix (`agentctl doctor`):** Interactive terminal dashboard, task sidecar manager, and automated transactional self-repair.
134
- * **Verified Test Suite:** Tested with **671 unit tests across 84 suites passing in < 10.0s**.
146
+ * **Verified Test Suite:** Tested with **706 unit tests across 85 suites passing in < 10.0s**.
135
147
 
136
148
  <br/>
137
149
 
@@ -146,18 +158,18 @@ To maximize PR merge rates, dispatch tasks according to deterministic boundaries
146
158
 
147
159
  | Command | Usage | Description | Exit Codes |
148
160
  | :--- | :--- | :--- | :--- |
149
- | `init` | `agentctl init [--interactive] [--tier pro]` | Interactive onboarding wizard & stack detector generating `.agent/config.yml`. | `0` (Created) |
161
+ | `init` | `agentctl init [--interactive] [--tier pro] [--force]` | Interactive onboarding wizard & stack detector. Generates `.agent/config.yml` and scaffolds `AGENTS.md`, the role prompts, the guardrails and the runtime `.gitignore` entries. Existing files are preserved unless `--force`. | `0` (Created) |
150
162
  | `budget` | `agentctl budget [--by-user] [--json] [reset]` | Reports rolling 24h task budget, quota headroom, and per-developer task attribution without external auth servers. | `0` (Status), `2` (Arg Error) |
151
- | `task create` | `agentctl task create [--title <t>] [--prompt <p>] [--template <id>] [--role <name>] [--tier fast\|complex]` | Interactively authors & scopes falsifiable task envelopes with secret scrubbing, preflight gate checks, and DAG dependency wiring. | `0` (Queued), `1` (Secret/Unfalsifiable) |
163
+ | `task create` | `agentctl task create [<prompt>] [--title <t>] [-p <prompt>] [-f <file>] [--template <id>] [--role <name>] [--tier fast\|complex]` | Interactively authors & scopes falsifiable task envelopes with secret scrubbing, preflight gate checks, and DAG dependency wiring. | `0` (Queued), `1` (Secret/Unfalsifiable) |
152
164
  | `task template` | `agentctl task template [<id>] [--list] [--json]` | Lists and synthesizes pre-calibrated task envelopes (Web, Deep Think & Agent Hardening: `web-cwv`, `web-wcag`, `web-seo`, `web-playwright`, `agent-dead-code-audit`, `web-flaky-heal`, `web-i18n`, `web-ai-access`, `agent-qa-mutation`, `agent-ci-falsify`, `agent-service-isolate`, `agent-error-paths`, `agent-security-audit`, `deep-debug`, `deep-feature`, `deep-optimize`, `deep-harden`). | `0` (Listed/Synthesized) |
153
- | `dispatch` | `agentctl dispatch [-p <prompt>] [-f <file>] [-r <role>] [-t <tier>] [--author <name>] [--check-premise] [--auto-pr] [--repoless] [--dry-run]` | Dispatches autonomous task to the active provider with pre-flight idempotency checks, payload limits, and role prompt resolution. | `0` (Dispatched), `1` (Error) |
165
+ | `dispatch` | `agentctl dispatch [<prompt>] [-p <prompt>] [-f <file>] [-r <role>] [-t <tier>] [--author <name>] [--check-premise] [--auto-pr] [--repoless] [--dry-run]` | Dispatches autonomous task to the active provider with pre-flight idempotency checks, payload limits, and role prompt resolution. `--dry-run` stops short of the provider call and reports itself as a rehearsal rather than a dispatch. | `0` (Dispatched), `1` (Error) |
154
166
  | `plan approve` | `agentctl plan approve <sessionId> [--dry-run] [--json]` | Approves pending execution plan for an active Jules session (`:approvePlan`) with automatic 404/503 retry backoff. | `0` (Approved), `1` (Error) |
155
167
  | `session get` | `agentctl session get <sessionId> [--dry-run] [--json]` | Retrieves live session lifecycle state from provider REST API with token rotation. | `0` (Fetched), `1` (Error) |
156
168
  | `pr harvest` | `agentctl pr harvest [--tier r0,r1] [--limit <n>] [--auto] [--allow-no-checks] [--dry-run]` | Discovers open agent PRs, evaluates CI checks & risk tiers, and auto-squashes green low-risk changes autonomously. A PR reporting **no** CI checks is skipped unless `--allow-no-checks` is passed, and an unavailable changed-file list blocks rather than classifying as low risk. | `0` (Triaged/Merged), `1` (Error) |
157
169
  | `doctor` | `agentctl doctor [--json]` | Diagnostic DAG check runner & automated transactional self-repair engine. | `0` (Healthy), `1` (Failures) |
158
170
  | `queue` | `agentctl queue [--dag] [--concurrency <n>] [--dry-run] [--json]` | Consumes and executes task envelopes in `.agent/jules-queue/` with Kahn's DAG dependency resolution. Non-task files (manifests, `README.md`) are skipped, and `--dry-run` previews without moving anything. | `0` (Complete) |
159
171
  | `swarm` | `agentctl swarm [--json]` | Runs parallel multi-agent swarm across worker slots with PID liveness detection. | `0` (Complete) |
160
- | `gate` / `audit`| `agentctl gate --mode working-tree [--json] [--json-report <path>]` | Runs security, secret scanning, and tiered verification gates (with declarative assertion support) against working tree or branch. | `0` (Approved), `3` (Scope), `5` (Diff >75K), `6` (Secret) |
172
+ | `gate` / `audit`| `agentctl gate --mode working-tree [--fix] [--json] [--json-report <path>]` | Runs security, secret scanning, and tiered verification gates (with declarative assertion support) against working tree or branch. Secret findings name the file and line; a failed verify stage reports its command, exit code and output. | `0` (Approved), `3` (Scope), `4` (Verify), `5` (Diff >75K), `6` (Secret), `8` (Flaky) |
161
173
  | `assert` | `agentctl assert [--dir <d>] [--file <f>] [--max-mb <n>] [--gzip] [--targets <g>] [--patterns <p>] [--json] [--json-report <p>]` | Runs declarative zero-dependency verification assertion primitives (`assert:dir-size`, `assert:file-size`, `assert:file-patterns`, `assert:exists`). | `0` (Passed), `1` (Assertion Failed) |
162
174
  | `rollback` | `agentctl rollback [sessionId \| --latest]` | Restores exact commit, uncommitted files, and cleans orphan task worktrees from pre-flight checkpoints. | `0` (Restored), `1` (Error) |
163
175
  | `resume` | `agentctl resume <sessionId> --response "<reply>"` | Streams engineer response back into active Google Jules warm session context window. | `0` (Resumed), `1` (Error) |
@@ -356,6 +368,8 @@ const result = await fast.dispatch({ prompt: "Fix a typo." }, { root: process.cw
356
368
 
357
369
  | Feature | Module / Command | Architectural Description | Status |
358
370
  | :--- | :--- | :--- | :---: |
371
+ | **Diagnostics That Reach the Operator** | `src/security.mjs`, `src/engine.mjs`, `bin/agentctl.mjs` | Secret findings name the file and line, a failed verify stage reports its command, exit code and output, and `queue`/`swarm` name each failed task and exit `1` rather than reporting success for a run that dispatched nothing. | **v0.41.1** *(Shipped)* |
372
+ | **One Scaffolding Path & First-Install Fixes** | `src/scaffold.mjs`, `src/security.mjs` | `agentctl init` and `jules-init` scaffold from one source and write the runtime `.gitignore` entries, so the kit's own bookkeeping no longer reaches its own gate; a lockfile bump no longer fails closed as a secret leak. | **v0.41.1** *(Shipped)* |
359
373
  | **Queue Runner Fidelity** | `src/dag-engine.mjs`, `src/engine.mjs` | Queue selection is by task shape rather than file extension, so manifests and READMEs are skipped instead of dispatched, and `--dry-run` leaves the queue untouched. | **v0.38.2** *(Shipped)* |
360
374
  | **Release Gate Enforcement & Wizard Smoke Test** | `.github/workflows/jules-audit.yml`, `scripts/release.mjs`, `test/wizard-smoke.test.mjs` | Doc-sync gate runs in CI rather than by hand, releases block on a green CI matrix for `HEAD`, per-test deadlines turn a hang into a failure, and the real `init` wizard is driven end to end over a fake TTY. | **v0.38.1** *(Shipped)* |
361
375
  | **Multi-OS CI Matrix & TUI Hardening** | `scripts/run-tests.mjs`, `src/state.mjs`, `src/git.mjs` | Automated 9-job CI matrix across Linux, macOS, and Windows on Node 20/22/24 with raw-mode TUI resilience and native Windows command quoting. | **v0.38.0** *(Shipped)* |
package/bin/agentctl.mjs CHANGED
@@ -76,6 +76,9 @@ Commands:
76
76
  version Output agentctl version
77
77
 
78
78
  Options:
79
+ --prompt, -p Task prompt text — dispatch, task create and task optimize
80
+ also accept it as a positional argument
81
+ --prompt-file, -f Read the prompt from a file (-f is --fix on task optimize)
79
82
  --role, -r Specify specialist agent role (overseer | bolt | sentinel | janitor)
80
83
  --tier Force routing tier when router.enabled (fast | complex) — see .agent/config.yml router:
81
84
  --check-premise Verify task goal/oracle passes locally before burning API budget
@@ -91,6 +94,96 @@ Options:
91
94
  `);
92
95
  }
93
96
 
97
+ /**
98
+ * Resolve the prompt text a command was given, from any of the three forms.
99
+ *
100
+ * The commands that take a prompt each accepted a different subset: `dispatch`
101
+ * took a flag, a file or a positional; `task create` took only `--prompt`; and
102
+ * `task optimize` took only a positional. The form an operator learned on one
103
+ * command then failed on the next — loudly on `task create "do the thing"`,
104
+ * which reported a missing prompt while holding one, and silently on
105
+ * `task optimize --prompt "..."`, which optimised an empty string.
106
+ *
107
+ * @param {Record<string, unknown>} values Parsed flags.
108
+ * @param {string[]} [positionals] Remaining free arguments.
109
+ * @returns {string} The prompt, or "" when none was supplied.
110
+ */
111
+ function resolvePromptInput(values, positionals = []) {
112
+ const file = values["prompt-file"] || values.file;
113
+ if (file) {
114
+ if (!existsSync(file)) {
115
+ console.error(`Error: prompt file not found: ${file}`);
116
+ process.exit(1);
117
+ }
118
+ return readFileSync(file, "utf-8");
119
+ }
120
+ if (values.prompt) return String(values.prompt);
121
+ return positionals.join(" ").trim();
122
+ }
123
+
124
+ /**
125
+ * Renders the outcome of a queue or swarm run and returns the exit code.
126
+ *
127
+ * The per-task failures used to be dropped on the floor. A run where every
128
+ * task was rejected — no API key is the common one — still printed
129
+ * "Processed 3 task(s)." and exited 0, so neither an operator nor a CI job
130
+ * could tell a dispatched queue from a dead one. The failures were already in
131
+ * the result object the whole time; only `--json` ever showed them.
132
+ *
133
+ * @param {{ processed?: number, results?: Array<{file?: string, ok?: boolean, status?: string, error?: string}> }} outcome
134
+ * @returns {number} 0 when every task succeeded, 1 when any failed.
135
+ */
136
+ function reportRunOutcome(outcome) {
137
+ const items = Array.isArray(outcome?.results) ? outcome.results : [];
138
+ const failed = items.filter((r) => r && r.ok === false);
139
+ const succeeded = items.length - failed.length;
140
+
141
+ console.log(`\nProcessed ${outcome?.processed ?? items.length} task(s): ${succeeded} ok, ${failed.length} failed.`);
142
+
143
+ if (failed.length > 0) {
144
+ console.error(`\n❌ ${failed.length} task(s) did not dispatch:`);
145
+ for (const f of failed) {
146
+ const reason = f.error || f.status || "Unknown error";
147
+ console.error(` - ${f.file || f.taskId || "task"}: ${reason}`);
148
+ }
149
+ // A failed task is left in the queue rather than moved to completed/, so
150
+ // fixing the cause and re-running is the whole recovery procedure.
151
+ console.error(`\n These tasks are still queued. Fix the cause above and re-run.`);
152
+ return 1;
153
+ }
154
+ if (items.some((r) => r && r.dryRun)) {
155
+ console.log(` Dry run — no provider call was made and nothing was dispatched.`);
156
+ }
157
+ return 0;
158
+ }
159
+
160
+ /**
161
+ * Render the stage, exit code and captured output of a failed verify phase.
162
+ *
163
+ * VERIFY is the one gate phase whose failure the operator has to fix in their
164
+ * own code, and it was the only one that printed nothing beyond "❌ FAIL" —
165
+ * the command's output was captured, hashed into the evidence manifest, and
166
+ * then discarded before anyone could read it.
167
+ *
168
+ * @param {{ stageId?: string, command?: string|null, exitCode?: number|null, stdout?: string, stderr?: string, diagnostics?: string[] }} failure
169
+ */
170
+ const VERIFY_OUTPUT_TAIL_LINES = 20;
171
+
172
+ function printVerifyFailure(failure) {
173
+ const exit = failure.exitCode === null || failure.exitCode === undefined ? "n/a" : failure.exitCode;
174
+ console.log(` - Stage: ${failure.stageId || "verify"} (exit ${exit})`);
175
+ if (failure.command) console.log(` - Command: ${failure.command}`);
176
+ for (const d of failure.diagnostics || []) console.log(` - ${d}`);
177
+
178
+ // stderr is where a failing suite says what it expected; stdout is the
179
+ // fallback for the runners that report everything there.
180
+ const output = (failure.stderr || "").trim() || (failure.stdout || "").trim();
181
+ if (!output) return;
182
+ const lines = output.split("\n");
183
+ const tail = lines.slice(-VERIFY_OUTPUT_TAIL_LINES);
184
+ console.log(` - Output${tail.length < lines.length ? ` (last ${VERIFY_OUTPUT_TAIL_LINES} of ${lines.length} lines)` : ""}:`);
185
+ for (const line of tail) console.log(` ${line}`);
186
+ }
94
187
 
95
188
  async function main() {
96
189
  if (command === "--help" || command === "-h") {
@@ -140,7 +233,7 @@ async function main() {
140
233
  switch (command) {
141
234
  case "dispatch":
142
235
  case "create": {
143
- const { values } = parseArgs({
236
+ const { values, positionals } = parseArgs({
144
237
  args: args.slice(1),
145
238
  options: {
146
239
  title: { type: "string", short: "t" },
@@ -162,20 +255,28 @@ async function main() {
162
255
  allowPositionals: true,
163
256
  });
164
257
 
165
- let promptContent = values.prompt || "";
166
- if (values["prompt-file"] && existsSync(values["prompt-file"])) {
167
- promptContent = readFileSync(values["prompt-file"], "utf-8");
168
- }
169
-
170
- if (!promptContent && args[1] && !args[1].startsWith("-")) {
171
- promptContent = args.slice(1).join(" ");
172
- }
173
-
258
+ const promptContent = resolvePromptInput(values, positionals);
174
259
  if (!promptContent) {
175
- console.error("Error: --prompt or --prompt-file is required.");
260
+ console.error("Error: a prompt is required — pass it as --prompt, --prompt-file, or a positional argument.");
176
261
  process.exit(1);
177
262
  }
178
263
 
264
+ // `task create --role` has always rejected a role it cannot resolve;
265
+ // `dispatch --role` used to drop it without a word and hand the work to a
266
+ // generic agent, so the two commands disagreed about what the same flag
267
+ // means. An explicitly typed role is a statement of intent — failing here
268
+ // is the only way the operator learns the prompt file is missing.
269
+ if (values.role) {
270
+ const { resolveRolePrompt } = await import("../src/role-resolver.mjs");
271
+ if (!resolveRolePrompt(root, values.role, { config })) {
272
+ console.error(
273
+ `Error: Unknown agent role '${values.role}'. Expected matching prompt file in .agent/prompts/ (e.g. Overseer, Bolt, Sentinel, Janitor).`
274
+ );
275
+ console.error(` Run 'agentctl init' to scaffold the shipped role prompts.`);
276
+ process.exit(1);
277
+ }
278
+ }
279
+
179
280
  const task = {
180
281
  title: values.title || "CLI Dispatch Task",
181
282
  prompt: promptContent,
@@ -206,6 +307,20 @@ async function main() {
206
307
  if (session.status === "ALREADY_SATISFIED" || session.skipped) {
207
308
  console.log(`\n⚡ Task Already Satisfied (skipped dispatch):`);
208
309
  console.log(` Reason: ${session.reason || "Verification oracle already passing on base branch."}`);
310
+ } else if (values["dry-run"]) {
311
+ // The dry run reached the provider adapter and stopped short of the
312
+ // call. Printing the same "Dispatched Successfully!" banner as a
313
+ // real dispatch made the two indistinguishable in a terminal, so a
314
+ // rehearsal read as work in flight and the operator waited for a
315
+ // session that was never going to exist.
316
+ console.log(`\n🧪 Dry Run — nothing was dispatched.`);
317
+ console.log(` Title : ${task.title}`);
318
+ console.log(` Provider : ${session.provider || config.provider || "jules"}`);
319
+ if (task.role) console.log(` Role : ${task.role}`);
320
+ if (session._routeTier) {
321
+ console.log(` Router Tier : ${session._routeTier} (${session._routeReason || "n/a"})`);
322
+ }
323
+ console.log(`\n Re-run without --dry-run to dispatch for real.`);
209
324
  } else {
210
325
  console.log(`\n✅ Task Dispatched Successfully!`);
211
326
  console.log(` Session ID : ${session.id}`);
@@ -274,13 +389,32 @@ async function main() {
274
389
  p.violations.forEach((v) => console.log(` - Violation: ${v.file} (Rule: ${v.rule})`));
275
390
  }
276
391
  if (p.findings) {
277
- p.findings.forEach((f) => console.log(` - Finding: ${f.id} at line ${f.line}`));
392
+ p.findings.forEach((f) => console.log(` - [${f.severity}] ${f.type}: ${f.description}`));
393
+ }
394
+ if (!p.ok && p.failure) {
395
+ printVerifyFailure(p.failure);
278
396
  }
279
397
  }
280
398
  console.log(`-----------------------------------------------------`);
281
399
  console.log(`Overall Result: ${res.ok ? "APPROVED (Exit 0)" : `REJECTED (Exit ${res.code})`}\n`);
282
400
  if (!res.ok) {
283
- if (res.code === 3) {
401
+ const failedPhase = res.phases.find((p) => !p.ok)?.phase;
402
+ // Exit 3 is also what a strictTestLock tamper verdict returns, so the
403
+ // code alone cannot pick the hint — a scope remediation for a rewritten
404
+ // test file sends the operator to the wrong flag entirely.
405
+ if (res.repairs) {
406
+ console.log(`💡 Remediation Hint (Exit 4 OODA Repair Exhausted):`);
407
+ console.log(` • Automated self-repair could not pass tests cleanly.`);
408
+ console.log(` • Review error fingerprints via: agentctl doctor\n`);
409
+ } else if (res.flakyVerdict?.verdict === "QUARANTINED") {
410
+ console.log(`💡 Remediation Hint (Exit 8 Flaky Test Quarantined):`);
411
+ console.log(` • This command has alternated between pass and fail across recent runs.`);
412
+ console.log(` • Fix the test's non-determinism — re-running will not clear the verdict.\n`);
413
+ } else if (failedPhase === "verify" || failedPhase === "evidence") {
414
+ console.log(`💡 Remediation Hint (Exit ${res.code} Verification Failed):`);
415
+ console.log(` • The stage above exited non-zero. Reproduce it locally, then re-run the gate.`);
416
+ console.log(` • To let agentctl attempt the repair loop itself, pass: agentctl gate --fix\n`);
417
+ } else if (res.code === 3) {
284
418
  console.log(`💡 Remediation Hint (Exit 3 Scope Violation):`);
285
419
  console.log(` • To allow protected files in this run, pass: agentctl gate --allow-protected`);
286
420
  console.log(` • Or remove protected/denied paths from the diff before dispatching.\n`);
@@ -292,10 +426,6 @@ async function main() {
292
426
  console.log(`💡 Remediation Hint (Exit 6 Secret Leak Prevented):`);
293
427
  console.log(` • High-entropy credential or secret detected in patch.`);
294
428
  console.log(` • Scrub credential from source and rotate any exposed keys immediately.\n`);
295
- } else if (res.code === 4) {
296
- console.log(`💡 Remediation Hint (Exit 4 OODA Repair Exhausted):`);
297
- console.log(` • Automated self-repair could not pass tests cleanly.`);
298
- console.log(` • Review error fingerprints via: agentctl doctor\n`);
299
429
  }
300
430
  }
301
431
  }
@@ -417,9 +547,10 @@ async function main() {
417
547
  });
418
548
  if (values.json) {
419
549
  console.log(JSON.stringify(results, null, 2));
420
- } else {
421
- console.log(`\nProcessed ${results.processed || results.results?.length || 0} task(s).`);
550
+ const anyFailed = (results.results || []).some((r) => r && r.ok === false);
551
+ process.exit(anyFailed ? 1 : 0);
422
552
  }
553
+ process.exit(reportRunOutcome(results));
423
554
  }
424
555
  process.exit(0);
425
556
  break;
@@ -439,8 +570,7 @@ async function main() {
439
570
  prompt: readFileSync(join(queueDir, f), "utf-8"),
440
571
  }));
441
572
  const results = await run(tasks, { root, config, concurrency: config.limits.concurrency || 3 });
442
- console.log(`Swarm completed ${results.length} tasks.`);
443
- process.exit(0);
573
+ process.exit(reportRunOutcome(results));
444
574
  break;
445
575
  }
446
576
 
@@ -679,6 +809,7 @@ async function main() {
679
809
  tier: { type: "string", short: "t" },
680
810
  json: { type: "boolean", short: "j" },
681
811
  "dry-run": { type: "boolean", short: "d" },
812
+ force: { type: "boolean", short: "f" },
682
813
  },
683
814
  allowPositionals: true,
684
815
  });
@@ -694,13 +825,38 @@ async function main() {
694
825
  allowDefaults: true,
695
826
  });
696
827
 
828
+ // The wizard writes the manifest; the assets the CLI's documented
829
+ // features actually need — AGENTS.md, the role prompts, the guardrails,
830
+ // the gitignore entries — used to be scaffolded only by the separate
831
+ // `jules-init` binary that the README's quickstart never mentions.
832
+ const { scaffoldRepoAssets } = await import("../src/scaffold.mjs");
833
+ const scaffold = scaffoldRepoAssets(root, { force: values.force });
834
+
697
835
  if (values.json) {
698
- console.log(JSON.stringify(res, null, 2));
836
+ console.log(JSON.stringify({ ...res, scaffold }, null, 2));
699
837
  } else {
700
838
  console.log(`✅ Onboarding complete! Manifest generated at ${res.configPath}`);
701
839
  console.log(` Tier: ${res.plan.tier.toUpperCase()} (${res.plan.limits.concurrency} worker(s), ${res.plan.limits.daily_tasks} daily tasks)`);
702
840
  console.log(` Verification Test Command : "${res.plan.verify.test}"`);
703
841
  console.log(` Active Presets : ${res.plan.presets.join(", ")}`);
842
+ for (const item of scaffold.created) {
843
+ console.log(` Scaffolded : ${item}`);
844
+ }
845
+ if (scaffold.gitignore.length > 0) {
846
+ console.log(` Ignored runtime state : ${scaffold.gitignore.length} entries added to .gitignore`);
847
+ }
848
+
849
+ // `.agent/config.yml` and `.agent/jules.yml` are both on the gate's deny
850
+ // list, by design — the agent must not edit its own rules. Leaving them
851
+ // uncommitted meant the very first `agentctl gate` rejected the working
852
+ // tree for files init had just written, which reads as the tool
853
+ // catching the user cheating on step three.
854
+ console.log(`\n Commit the manifest so the gate does not read it as an agent edit:`);
855
+ console.log(` git add .agent AGENTS.md .gitignore && git commit -m "chore: add agent config"`);
856
+
857
+ const { resolveNextStep, renderNextStep } = await import("../src/ops/next-step.mjs");
858
+ const next = resolveNextStep(root);
859
+ console.log(renderNextStep({ version: VERSION, root, next, budgetLine: "" }));
704
860
  }
705
861
  process.exit(0);
706
862
  break;
@@ -709,11 +865,12 @@ async function main() {
709
865
  case "task": {
710
866
  const subCommand = args[1] || "create";
711
867
  if (subCommand === "create") {
712
- const { values } = parseArgs({
868
+ const { values, positionals } = parseArgs({
713
869
  args: args.slice(2),
714
870
  options: {
715
871
  title: { type: "string", short: "t" },
716
872
  prompt: { type: "string", short: "p" },
873
+ "prompt-file": { type: "string", short: "f" },
717
874
  role: { type: "string", short: "r" },
718
875
  tier: { type: "string" },
719
876
  template: { type: "string" },
@@ -732,7 +889,7 @@ async function main() {
732
889
  const { runTaskCreateWizard } = await import("../src/wizard-task.mjs");
733
890
  const res = await runTaskCreateWizard(root, {
734
891
  title: values.title,
735
- prompt: values.prompt,
892
+ prompt: resolvePromptInput(values, positionals),
736
893
  role: values.role,
737
894
  tier: values.tier,
738
895
  template: values.template,
@@ -806,6 +963,12 @@ async function main() {
806
963
  args: args.slice(2),
807
964
  options: {
808
965
  fix: { type: "boolean", short: "f" },
966
+ prompt: { type: "string", short: "p" },
967
+ // No short form here: `-f` is already --fix on this subcommand, and
968
+ // silently meaning two different things would be worse than one
969
+ // command having a flag short of full parity. `--file` predates
970
+ // `--prompt-file` and stays as an alias.
971
+ "prompt-file": { type: "string" },
809
972
  file: { type: "string" },
810
973
  dir: { type: "string", short: "d" },
811
974
  web: { type: "boolean", short: "w" },
@@ -818,16 +981,7 @@ async function main() {
818
981
 
819
982
  const { scorePromptFalsifiability, optimizeTaskPrompt } = await import("../src/task-optimizer.mjs");
820
983
  const targetDir = values.dir ? resolve(values.dir) : root;
821
- let promptText = positionals.join(" ");
822
-
823
- if (values.file) {
824
- if (existsSync(values.file)) {
825
- promptText = readFileSync(values.file, "utf-8");
826
- } else {
827
- console.error(`Error: File '${values.file}' does not exist.`);
828
- process.exit(1);
829
- }
830
- }
984
+ const promptText = resolvePromptInput(values, positionals);
831
985
 
832
986
  if (values.fix) {
833
987
  const opt = optimizeTaskPrompt(promptText, { rootDir: targetDir, verifyCmd: values["verify-cmd"], web: values.web });
@@ -910,6 +1064,12 @@ async function main() {
910
1064
  intent: { type: "string", short: "i" },
911
1065
  handover: { type: "boolean", default: true },
912
1066
  json: { type: "boolean", short: "j" },
1067
+ // Documented in the README as `rollback [sessionId | --latest]` but
1068
+ // never registered, so the documented spelling died in parseArgs
1069
+ // before restoreCheckpoint — which has always accepted it — was
1070
+ // reached. Restoring the newest checkpoint is also the no-argument
1071
+ // default, so the flag is explicit-intent sugar rather than a mode.
1072
+ latest: { type: "boolean" },
913
1073
  },
914
1074
  allowPositionals: true,
915
1075
  });
package/bin/init.js CHANGED
@@ -117,53 +117,17 @@ if (detected.testCmd || detected.buildCmd) {
117
117
  console.log(` - Build Command: ${detected.buildCmd || "(none)"}`);
118
118
  }
119
119
 
120
- // 2. Scaffold AGENTS.md / .agent/jules-protocol.md
121
- const agentsFile = path.join(targetDir, "AGENTS.md");
122
- const julesRulesSource = path.join(kitRoot, "JULES_RULES_TEMPLATE.md");
123
-
124
- if (!fs.existsSync(agentsFile) || isForce) {
125
- if (fs.existsSync(julesRulesSource)) {
126
- fs.copyFileSync(julesRulesSource, agentsFile);
127
- console.log("✅ Created: AGENTS.md");
128
- }
129
- } else {
130
- const existingContent = fs.readFileSync(agentsFile, "utf-8");
131
- if (!existingContent.includes("<MCP_DIRECTIVE>")) {
132
- if (fs.existsSync(julesRulesSource)) {
133
- const templateContent = fs.readFileSync(julesRulesSource, "utf-8");
134
- fs.appendFileSync(agentsFile, `\n\n---\n\n${templateContent}`, "utf-8");
135
- console.log("✅ Appended Google Jules directives to existing AGENTS.md");
136
- }
137
- } else {
138
- console.log("ℹ️ AGENTS.md already contains Jules directives (skipped overwrite).");
139
- }
120
+ // 2-3. Scaffold AGENTS.md, .agent/ structure, role prompts, rules and workflows.
121
+ // Shared with `agentctl init` so the two entry points cannot scaffold different
122
+ // repositories — which is exactly what they used to do, with the README's
123
+ // quickstart pointing at the one that scaffolded less.
124
+ const { scaffoldRepoAssets } = await import("../src/scaffold.mjs");
125
+ const scaffolded = scaffoldRepoAssets(targetDir, { force: isForce });
126
+ for (const item of scaffolded.created) {
127
+ console.log(`✅ Created: ${item}`);
140
128
  }
141
129
 
142
- // 3. Scaffold .agent/ structure
143
130
  const agentDir = path.join(targetDir, ".agent");
144
- const rulesDir = path.join(agentDir, "rules");
145
- const queueDir = path.join(agentDir, "jules-queue");
146
- const completedQueueDir = path.join(queueDir, "completed");
147
- const workflowsDir = path.join(agentDir, "workflows");
148
- const promptsDir = path.join(agentDir, "prompts");
149
-
150
- [agentDir, rulesDir, queueDir, completedQueueDir, workflowsDir, promptsDir].forEach((d) => {
151
- if (!fs.existsSync(d)) fs.mkdirSync(d, { recursive: true });
152
- });
153
-
154
- // Scaffold .agent/prompts files
155
- const sourcePromptsDir = path.join(kitRoot, ".agent/prompts");
156
- if (fs.existsSync(sourcePromptsDir)) {
157
- const promptFiles = fs.readdirSync(sourcePromptsDir);
158
- promptFiles.forEach((file) => {
159
- const srcPrompt = path.join(sourcePromptsDir, file);
160
- const destPrompt = path.join(promptsDir, file);
161
- if (!fs.existsSync(destPrompt) || isForce) {
162
- fs.copyFileSync(srcPrompt, destPrompt);
163
- }
164
- });
165
- console.log("✅ Created: .agent/prompts presets (Overseer, Bolt, Sentinel, Janitor, Task_Template)");
166
- }
167
131
 
168
132
  // Scaffold .agent/jules.yml
169
133
  const yamlConfigPath = path.join(agentDir, "jules.yml");
@@ -178,22 +142,6 @@ forbidden_paths: [".github/**", "**/.env*", "**/*.pem", "**/lock-manager*"]
178
142
  console.log("✅ Created: .agent/jules.yml");
179
143
  }
180
144
 
181
- // Scaffold .agent/rules/dynamic-guardrails.json
182
- const dgcSource = path.join(kitRoot, ".agent/rules/dynamic-guardrails.json");
183
- const dgcTarget = path.join(rulesDir, "dynamic-guardrails.json");
184
- if ((!fs.existsSync(dgcTarget) || isForce) && fs.existsSync(dgcSource)) {
185
- fs.copyFileSync(dgcSource, dgcTarget);
186
- console.log("✅ Created: .agent/rules/dynamic-guardrails.json");
187
- }
188
-
189
- // Scaffold .agent/workflows/jules-review.md
190
- const reviewSource = path.join(kitRoot, ".agent/workflows/jules-review.md");
191
- const reviewTarget = path.join(workflowsDir, "jules-review.md");
192
- if ((!fs.existsSync(reviewTarget) || isForce) && fs.existsSync(reviewSource)) {
193
- fs.copyFileSync(reviewSource, reviewTarget);
194
- console.log("✅ Created: .agent/workflows/jules-review.md");
195
- }
196
-
197
145
  // Scaffold .github/workflows/jules-audit.yml
198
146
  const githubWorkflowsDir = path.join(targetDir, ".github/workflows");
199
147
  const auditWfSource = path.join(kitRoot, ".github/workflows/jules-audit.yml");
@@ -275,32 +223,10 @@ if (fs.existsSync(targetPkgPath) && targetDir !== kitRoot) {
275
223
  }
276
224
  }
277
225
 
278
- // 5b. Ensure target repository has .gitignore entries for sensitive and runtime state
279
- const targetGitignorePath = path.join(targetDir, ".gitignore");
280
- const requiredGitignoreEntries = [
281
- ".env",
282
- ".agent/history/",
283
- ".agent/state/",
284
- ".agent/jules-queue/.state/",
285
- ".agent/jules-queue/failed/",
286
- ".agent/jules-queue/.processing/",
287
- ".agent/jules-queue/*.md",
288
- "!.agent/jules-queue/README.md"
289
- ];
290
-
291
- let gitignoreContent = fs.existsSync(targetGitignorePath)
292
- ? fs.readFileSync(targetGitignorePath, "utf-8")
293
- : "";
294
-
295
- const missingEntries = requiredGitignoreEntries.filter(
296
- (entry) => !gitignoreContent.includes(entry)
297
- );
298
-
299
- if (missingEntries.length > 0) {
300
- const prefix = gitignoreContent && !gitignoreContent.endsWith("\n") ? "\n" : "";
301
- const addedBlock = `${prefix}# Jules Orchestrator Runtime State & Credentials\n${missingEntries.join("\n")}\n`;
302
- fs.appendFileSync(targetGitignorePath, addedBlock, "utf-8");
303
- console.log("✅ Added required security entries to .gitignore");
226
+ // 5b. The .gitignore entries are written by scaffoldRepoAssets above, so the
227
+ // two entry points cannot disagree about which runtime paths stay untracked.
228
+ if (scaffolded.gitignore.length > 0) {
229
+ console.log(`✅ Added ${scaffolded.gitignore.length} runtime state entries to .gitignore`);
304
230
  }
305
231
 
306
232
  console.log("\n🎉 Google Jules Orchestration Kit successfully initialized!");
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "jules-orchestrator-kit",
3
- "version": "0.41.0",
3
+ "version": "0.41.1",
4
4
  "description": "Zero-dependency safety gatekeeper, test oracle generator, and multi-agent coordination protocol for Google Jules (jules) autonomous agents.",
5
5
  "repository": {
6
6
  "type": "git",
package/src/engine.mjs CHANGED
@@ -357,7 +357,22 @@ export async function gate(opts = {}) {
357
357
  const runs = readVerifyRuns(root, stage.cmd);
358
358
  flakyVerdictResult = flakyVerdict(runs);
359
359
  if (flakyVerdictResult.verdict === "QUARANTINED") {
360
- phases.push({ phase: "verify", ok: false, testResult, buildResult, flakyVerdict: flakyVerdictResult, executionRecords });
360
+ phases.push({
361
+ phase: "verify",
362
+ ok: false,
363
+ testResult,
364
+ buildResult,
365
+ failure: {
366
+ stageId: stage.id || "unit",
367
+ command: stage.cmd || null,
368
+ exitCode: res.status ?? null,
369
+ stdout: stdoutRedacted,
370
+ stderr: stderrRedacted,
371
+ diagnostics: [`Quarantined as flaky: ${flakyVerdictResult.reason || "alternating pass/fail across recent runs"}`],
372
+ },
373
+ flakyVerdict: flakyVerdictResult,
374
+ executionRecords,
375
+ });
361
376
  appendTelemetry(root, "gate_phase", { phase: "verify", ok: false, quarantined: true });
362
377
  appendTelemetry(root, "gate_finished", { ok: false, code: 8 });
363
378
  return { ok: false, code: 8, phases, flakyVerdict: flakyVerdictResult };
@@ -413,6 +428,31 @@ export async function gate(opts = {}) {
413
428
 
414
429
  const verifyOk = !failingCmd && !testTampered;
415
430
 
431
+ // What actually broke. Without this the verify phase reported `ok: false` and
432
+ // nothing else — not the stage, not the exit code, not a line of output — so
433
+ // the one gate failure the operator is expected to fix themselves was the
434
+ // only one that told them nothing about how. Both streams are already
435
+ // redacted at the point they were captured.
436
+ const verifyFailure = failingCmd
437
+ ? {
438
+ stageId: failingCmd.stageId || failingCmd.phase || "verify",
439
+ command: failingCmd.command || null,
440
+ exitCode: failingCmd.status ?? null,
441
+ stdout: failingCmd.stdout || "",
442
+ stderr: failingCmd.stderr || "",
443
+ diagnostics: failingCmd.diagnostics || [],
444
+ }
445
+ : testTampered
446
+ ? {
447
+ stageId: "test-integrity",
448
+ command: null,
449
+ exitCode: null,
450
+ stdout: "",
451
+ stderr: `Test files changed during the run (${preTestHash.slice(0, 12)} → ${postTestHash.slice(0, 12)}). evidence.strictTestLock treats a passing suite that the diff also rewrote as unproven.`,
452
+ diagnostics: [],
453
+ }
454
+ : null;
455
+
416
456
  // Generate & persist Evidence Manifest
417
457
  const evidenceManifest = generateEvidenceManifest(root, {
418
458
  taskId: opts.taskId,
@@ -449,6 +489,7 @@ export async function gate(opts = {}) {
449
489
  testResult,
450
490
  buildResult,
451
491
  serverResult,
492
+ failure: verifyFailure,
452
493
  executionRecords,
453
494
  testIntegrity: {
454
495
  preTestHash,
@@ -834,6 +875,15 @@ export async function dispatch(task = {}, opts = {}) {
834
875
  const roleObj = resolveRolePrompt(root, task.role);
835
876
  if (roleObj) {
836
877
  cleanPrompt = `${roleObj.content}\n\n${cleanPrompt}`.trim();
878
+ } else {
879
+ // A role reaching here comes from a task envelope or an internal
880
+ // synthesis rather than a typed flag, so the dispatch still proceeds with
881
+ // a generic agent — failing an automated heal swarm over a missing
882
+ // prompt file helps nobody. It must not proceed *silently* though: the
883
+ // caller asked for a specialist and is not getting one.
884
+ console.warn(
885
+ `⚠️ Role '${task.role}' has no prompt in .agent/prompts/ — dispatching without specialist context. Run 'agentctl init' to scaffold the shipped roles.`
886
+ );
837
887
  }
838
888
  }
839
889
 
package/src/git.mjs CHANGED
@@ -27,6 +27,9 @@ export function runCmd(command, opts = {}) {
27
27
  let args = [];
28
28
  let useShell = false;
29
29
  let shellCmd = "";
30
+ // True when `args` came from splitting a whitespace-separated string, which
31
+ // means no element can itself contain whitespace. See the Windows note below.
32
+ let tokenized = false;
30
33
 
31
34
  if (Array.isArray(command)) {
32
35
  binary = command[0];
@@ -40,9 +43,25 @@ export function runCmd(command, opts = {}) {
40
43
  const tokens = trimmed.split(/\s+/).filter(Boolean);
41
44
  binary = tokens[0] || "";
42
45
  args = tokens.slice(1);
46
+ tokenized = true;
43
47
  }
44
48
  }
45
49
 
50
+ // Every package-manager entry point on Windows is a `.cmd` shim, and
51
+ // execFileSync cannot spawn one directly. `npm test` — the kit's own default
52
+ // verify command — therefore failed with `spawnSync npm ENOENT` on every
53
+ // Windows install, and the gate reported it as a plain non-zero verification
54
+ // rather than as an environment problem.
55
+ //
56
+ // Node's `shell: true` rebuilds the command line as `[file, ...args].join(" ")`
57
+ // and passes it to cmd.exe verbatim. That reconstruction is normally lossy —
58
+ // it does not quote an argument containing a space — but it is exact here,
59
+ // because these tokens were produced by splitting on whitespace in the first
60
+ // place. It is also only reached when the command contains no shell
61
+ // metacharacter, since those take the execSync branch above. An array command
62
+ // is excluded: its elements may legitimately contain spaces.
63
+ const winShim = tokenized && process.platform === "win32";
64
+
46
65
  if (!binary && !useShell) {
47
66
  if (opts.ignoreError) return { status: 0, stdout: "", stderr: "" };
48
67
  throw new GateError("Empty command provided");
@@ -61,7 +80,7 @@ export function runCmd(command, opts = {}) {
61
80
  : execFileSync(binary, args, {
62
81
  cwd,
63
82
  encoding: "utf-8",
64
- shell: false,
83
+ shell: winShim,
65
84
  stdio: ["ignore", "pipe", "pipe"],
66
85
  env: opts.env || process.env,
67
86
  timeout,
@@ -0,0 +1,146 @@
1
+ import { existsSync, mkdirSync, readdirSync, copyFileSync, readFileSync, appendFileSync, writeFileSync } from "node:fs";
2
+ import { join, dirname } from "node:path";
3
+ import { fileURLToPath } from "node:url";
4
+
5
+ const KIT_ROOT = join(dirname(fileURLToPath(import.meta.url)), "..");
6
+
7
+ /**
8
+ * Paths the kit writes at runtime and must never hand to its own gate.
9
+ *
10
+ * Without these, `agentctl init` left every ledger, evidence manifest and
11
+ * telemetry line untracked in the working tree. The gate audits that tree, so
12
+ * the kit's own bookkeeping showed up as a diff the agent was accused of
13
+ * making: first as scope violations against `.agent/config.yml`, then — once
14
+ * enough evidence files accumulated — as a CRITICAL secret verdict. A new user
15
+ * met both before dispatching a single task.
16
+ */
17
+ export const RUNTIME_GITIGNORE_ENTRIES = [
18
+ ".env",
19
+ ".agent/history/",
20
+ ".agent/state/",
21
+ ".agent/evidence/",
22
+ ".agent/handovers/",
23
+ ".agent/jules-queue/.state/",
24
+ ".agent/jules-queue/failed/",
25
+ ".agent/jules-queue/.processing/",
26
+ ".agent/jules-queue/completed/",
27
+ // A queued envelope is work-in-flight, not source. `.agent/jules-queue/**` is
28
+ // on the gate's deny list, so tracking the envelopes means the first gate
29
+ // after `task create` rejects the tree for the file `task create` just wrote.
30
+ // The negation has to follow the pattern it re-includes.
31
+ ".agent/jules-queue/*.md",
32
+ "!.agent/jules-queue/README.md",
33
+ ];
34
+
35
+ /**
36
+ * Ensure `.gitignore` lists every runtime path in {@link RUNTIME_GITIGNORE_ENTRIES}.
37
+ *
38
+ * @param {string} root
39
+ * @returns {string[]} Entries newly appended (empty when already covered).
40
+ */
41
+ export function ensureGitignore(root) {
42
+ const gitignorePath = join(root, ".gitignore");
43
+ const current = existsSync(gitignorePath) ? readFileSync(gitignorePath, "utf-8") : "";
44
+ const lines = new Set(current.split("\n").map((l) => l.trim()));
45
+ const missing = RUNTIME_GITIGNORE_ENTRIES.filter((e) => !lines.has(e));
46
+ if (missing.length === 0) return [];
47
+
48
+ const prefix = current && !current.endsWith("\n") ? "\n" : "";
49
+ appendFileSync(
50
+ gitignorePath,
51
+ `${prefix}\n# Jules Orchestrator runtime state & credentials\n${missing.join("\n")}\n`,
52
+ "utf-8"
53
+ );
54
+ return missing;
55
+ }
56
+
57
+ /**
58
+ * Copy every file from a directory shipped in the package into the target repo.
59
+ *
60
+ * @param {string} srcDir
61
+ * @param {string} destDir
62
+ * @param {boolean} force
63
+ * @returns {number} Files written.
64
+ */
65
+ function copyDir(srcDir, destDir, force) {
66
+ if (!existsSync(srcDir) || srcDir === destDir) return 0;
67
+ mkdirSync(destDir, { recursive: true });
68
+ let written = 0;
69
+ for (const file of readdirSync(srcDir)) {
70
+ const src = join(srcDir, file);
71
+ const dest = join(destDir, file);
72
+ if (!existsSync(dest) || force) {
73
+ copyFileSync(src, dest);
74
+ written++;
75
+ }
76
+ }
77
+ return written;
78
+ }
79
+
80
+ /**
81
+ * Scaffold the repository assets the CLI's documented features depend on.
82
+ *
83
+ * This is the single source of truth for both entry points. `agentctl init`
84
+ * and `jules-init` used to scaffold different things: only the latter wrote
85
+ * AGENTS.md, the role prompts and the guardrails, while the README's quickstart
86
+ * pointed at the former. Anyone following the quickstart got a Jules that never
87
+ * saw the protocol and a `--role` flag with nothing to resolve against.
88
+ *
89
+ * Existing files are preserved unless `force` is set — re-running init is a
90
+ * routine way to pick up new presets and must not overwrite local edits.
91
+ *
92
+ * @param {string} [root=process.cwd()]
93
+ * @param {{ force?: boolean }} [options]
94
+ * @returns {{ created: string[], gitignore: string[] }}
95
+ */
96
+ export function scaffoldRepoAssets(root = process.cwd(), options = {}) {
97
+ const force = Boolean(options.force);
98
+ const created = [];
99
+
100
+ const agentDir = join(root, ".agent");
101
+ const queueDir = join(agentDir, "jules-queue");
102
+ for (const d of [agentDir, queueDir, join(queueDir, "completed"), join(agentDir, "rules"), join(agentDir, "prompts"), join(agentDir, "workflows")]) {
103
+ if (!existsSync(d)) mkdirSync(d, { recursive: true });
104
+ }
105
+
106
+ // AGENTS.md is how the agent learns the protocol at all, so an existing file
107
+ // is appended to rather than replaced: repositories that already brief their
108
+ // agents must not lose that briefing to a scaffolding step.
109
+ const agentsFile = join(root, "AGENTS.md");
110
+ const template = join(KIT_ROOT, "JULES_RULES_TEMPLATE.md");
111
+ if (existsSync(template) && template !== agentsFile) {
112
+ if (!existsSync(agentsFile) || force) {
113
+ copyFileSync(template, agentsFile);
114
+ created.push("AGENTS.md");
115
+ } else if (!readFileSync(agentsFile, "utf-8").includes("<MCP_DIRECTIVE>")) {
116
+ appendFileSync(agentsFile, `\n\n---\n\n${readFileSync(template, "utf-8")}`, "utf-8");
117
+ created.push("AGENTS.md (appended)");
118
+ }
119
+ }
120
+
121
+ if (copyDir(join(KIT_ROOT, ".agent/prompts"), join(agentDir, "prompts"), force) > 0) {
122
+ created.push(".agent/prompts/ (Overseer, Bolt, Sentinel, Janitor)");
123
+ }
124
+ if (copyDir(join(KIT_ROOT, ".agent/rules"), join(agentDir, "rules"), force) > 0) {
125
+ created.push(".agent/rules/");
126
+ }
127
+ if (copyDir(join(KIT_ROOT, ".agent/workflows"), join(agentDir, "workflows"), force) > 0) {
128
+ created.push(".agent/workflows/");
129
+ }
130
+
131
+ // isTaskFile() skips README.md, so the queue can carry its own explanation
132
+ // without the runner mistaking it for a task envelope.
133
+ const queueReadme = join(queueDir, "README.md");
134
+ if (!existsSync(queueReadme)) {
135
+ writeFileSync(
136
+ queueReadme,
137
+ "# Task Queue\n\nEach `TASK-*.md` here is one queued task envelope.\n\n" +
138
+ "- `agentctl task create` writes them\n- `agentctl queue` dispatches them\n" +
139
+ "- Dispatched envelopes move to `completed/`; failures stay put for a re-run\n",
140
+ "utf-8"
141
+ );
142
+ created.push(".agent/jules-queue/README.md");
143
+ }
144
+
145
+ return { created, gitignore: ensureGitignore(root) };
146
+ }
package/src/security.mjs CHANGED
@@ -454,8 +454,26 @@ function secretScanVariants(addedLines) {
454
454
  // get encoded into a single CI variable.
455
455
  const BASE64_CANDIDATE = /[A-Za-z0-9+/\-_]{20,}={0,2}/g;
456
456
 
457
+ // Budgets the decoder spends before it gives up and reports `capped`.
458
+ //
459
+ // The count that matters is payloads *retained* — blobs that decoded to text
460
+ // and so could be carrying a credential. Counting every token that merely
461
+ // matches the base64 alphabet instead made a digest indistinguishable from a
462
+ // payload: a sha256 hex string is 64 characters of that alphabet, decodes to
463
+ // binary, gets discarded, and used to consume a slot anyway. Any diff holding
464
+ // 65 hashes — every lockfile bump — then tripped the cap and failed closed as
465
+ // a CRITICAL credential leak with no credential anywhere in it.
457
466
  const BASE64_MAX_CANDIDATES = 64;
467
+ const BASE64_MAX_TOKENS_EXAMINED = 8192;
458
468
  const BASE64_MAX_DECODED_BYTES = 64 * 1024;
469
+ // Per-blob ceiling, so one oversized payload cannot spend the whole budget and
470
+ // starve the blobs after it. The trade-off is deliberate: a credential buried
471
+ // past 8 KB inside a single blob is missed, where the old code caught it only
472
+ // by refusing to decode and then failing the entire diff closed. That refusal
473
+ // fired on every checked-in base64 asset, and a gate that cries wolf on
474
+ // ordinary input gets switched off. The cleartext scanners still run over the
475
+ // raw diff regardless.
476
+ const BASE64_MAX_BLOB_BYTES = 8 * 1024;
459
477
 
460
478
  /**
461
479
  * Share of characters that are printable ASCII (plus tab/newline/return).
@@ -480,14 +498,15 @@ function printableRatio(str) {
480
498
  function decodeBase64Blobs(text, onDecoded) {
481
499
  if (!text) return { decoded: [], capped: false };
482
500
  const decoded = [];
483
- let candidates = 0;
501
+ let examined = 0;
502
+ let retained = 0;
484
503
  let bytes = 0;
485
504
  let capped = false;
486
505
 
487
506
  BASE64_CANDIDATE.lastIndex = 0;
488
507
  let match;
489
508
  while ((match = BASE64_CANDIDATE.exec(text)) !== null) {
490
- if (candidates++ >= BASE64_MAX_CANDIDATES) {
509
+ if (examined++ >= BASE64_MAX_TOKENS_EXAMINED) {
491
510
  capped = true;
492
511
  break;
493
512
  }
@@ -497,10 +516,17 @@ function decodeBase64Blobs(text, onDecoded) {
497
516
  stdBlob += "=";
498
517
  }
499
518
 
500
- if (bytes + (stdBlob.length * 3) / 4 > BASE64_MAX_DECODED_BYTES) {
519
+ // An oversized blob is decoded up to a bounded prefix rather than skipped
520
+ // outright. Base64 decodes in independent 4-character groups, so a prefix
521
+ // is exact, and a credential near the head of a large payload still
522
+ // surfaces — where skipping used to hide it and then blame the whole diff.
523
+ const budget = Math.min(BASE64_MAX_BLOB_BYTES, BASE64_MAX_DECODED_BYTES - bytes);
524
+ if (budget <= 0) {
501
525
  capped = true;
502
- continue;
526
+ break;
503
527
  }
528
+ const maxChars = Math.floor(budget / 3) * 4;
529
+ if (stdBlob.length > maxChars) stdBlob = stdBlob.slice(0, maxChars);
504
530
 
505
531
  let plain;
506
532
  try {
@@ -509,8 +535,16 @@ function decodeBase64Blobs(text, onDecoded) {
509
535
  continue;
510
536
  }
511
537
  bytes += plain.length;
538
+
539
+ // A blob that decodes to binary has been examined and cleared. It is not a
540
+ // blind spot, so it must not spend a payload slot.
512
541
  if (printableRatio(plain) < 0.9) continue;
513
542
 
543
+ if (retained++ >= BASE64_MAX_CANDIDATES) {
544
+ capped = true;
545
+ break;
546
+ }
547
+
514
548
  decoded.push(plain);
515
549
  if (onDecoded) onDecoded(plain, rawBlob);
516
550
 
@@ -543,35 +577,141 @@ export function hasEncodedSecret(text) {
543
577
  return result.decoded.some((plain) => hasHighConfidenceSecret(plain));
544
578
  }
545
579
 
546
- export function scanDiff(diffTextStr = "", options = {}) {
547
- if (!diffTextStr) return { ok: true, findings: [] };
548
- const addedLines = diffTextStr
549
- .split("\n")
550
- .filter((line) => line.startsWith("+") && !line.startsWith("+++"))
551
- .map((line) => line.slice(1))
552
- .join("\n");
580
+ /**
581
+ * Group the added lines of a unified diff by the file they belong to.
582
+ *
583
+ * Line numbers come from the `@@` hunk headers and count the post-image, so a
584
+ * reported number matches what an editor shows after the change is applied.
585
+ * Both the file and the number are best-effort: a fragment with no headers —
586
+ * the shape `wizard-task.mjs` synthesises from a prompt — yields one anonymous
587
+ * segment, which is exactly the old whole-diff behaviour.
588
+ *
589
+ * @param {string} diffText
590
+ * @returns {Array<{ file: string|null, lines: Array<{ text: string, no: number|null }> }>}
591
+ */
592
+ function splitDiffByFile(diffText) {
593
+ const byFile = new Map();
594
+ let current = null;
595
+ let lineNo = null;
596
+
597
+ const select = (file) => {
598
+ if (!byFile.has(file)) byFile.set(file, { file, lines: [] });
599
+ current = byFile.get(file);
600
+ };
553
601
 
602
+ for (const line of diffText.split("\n")) {
603
+ if (line.startsWith("+++")) {
604
+ const name = line.slice(3).split("\t")[0].trim().replace(/^b\//, "");
605
+ select(name && name !== "/dev/null" ? name : null);
606
+ lineNo = null;
607
+ continue;
608
+ }
609
+ const hunk = /^@@ -\d+(?:,\d+)? \+(\d+)/.exec(line);
610
+ if (hunk) {
611
+ lineNo = Number(hunk[1]);
612
+ continue;
613
+ }
614
+ if (line.startsWith("+")) {
615
+ if (!current) select(null);
616
+ current.lines.push({ text: line.slice(1), no: lineNo });
617
+ if (lineNo !== null) lineNo++;
618
+ } else if (lineNo !== null && !line.startsWith("-") && !line.startsWith("\\")) {
619
+ lineNo++; // A context line advances the post-image just as an added one does.
620
+ }
621
+ }
622
+
623
+ return [...byFile.values()].filter((s) => s.lines.length > 0);
624
+ }
625
+
626
+ /**
627
+ * Classify a block of added lines. Returns the single most severe finding, or
628
+ * null when the block is clean.
629
+ *
630
+ * @param {string} addedLines
631
+ * @returns {{ severity: string, type: string, description: string, encoded: boolean }|null}
632
+ */
633
+ function classifyAddedLines(addedLines) {
554
634
  const { all: variants, normalized } = secretScanVariants(addedLines);
555
- const hasHigh = variants.some((v) => hasHighConfidenceSecret(v));
635
+ if (variants.some((v) => hasHighConfidenceSecret(v))) {
636
+ return { severity: "CRITICAL", type: "HIGH_CONFIDENCE_SECRET", encoded: false, description: "High-confidence secret pattern detected in added diff lines" };
637
+ }
556
638
  // Only worth decoding when nothing was found in the clear, and only against
557
639
  // the fully-normalised text: decoding is the expensive step, and the
558
640
  // intermediate variants differ from it in ways base64 blobs do not care about.
559
- const hasEncoded = !hasHigh && hasEncodedSecret(normalized);
560
- const hasLow = !hasHigh && !hasEncoded && variants.some((v) => hasLowConfidenceSecret(v));
561
- const findings = [];
562
-
563
- if (hasHigh) {
564
- findings.push({ severity: "CRITICAL", type: "HIGH_CONFIDENCE_SECRET", description: "High-confidence secret pattern detected in added diff lines" });
565
- } else if (hasEncoded) {
641
+ if (hasEncodedSecret(normalized)) {
566
642
  // Same type as the cleartext case: every gate that blocks on
567
643
  // HIGH_CONFIDENCE_SECRET should block on this too, and a new type would
568
644
  // have silently passed through the ones not updated. The description
569
645
  // carries the difference the operator needs.
570
- findings.push({ severity: "CRITICAL", type: "HIGH_CONFIDENCE_SECRET", description: "High-confidence secret pattern detected inside a base64-encoded value on an added diff line" });
571
- } else if (hasLow) {
572
- findings.push({ severity: "HIGH", type: "LOW_CONFIDENCE_SECRET", description: "Low-confidence secret or authorization token detected in added diff lines" });
646
+ return { severity: "CRITICAL", type: "HIGH_CONFIDENCE_SECRET", encoded: true, description: "High-confidence secret pattern detected inside a base64-encoded value on an added diff line" };
647
+ }
648
+ if (variants.some((v) => hasLowConfidenceSecret(v))) {
649
+ return { severity: "HIGH", type: "LOW_CONFIDENCE_SECRET", encoded: false, description: "Low-confidence secret or authorization token detected in added diff lines" };
650
+ }
651
+ return null;
652
+ }
653
+
654
+ /**
655
+ * Narrow a segment-level finding to the line that produced it.
656
+ *
657
+ * Only runs on a segment that has already been flagged, so the extra pass costs
658
+ * nothing on a clean diff. Returns null when no single line reproduces the
659
+ * verdict — a credential split across a concatenation belongs to the block, not
660
+ * to either half of it, and guessing one of them would point the operator at an
661
+ * innocent line.
662
+ *
663
+ * @param {Array<{ text: string, no: number|null }>} lines
664
+ * @param {string} type
665
+ * @returns {number|null}
666
+ */
667
+ function locateFindingLine(lines, type) {
668
+ for (const line of lines) {
669
+ if (line.no === null) continue;
670
+ const hit = classifyAddedLines(line.text);
671
+ if (hit && hit.type === type) return line.no;
672
+ }
673
+ return null;
674
+ }
675
+
676
+ export function scanDiff(diffTextStr = "", options = {}) {
677
+ if (!diffTextStr) return { ok: true, findings: [] };
678
+
679
+ const segments = splitDiffByFile(diffTextStr);
680
+ const findings = [];
681
+
682
+ for (const segment of segments) {
683
+ const hit = classifyAddedLines(segment.lines.map((l) => l.text).join("\n"));
684
+ if (!hit) continue;
685
+ const line = segment.file ? locateFindingLine(segment.lines, hit.type) : null;
686
+ const at = segment.file ? ` (${segment.file}${line ? `:${line}` : ""})` : "";
687
+ findings.push({
688
+ severity: hit.severity,
689
+ type: hit.type,
690
+ file: segment.file,
691
+ line,
692
+ description: `${hit.description}${at}`,
693
+ });
694
+ }
695
+
696
+ // Scanning per file loses anything that only matches across a file boundary,
697
+ // which the previous whole-diff join happened to catch. Rather than trade
698
+ // detection for attribution, fall back to the joined text when every file
699
+ // came back clean — the cost lands only on diffs with nothing to report.
700
+ if (findings.length === 0 && segments.length > 1) {
701
+ const hit = classifyAddedLines(segments.flatMap((s) => s.lines.map((l) => l.text)).join("\n"));
702
+ if (hit) {
703
+ findings.push({
704
+ severity: hit.severity,
705
+ type: hit.type,
706
+ file: null,
707
+ line: null,
708
+ description: `${hit.description} (spanning more than one file)`,
709
+ });
710
+ }
573
711
  }
574
712
 
713
+ const secretsOk = findings.length === 0;
714
+
575
715
  const edgeRes = checkEdgeRuntimeImports(diffTextStr, options);
576
716
  if (!edgeRes.ok) {
577
717
  for (const v of edgeRes.violations) {
@@ -583,12 +723,12 @@ export function scanDiff(diffTextStr = "", options = {}) {
583
723
  const crossPkgRes = checkCrossPackageImports(diffTextStr, root, options);
584
724
  if (!crossPkgRes.ok) {
585
725
  for (const v of crossPkgRes.violations) {
586
- findings.push({ severity: "HIGH", type: "CROSS_PACKAGE_BOUNDARY_VIOLATION", description: v.reason });
726
+ findings.push({ severity: "HIGH", type: "CROSS_PACKAGE_BOUNDARY_VIOLATION", file: v.file ?? null, description: v.reason });
587
727
  }
588
728
  }
589
729
 
590
730
  return {
591
- ok: !hasHigh && !hasEncoded && !hasLow && edgeRes.ok && crossPkgRes.ok,
731
+ ok: secretsOk && edgeRes.ok && crossPkgRes.ok,
592
732
  findings,
593
733
  };
594
734
  }
@@ -280,7 +280,7 @@ export function scorePromptFalsifiability(promptText, options = {}) {
280
280
  } else if (!verifyCmd) {
281
281
  score -= 15;
282
282
  issues.push({ type: "MISSING_ORACLE", message: "No automated test or build verification command specified.", penalty: 15 });
283
- suggestions.push("Specify a verification command using --verify (e.g. 'npm test').");
283
+ suggestions.push("Specify a verification command using --verify-cmd (e.g. 'npm test').");
284
284
  }
285
285
 
286
286
  // Final score clamping & letter grade assignment