jules-orchestrator-kit 0.36.0 → 0.37.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -104,7 +104,7 @@ Autonomous coding agents can write software at 100× human speed—but unconstra
104
104
 
105
105
  * **🔒 Zero Runtime Dependencies:** Built exclusively on Node.js 20+ built-ins (`node:fs`, `node:child_process`, `node:crypto`, `node:path`, `node:http`, `node:tty`, `node:test`). Zero third-party npm packages mean zero supply-chain CVE risk.
106
106
 
107
- * **🛡️ Fail-Closed Security Gatekeeper:** Unconditionally evaluates explicit Deny rules *before* Allow rules, matching against **canonicalised, case-folded paths** so `./`, `..`, mixed separators or a `.GitHub/` spelling cannot walk past a rule (the same repo is checked out on case-insensitive macOS and Windows filesystems). Redacts high-entropy secrets and PII from dry-runs and git diffs, blocks unsupported Node.js native module imports in Edge environments (Cloudflare Workers, Vercel Edge, Netlify Edge), and rejects PRs exceeding the 75 KB Diff Payload governor.
107
+ * **🛡️ Fail-Closed Security Gatekeeper:** Unconditionally evaluates explicit Deny rules *before* Allow rules, matching against **canonicalised, case-folded paths** so `./`, `..`, mixed separators or a `.GitHub/` spelling cannot walk past a rule (the same repo is checked out on case-insensitive macOS and Windows filesystems). Redacts high-entropy secrets and PII from dry-runs and git diffs — **including credentials wrapped in base64**, so a key inside a Kubernetes `Secret` manifest is not invisible to a line-oriented scanner — blocks unsupported Node.js native module imports in Edge environments (Cloudflare Workers, Vercel Edge, Netlify Edge), and rejects PRs exceeding the 75 KB Diff Payload governor.
108
108
 
109
109
  * **🔄 Autonomous OODA Self-Healing:** Captures test stderr/stdout, normalizes failure fingerprints, and feeds structured error contexts back into repair iterations (up to 3 automatic attempts) before human escalation.
110
110
 
@@ -122,7 +122,7 @@ Autonomous coding agents can write software at 100× human speed—but unconstra
122
122
 
123
123
  * **🚀 Zero-Test Bootstrapping (`agentctl bootstrap`):** Synthesizes deterministic syntax-check and smoke-test verification oracles for untested legacy repositories so agents always operate against a falsifiable feedback loop.
124
124
 
125
- * **📈 Proven Scale & Reliability:** Empirically tested with **532 unit tests across 79 suites passing in < 10.0s**. An adversarial red-team suite (`test/adversarial-claims.test.mjs`) continuously attempts to falsify the safety guarantees documented above — including cross-platform probes for the case-insensitive filesystems on macOS and Windows — and a documentation-sync gate (`scripts/doc-sync-check.mjs`) blocks any release whose docs have drifted from the code.
125
+ * **📈 Proven Scale & Reliability:** Empirically tested with **547 unit tests across 80 suites passing in < 10.0s**. An adversarial red-team suite (`test/adversarial-claims.test.mjs`) continuously attempts to falsify the safety guarantees documented above — every probe in it currently holds, with no open gaps — including cross-platform probes for the case-insensitive filesystems on macOS and Windows — and a documentation-sync gate (`scripts/doc-sync-check.mjs`) blocks any release whose docs have drifted from the code.
126
126
 
127
127
  <br/>
128
128
 
package/bin/agentctl.mjs CHANGED
@@ -334,18 +334,41 @@ async function main() {
334
334
  // The count is local-only, so an operator who knows their real usage
335
335
  // must be able to correct it. Appending `budget_released` keeps the
336
336
  // hash chain intact — the ledger is corrected forwards, never edited.
337
+ // An unrecognised flag is refused rather than ignored. `reset` writes to
338
+ // the ledger, and a misremembered flag — `--root`, `--force` — silently
339
+ // dropping through to a full release is the kind of misfire that only
340
+ // becomes visible after the count is already gone.
341
+ const known = new Set(["--dry-run", "--yes", "-y", "--all"]);
342
+ const unknown = args.slice(2).filter((a) => !known.has(a));
343
+ if (unknown.length) {
344
+ console.error(`Unrecognised option${unknown.length > 1 ? "s" : ""} for \`budget reset\`: ${unknown.join(", ")}`);
345
+ console.error("Accepted: --dry-run, --yes/-y, --all. Nothing was released.");
346
+ process.exit(2);
347
+ }
337
348
  const dryRun = args.includes("--dry-run");
338
349
  const confirmed = args.includes("--yes") || args.includes("-y");
350
+ // Committed reservations reached the provider, so releasing them makes
351
+ // the local count understate real usage. That has to be deliberate.
352
+ const includeCommitted = args.includes("--all");
339
353
  if (!dryRun && !confirmed) {
340
354
  const open = listOpenReservations(root);
341
- console.log(`Would release ${open.length} open reservation(s) from the last 24 hours.`);
355
+ const committed = open.filter((r) => r.committed).length;
356
+ const target = includeCommitted ? open.length : open.length - committed;
357
+ console.log(`Would release ${target} open reservation(s) from the last 24 hours.`);
358
+ if (committed > 0 && !includeCommitted) {
359
+ console.log(`Keeping ${committed} that reached Jules — those sessions really spent quota.`);
360
+ console.log("Add --all to release them too, if you know the count is still wrong.");
361
+ }
342
362
  console.log("This rewrites nothing — it appends `budget_released` entries.");
343
363
  console.log("Re-run with --yes to confirm, or --dry-run for detail.");
344
364
  process.exit(0);
345
365
  }
346
- const res = releaseOpenReservations({ root, dryRun, reason: "operator-reconcile" });
366
+ const res = releaseOpenReservations({ root, dryRun, includeCommitted, reason: "operator-reconcile" });
347
367
  const verb = dryRun ? "Would release" : "Released";
348
- console.log(`${verb} ${res.released} reservation(s) (${res.committed} committed, ${res.uncommitted} never closed).`);
368
+ console.log(`${verb} ${res.released} of ${res.open} open reservation(s) — ${res.uncommitted} never closed, ${res.committed} committed.`);
369
+ if (res.kept > 0) {
370
+ console.log(`Kept ${res.kept} committed reservation(s); re-run with --all to release those as well.`);
371
+ }
349
372
  if (!dryRun) {
350
373
  const after = budgetStatus(loadConfig(root), root);
351
374
  console.log(`Daily Budget : ${formatBudgetLine(after)}`);
@@ -363,14 +386,17 @@ async function main() {
363
386
  console.log(` ${b.note}`);
364
387
  console.log(` Window opened at ${b.windowStart} — the quota resets ${b.windowHours}h after each task,`);
365
388
  console.log(" not at midnight, so this count spans yesterday's ledger too.");
366
- console.log(` Open reservations in the window: ${listOpenReservations(root).length}`);
389
+ const openNow = listOpenReservations(root);
390
+ const committedNow = openNow.filter((r) => r.committed).length;
391
+ console.log(` Open reservations in the window: ${openNow.length} (${committedNow} confirmed dispatched, ${openNow.length - committedNow} never closed)`);
367
392
  console.log("");
368
393
  console.log(`Worker Slots : ${slots.concurrency} concurrent`);
369
394
  console.log(` ${slots.note}`);
370
395
  console.log("");
371
396
  console.log("The ledger counts this checkout only — sessions started from the Jules");
372
397
  console.log("web UI or another machine spend the same quota without appearing here.");
373
- console.log("Use `agentctl budget reset --yes` to reconcile a count you know is wrong.");
398
+ console.log("Use `agentctl budget reset --yes` to clear the ones that never closed,");
399
+ console.log("or `--all` to also give back the ones that did reach Jules.");
374
400
  process.exit(0);
375
401
  break;
376
402
  }
package/index.mjs CHANGED
@@ -13,6 +13,7 @@ export {
13
13
  matchesGlob,
14
14
  checkScope,
15
15
  scanDiff,
16
+ hasEncodedSecret,
16
17
  checkEdgeRuntimeImports,
17
18
  checkCrossPackageImports,
18
19
  } from "./src/security.mjs";
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "jules-orchestrator-kit",
3
- "version": "0.36.0",
3
+ "version": "0.37.0",
4
4
  "description": "Orchestration kit for running Google Jules autonomous agents.",
5
5
  "repository": {
6
6
  "type": "git",
package/src/budget.mjs CHANGED
@@ -276,32 +276,54 @@ export function listOpenReservations(root = resolveRoot(), opts = {}) {
276
276
  * forwards — never by editing or deleting the file, which would break the chain
277
277
  * and destroy the audit trail the ledger exists to provide.
278
278
  *
279
- * This is an operator override, not an inference. The kit cannot tell a
280
- * reservation that reached the provider from one whose process died first, so
281
- * only the operator knows whether the local count still reflects reality.
279
+ * This is an operator override, not an inference — but the two kinds of open
280
+ * reservation are not equally likely to be wrong, and by default only one of
281
+ * them is released.
282
+ *
283
+ * An **uncommitted** reservation was taken and never resolved: the process died
284
+ * between reserving the slot and learning what happened to it. That is the
285
+ * phantom this function exists to clear.
286
+ *
287
+ * A **committed** one carries proof that the dispatch reached the provider, so
288
+ * a real session exists and the quota really was spent. Releasing it makes the
289
+ * local count *understate* actual usage, which is the dangerous direction: the
290
+ * kit then dispatches confidently past the provider's real limit and gets
291
+ * refused. `includeCommitted` therefore has to be asked for.
292
+ *
293
+ * Legacy id-less reservations have no `budget_committed` entry that could ever
294
+ * name them, so they always read as uncommitted and are always released.
282
295
  *
283
296
  * @param {object} [opts]
284
297
  * @param {string} [opts.root]
285
298
  * @param {string} [opts.reason] - Recorded on every released entry.
286
299
  * @param {boolean} [opts.dryRun] - Report what would be released, write nothing.
287
- * @returns {{ released: number, committed: number, uncommitted: number, ids: string[], dryRun: boolean }}
300
+ * @param {boolean} [opts.includeCommitted] - Also release reservations that
301
+ * demonstrably reached the provider. Off by default.
302
+ * @returns {{ released: number, kept: number, open: number, committed: number,
303
+ * uncommitted: number, anonymous: number, ids: string[],
304
+ * includeCommitted: boolean, dryRun: boolean }}
288
305
  */
289
306
  export function releaseOpenReservations(opts = {}) {
290
307
  const root = opts.root || resolveRoot();
291
308
  const openRecords = listOpenReservations(root);
292
309
  const committed = openRecords.filter((r) => r.committed).length;
310
+ const includeCommitted = Boolean(opts.includeCommitted);
311
+ const targets = includeCommitted ? openRecords : openRecords.filter((r) => !r.committed);
293
312
 
294
313
  const result = {
295
- released: openRecords.length,
314
+ released: targets.length,
315
+ kept: openRecords.length - targets.length,
316
+ open: openRecords.length,
296
317
  committed,
297
318
  uncommitted: openRecords.length - committed,
298
- anonymous: openRecords.filter((r) => !r.reservationId).length,
299
- ids: openRecords.map((r) => r.reservationId).filter(Boolean),
319
+ anonymous: targets.filter((r) => !r.reservationId).length,
320
+ ids: targets.map((r) => r.reservationId).filter(Boolean),
321
+ includeCommitted,
300
322
  dryRun: Boolean(opts.dryRun),
301
323
  };
302
- if (opts.dryRun || openRecords.length === 0) return result;
324
+ if (opts.dryRun || targets.length === 0) return result;
303
325
 
304
- for (const rec of openRecords) {
326
+ for (const rec of targets) {
305
327
  const entry = { event: "budget_released", reason: opts.reason || "operator-reconcile" };
306
328
  if (rec.reservationId) {
307
329
  entry.reservationId = rec.reservationId;
package/src/security.mjs CHANGED
@@ -142,6 +142,19 @@ export function redactSecrets(text) {
142
142
  sanitized = sanitized.replace(pat, "[REDACTED_BY_SECURITY_GATE]");
143
143
  }
144
144
 
145
+ // A key the scanner can find inside a base64 blob must not survive redaction
146
+ // just because the literal bytes differ — otherwise scanDiff blocks the
147
+ // dispatch and the escalation payload leaks the very value it blocked on. The
148
+ // whole blob goes, not part of it: a partially-redacted encoding still
149
+ // decodes to the key.
150
+ const encoded = new Set();
151
+ decodeBase64Blobs(sanitized, (plain, blob) => {
152
+ if (hasHighConfidenceSecret(plain)) encoded.add(blob);
153
+ });
154
+ for (const blob of encoded) {
155
+ sanitized = sanitized.split(blob).join("[REDACTED_ENCODED_SECRET]");
156
+ }
157
+
145
158
  return sanitized;
146
159
  }
147
160
 
@@ -374,14 +387,118 @@ const STRING_CONCAT_JOIN = /(["'`])\s*\+\s*(["'`])/g;
374
387
  * Produces the variants of the added-line text that secret patterns are run
375
388
  * against: as-written, with invisible characters stripped, and with
376
389
  * source-level string concatenation collapsed.
390
+ *
391
+ * `normalized` is the last of those — every normalisation applied. It is
392
+ * returned separately for checks that are too expensive to run three times and
393
+ * gain nothing from the intermediate forms.
394
+ *
377
395
  * @param {string} addedLines
378
- * @returns {string[]}
396
+ * @returns {{ all: string[], normalized: string }}
379
397
  */
380
398
  function secretScanVariants(addedLines) {
381
399
  const stripped = addedLines.replace(INVISIBLE_CHARS, "");
382
400
  // Collapse `"AAA" +\n "BBB"` into `"AAABBB"` before matching.
383
401
  const dejoined = stripped.replace(/\s*\n\s*/g, " ").replace(STRING_CONCAT_JOIN, "");
384
- return [...new Set([addedLines, stripped, dejoined])];
402
+ return { all: [...new Set([addedLines, stripped, dejoined])], normalized: dejoined };
403
+ }
404
+
405
+ // Base64 is less an evasion technique than a file format. Every value in a
406
+ // Kubernetes Secret manifest is base64 by specification, and whole `.env` files
407
+ // get encoded into a single CI variable. A credential arriving that way is
408
+ // ordinary rather than adversarial — and a line-oriented scanner walks straight
409
+ // past it, which makes this the encoding most likely to carry a live key
410
+ // through the gate.
411
+ const BASE64_CANDIDATE = /[A-Za-z0-9+/]{24,}={0,2}/g;
412
+
413
+ // Decoding is cheap per blob and ruinous per diff if left unbounded. A patch
414
+ // that checks in a binary, a source map or a bundled font is otherwise enough
415
+ // to turn one scan into a memory event, so both the number of candidates and
416
+ // the total decoded size are capped. Exceeding a cap skips the remainder; it
417
+ // does not fail the scan, because a large diff is not evidence of a leak.
418
+ const BASE64_MAX_CANDIDATES = 64;
419
+ const BASE64_MAX_DECODED_BYTES = 64 * 1024;
420
+
421
+ /**
422
+ * Share of characters that are printable ASCII (plus tab/newline/return).
423
+ *
424
+ * `Buffer.from(str, "base64")` never throws — it discards what it cannot parse
425
+ * and returns whatever it managed to decode. So a hex digest or a random
426
+ * identifier of the right length "decodes" successfully into bytes that mean
427
+ * nothing. A wrapped credential, on the other hand, decodes to text: keys,
428
+ * PEM blocks and `.env` bodies are all ASCII. This ratio is what separates the
429
+ * two, and it removes nearly all of the noise before any pattern runs.
430
+ */
431
+ function printableRatio(str) {
432
+ if (!str) return 0;
433
+ let printable = 0;
434
+ for (let i = 0; i < str.length; i++) {
435
+ const c = str.charCodeAt(i);
436
+ if (c === 9 || c === 10 || c === 13 || (c >= 32 && c <= 126)) printable++;
437
+ }
438
+ return printable / str.length;
439
+ }
440
+
441
+ /**
442
+ * Decode the base64-looking blobs in `text` that plausibly hold text.
443
+ *
444
+ * @param {string} text
445
+ * @param {(plain: string, blob: string) => void} [onDecoded] - Called per blob.
446
+ * @returns {string[]}
447
+ */
448
+ function decodeBase64Blobs(text, onDecoded) {
449
+ if (!text) return [];
450
+ const decoded = [];
451
+ let candidates = 0;
452
+ let bytes = 0;
453
+
454
+ BASE64_CANDIDATE.lastIndex = 0;
455
+ let match;
456
+ while ((match = BASE64_CANDIDATE.exec(text)) !== null) {
457
+ if (candidates++ >= BASE64_MAX_CANDIDATES) break;
458
+ const blob = match[0];
459
+ // Valid base64 is a multiple of four characters including padding. This
460
+ // costs nothing and rejects three quarters of the alphanumeric runs — commit
461
+ // hashes, minified identifiers — that would otherwise be decoded for nothing.
462
+ if (blob.length % 4 !== 0) continue;
463
+ // Check the budget against what this blob *would* cost, not against what
464
+ // has already been spent — otherwise the first candidate decodes in full
465
+ // however large it is, and a single checked-in binary costs more than the
466
+ // cap was meant to allow. `continue`, not `break`: one oversized blob must
467
+ // not hide the smaller ones after it.
468
+ if (bytes + (blob.length * 3) / 4 > BASE64_MAX_DECODED_BYTES) continue;
469
+
470
+ let plain;
471
+ try {
472
+ plain = Buffer.from(blob, "base64").toString("utf-8");
473
+ } catch (_) {
474
+ continue;
475
+ }
476
+ bytes += plain.length;
477
+ if (printableRatio(plain) < 0.9) continue;
478
+
479
+ decoded.push(plain);
480
+ if (onDecoded) onDecoded(plain, blob);
481
+ }
482
+ BASE64_CANDIDATE.lastIndex = 0;
483
+ return decoded;
484
+ }
485
+
486
+ /**
487
+ * True when a base64-encoded value on an added line decodes to a structured
488
+ * credential.
489
+ *
490
+ * Deliberately runs the high-confidence patterns *only*. The low-confidence set
491
+ * is entropy- and keyword-driven, and decoded bytes are high-entropy by
492
+ * construction — pointing it at this output would flag close to every encoded
493
+ * blob in every repository. `AKIA[0-9A-Z]{16}` cannot match decoded noise;
494
+ * "looks secret-ish" always can. That asymmetry is the whole reason this check
495
+ * is safe to enable by default.
496
+ *
497
+ * @param {string} text
498
+ * @returns {boolean}
499
+ */
500
+ export function hasEncodedSecret(text) {
501
+ return decodeBase64Blobs(text).some((plain) => hasHighConfidenceSecret(plain));
385
502
  }
386
503
 
387
504
  export function scanDiff(diffTextStr = "", options = {}) {
@@ -392,13 +509,23 @@ export function scanDiff(diffTextStr = "", options = {}) {
392
509
  .map((line) => line.slice(1))
393
510
  .join("\n");
394
511
 
395
- const variants = secretScanVariants(addedLines);
512
+ const { all: variants, normalized } = secretScanVariants(addedLines);
396
513
  const hasHigh = variants.some((v) => hasHighConfidenceSecret(v));
397
- const hasLow = !hasHigh && variants.some((v) => hasLowConfidenceSecret(v));
514
+ // Only worth decoding when nothing was found in the clear, and only against
515
+ // the fully-normalised text: decoding is the expensive step, and the
516
+ // intermediate variants differ from it in ways base64 blobs do not care about.
517
+ const hasEncoded = !hasHigh && hasEncodedSecret(normalized);
518
+ const hasLow = !hasHigh && !hasEncoded && variants.some((v) => hasLowConfidenceSecret(v));
398
519
  const findings = [];
399
520
 
400
521
  if (hasHigh) {
401
522
  findings.push({ severity: "CRITICAL", type: "HIGH_CONFIDENCE_SECRET", description: "High-confidence secret pattern detected in added diff lines" });
523
+ } else if (hasEncoded) {
524
+ // Same type as the cleartext case: every gate that blocks on
525
+ // HIGH_CONFIDENCE_SECRET should block on this too, and a new type would
526
+ // have silently passed through the ones not updated. The description
527
+ // carries the difference the operator needs.
528
+ findings.push({ severity: "CRITICAL", type: "HIGH_CONFIDENCE_SECRET", description: "High-confidence secret pattern detected inside a base64-encoded value on an added diff line" });
402
529
  } else if (hasLow) {
403
530
  findings.push({ severity: "HIGH", type: "LOW_CONFIDENCE_SECRET", description: "Low-confidence secret or authorization token detected in added diff lines" });
404
531
  }
@@ -419,7 +546,7 @@ export function scanDiff(diffTextStr = "", options = {}) {
419
546
  }
420
547
 
421
548
  return {
422
- ok: !hasHigh && !hasLow && edgeRes.ok && crossPkgRes.ok,
549
+ ok: !hasHigh && !hasEncoded && !hasLow && edgeRes.ok && crossPkgRes.ok,
423
550
  findings,
424
551
  };
425
552
  }