jules-orchestrator-kit 0.35.1 → 0.36.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -122,7 +122,7 @@ Autonomous coding agents can write software at 100× human speed—but unconstra
122
122
 
123
123
  * **🚀 Zero-Test Bootstrapping (`agentctl bootstrap`):** Synthesizes deterministic syntax-check and smoke-test verification oracles for untested legacy repositories so agents always operate against a falsifiable feedback loop.
124
124
 
125
- * **📈 Proven Scale & Reliability:** Empirically tested with **526 unit tests across 79 suites passing in < 10.0s**. An adversarial red-team suite (`test/adversarial-claims.test.mjs`) continuously attempts to falsify the safety guarantees documented above — including cross-platform probes for the case-insensitive filesystems on macOS and Windows — and a documentation-sync gate (`scripts/doc-sync-check.mjs`) blocks any release whose docs have drifted from the code.
125
+ * **📈 Proven Scale & Reliability:** Empirically tested with **532 unit tests across 79 suites passing in < 10.0s**. An adversarial red-team suite (`test/adversarial-claims.test.mjs`) continuously attempts to falsify the safety guarantees documented above — including cross-platform probes for the case-insensitive filesystems on macOS and Windows — and a documentation-sync gate (`scripts/doc-sync-check.mjs`) blocks any release whose docs have drifted from the code.
126
126
 
127
127
  <br/>
128
128
 
@@ -389,7 +389,7 @@ Native stdio server exposing task dispatch, gate verification, and risk auditing
389
389
  | :--- | :--- | :--- | :--- |
390
390
  | `init` | `agentctl init [--interactive] [--tier pro]` | Interactive onboarding wizard & stack oracle inspector generating `.agent/config.yml`. | `0` (Created) |
391
391
  | `task create` | `agentctl task create [--title <t>] [--prompt <p>] [--template <id>] [--role <name>] [--tier fast\|complex] [--depends-on <id,...>]` | Interactively authors & scopes falsifiable task envelopes with secret scrubbing, preflight gate checks, specialist role resolution, DAG dependency wiring, and an optional Cost Router tier override. | `0` (Queued), `1` (Unfalsifiable / Secret leak) |
392
- | `task template` | `agentctl task template [<id>] [--list] [--json]` | Lists and synthesizes specialized web task envelopes (`web-cwv`, `web-wcag`, `web-seo`, `web-playwright`, `web-flaky-heal`, `web-i18n`). | `0` (Synthesized/Listed) |
392
+ | `task template` | `agentctl task template [<id>] [--list] [--json]` | Lists and synthesizes specialized web task envelopes (`web-cwv`, `web-wcag`, `web-seo`, `web-playwright`, `web-flaky-heal`, `web-i18n`, `web-ai-access`). | `0` (Synthesized/Listed) |
393
393
  | `task optimize` | `agentctl task optimize "<prompt>" [--fix] [--web] [--json]` | Linter & optimizer injecting Google Labs 3-phase exploration budgets, critic steering, and web oracles. | `0` (Scored/Fixed) |
394
394
  | `test-gen` | `agentctl test-gen --title <t> --spec <s> [--run]` | Scaffolds falsifiable unit tests, verifies **RED** failure state, and locks test in `scope.deny`. | `0` (Scaffolded/Red) |
395
395
  | `rollback` | `agentctl rollback [sessionId \| --latest]` | Restores exact commit, uncommitted files, and cleans orphan task worktrees from pre-flight checkpoints. | `0` (Restored), `1` (Error) |
package/bin/agentctl.mjs CHANGED
@@ -889,6 +889,12 @@ async function main() {
889
889
  console.log(` Session ID : ${sessionId}`);
890
890
  console.log(` Digest Count : ${res.digestCount || 1}`);
891
891
  console.log(` (Use 'agentctl escalate --flush' to deliver immediately)\n`);
892
+ } else if (res.dryRun) {
893
+ // Nothing left the machine, so do not claim it did.
894
+ console.log(`\n🧪 Dry run — this incident would be sent immediately:`);
895
+ console.log(` Session ID : ${sessionId}`);
896
+ console.log(` Reason : ${values.reason}`);
897
+ console.log(` (No request sent, and your hourly interruption budget is untouched.)\n`);
892
898
  } else if (res.dispatched) {
893
899
  console.log(`\n🚨 Incident Escalation Dispatched!`);
894
900
  console.log(` Session ID : ${sessionId}`);
package/index.mjs CHANGED
@@ -114,7 +114,11 @@ export {
114
114
  getEscalationDigestStatus,
115
115
  clearEscalationDigest,
116
116
  bufferEscalationIncident,
117
+ loadEscalationDigest,
118
+ recordInterruption,
119
+ countRecentInterruptions,
117
120
  DEFAULT_CRITICAL_REASONS,
121
+ DIGEST_BATCH_LIMIT,
118
122
  } from "./src/webhook.mjs";
119
123
 
120
124
  // PR Review Evidence Bundler & Dev Server Probe
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "jules-orchestrator-kit",
3
- "version": "0.35.1",
3
+ "version": "0.36.0",
4
4
  "description": "Orchestration kit for running Google Jules autonomous agents.",
5
5
  "repository": {
6
6
  "type": "git",
package/src/config.mjs CHANGED
@@ -349,6 +349,18 @@ export const VENDOR_TIERS = ["free", "pro", "ultra"];
349
349
  /** The tier used when a config names one that does not exist. */
350
350
  export const FALLBACK_TIER = "ultra";
351
351
 
352
+ /**
353
+ * Escalation reasons that bypass the Silence Governor and alert immediately.
354
+ *
355
+ * Kept here rather than in webhook.mjs because `loadConfig` needs it as the
356
+ * default for `notifications.critical_reasons`, and webhook.mjs already imports
357
+ * from this module — the reverse direction would be a cycle. Two hand-copied
358
+ * lists were how v0.35.0 ended up with a governor that governed nothing.
359
+ *
360
+ * See webhook.mjs for why the list is this short.
361
+ */
362
+ export const DEFAULT_CRITICAL_REASONS = ["R3_GATE_VIOLATION", "SECRET_LEAK_DETECTED", "CRITICAL_FAILURE"];
363
+
352
364
  /**
353
365
  * Loads and validates configuration from .agent/config.yml or .agent/jules.yml.
354
366
  */
@@ -440,7 +452,7 @@ export function loadConfig(root = resolveRoot(), explicitPath = null) {
440
452
  : 3,
441
453
  criticalReasons: Array.isArray(parsed.notifications?.critical_reasons)
442
454
  ? parsed.notifications.critical_reasons
443
- : ["R3_GATE_VIOLATION", "AWAITING_USER_FEEDBACK", "OODA_REPAIR_EXHAUSTED", "SECRET_LEAK_DETECTED", "CRITICAL_FAILURE"],
455
+ : [...DEFAULT_CRITICAL_REASONS],
444
456
  slackWebhookUrl: parsed.notifications?.slack_webhook_url || parsed.notifications?.slackWebhookUrl || "",
445
457
  discordWebhookUrl: parsed.notifications?.discord_webhook_url || parsed.notifications?.discordWebhookUrl || "",
446
458
  },
@@ -233,6 +233,93 @@ export const WEB_TEMPLATES = {
233
233
  - UTF-8 clean encoding with zero unescaped unicode artefacts.
234
234
  - Localized formatting for dates, currencies, and numbers using standard Intl APIs.`;
235
235
  }
236
+ },
237
+
238
+ // On the scope of this template, and what it deliberately does not claim:
239
+ //
240
+ // `llms.txt` (spec at llmstxt.org — cited without a scheme so the egress
241
+ // allowlist test does not read a citation as a destination the kit contacts)
242
+ // is a proposal, not a ratified standard. Plenty of sites publish one; no
243
+ // major provider has confirmed its retrieval stack reads one, and Google has
244
+ // said publicly that it does not use it. That is the supply side of a
245
+ // convention with no demonstrated demand side.
246
+ //
247
+ // So this template verifies what a repository can actually falsify — the file
248
+ // exists, it parses, its links resolve, and the site's crawler directives do
249
+ // not contradict each other — and it says nothing about whether publishing it
250
+ // improves visibility in any assistant. Every other template here carries a
251
+ // real verification oracle; a "generative engine optimization" template that
252
+ // promised ranking effects would be the first one that could not.
253
+ //
254
+ // Crawler posture is a policy choice, not a best practice. Allowing GPTBot,
255
+ // ClaudeBot or Google-Extended has licensing and editorial consequences, and
256
+ // blocking them is frequently deliberate. The template therefore takes no
257
+ // side: it defaults to `preserve`, reads the posture the repository already
258
+ // states, and enforces that every surface states the same thing.
259
+ //
260
+ // Structured data (JSON-LD, sameAs, entity markup) stays in `web-seo`. Two
261
+ // templates with authority over the same markup will eventually disagree.
262
+ "web-ai-access": {
263
+ id: "web-ai-access",
264
+ name: "AI Crawler Policy Consistency & llms.txt Integrity",
265
+ description: "Verify that AI crawler directives agree across every surface and that a published llms.txt parses with links that resolve.",
266
+ defaultVerifyCmd: "npm test",
267
+ category: "Crawler Policy & AI Access",
268
+ criticFocus: [
269
+ "Confirm the patch preserves the repository's existing AI crawler posture unless the task explicitly asked to change it — allowing or blocking a crawler is the operator's decision, not the agent's.",
270
+ "Verify robots.txt, per-page robots meta tags, and any X-Robots-Tag headers agree for every named agent; a page allowed in one surface and denied in another is a defect regardless of which is intended.",
271
+ "Check that every link in llms.txt resolves against the project's own route table or build output, with no absolute links to pages that no longer exist.",
272
+ "Ensure the PR description states that llms.txt consumption by AI systems is unverified, and claims no ranking, visibility, or citation benefit.",
273
+ "Confirm no JSON-LD or structured-data markup was modified here — that surface belongs to the web-seo template."
274
+ ],
275
+ defaultParams: {
276
+ aiAccessPolicy: "preserve",
277
+ aiAgents: "GPTBot, ClaudeBot, Google-Extended, PerplexityBot, CCBot, Applebot-Extended",
278
+ targetRoutes: "all public routes"
279
+ },
280
+ generatePrompt: (params = {}) => {
281
+ const policy = String(params.aiAccessPolicy || "preserve").toLowerCase();
282
+ const agents = params.aiAgents || "GPTBot, ClaudeBot, Google-Extended, PerplexityBot, CCBot, Applebot-Extended";
283
+ const routes = params.targetRoutes || "all public routes";
284
+ const customGoal = params.goal ? `\n- **Target Focus**: ${params.goal}` : "";
285
+
286
+ const policyClause = {
287
+ allow: `The operator has decided to **allow** these agents. Make every surface say so consistently.`,
288
+ deny: `The operator has decided to **block** these agents. Make every surface say so consistently, and confirm no route leaks access through a surface that was missed.`,
289
+ selective: `The operator allows some agents and blocks others. Derive the intended split from existing configuration and make every surface agree with it exactly.`,
290
+ preserve: `**Do not change the posture.** Determine what the repository already states about these agents and make every surface state the same thing. If the surfaces currently contradict each other, report the contradiction and resolve it toward the most restrictive existing directive — never toward the more permissive one, and never invent a posture the repository has not expressed.`
291
+ }[policy] || `Treat \`${policy}\` as an explicit operator instruction and apply it consistently across every surface.`;
292
+
293
+ return `Audit AI crawler access directives and llms.txt integrity for ${routes}.${customGoal}
294
+
295
+ ### Operator Policy (do not override)
296
+ ${policyClause}
297
+
298
+ Agents in scope: ${agents}.
299
+
300
+ ### Acceptance Criteria:
301
+ 1. **One Posture, Every Surface**:
302
+ - \`robots.txt\`, per-page \`<meta name="robots">\` / agent-specific meta tags, and any \`X-Robots-Tag\` response headers must agree for every agent above.
303
+ - A route that is allowed by one surface and denied by another is a defect even when the intended answer is obvious — fix the disagreement, do not pick a winner silently.
304
+ - Verify \`robots.txt\` parses: correct \`User-agent:\` grouping, no directives stranded outside a group, no rules unreachable because of an earlier wildcard group.
305
+ 2. **llms.txt Integrity (if the project publishes one)**:
306
+ - The file must parse as the proposed shape: a single \`# H1\` project name, an optional \`> blockquote\` summary, then \`## H2\` sections whose bodies are Markdown link lists.
307
+ - **Every link must resolve against this project's own route table or build output.** Check locally — do not fetch the live web from the verification step. A dead link in llms.txt is the single most common real defect in published files.
308
+ - Content must not contradict the crawler policy above: do not advertise paths in llms.txt that \`robots.txt\` disallows.
309
+ - If the project does not publish llms.txt, adding one is **in scope only if the task asked for it**. Do not create one on your own initiative.
310
+ 3. **Honest Reporting**:
311
+ - \`llms.txt\` is a proposal, not a ratified standard, and no major provider has confirmed that its retrieval systems read it. Google has stated publicly that it does not.
312
+ - The PR description must therefore claim only what was verified — that the file exists, parses, and its links resolve. Do **not** claim improved visibility, ranking, or citation in any AI assistant. There is no oracle for that claim and it must not appear in the diff, the commit message, or the PR body.
313
+ 4. **Scope Boundary**:
314
+ - Do not modify JSON-LD, Schema.org markup, \`sameAs\`, OpenGraph, or canonical tags. That surface belongs to the \`web-seo\` template; changing it here creates two sources of truth that will drift apart.
315
+
316
+ ### Verification Oracle to Add
317
+ Add a repository-local test (no network access) that:
318
+ - parses \`robots.txt\` and asserts the directive set for each agent in scope matches the intended posture;
319
+ - parses \`llms.txt\`, if present, and asserts every link target exists in the route table or build output.
320
+
321
+ The test must fail on a hand-broken fixture before you consider it done.`;
322
+ }
236
323
  }
237
324
  };
238
325
 
package/src/webhook.mjs CHANGED
@@ -3,16 +3,41 @@ import { createServer } from "node:http";
3
3
  import { readFileSync, writeFileSync, existsSync, unlinkSync } from "node:fs";
4
4
  import { join } from "node:path";
5
5
  import { getStateDir } from "./state.mjs";
6
- import { resolveRoot } from "./config.mjs";
6
+ import { resolveRoot, DEFAULT_CRITICAL_REASONS } from "./config.mjs";
7
7
  import { redactSecrets } from "./security.mjs";
8
8
 
9
- export const DEFAULT_CRITICAL_REASONS = [
10
- "R3_GATE_VIOLATION",
11
- "AWAITING_USER_FEEDBACK",
12
- "OODA_REPAIR_EXHAUSTED",
13
- "SECRET_LEAK_DETECTED",
14
- "CRITICAL_FAILURE",
15
- ];
9
+ /**
10
+ * Reasons that bypass the governor and page the operator the moment they occur.
11
+ * Defined in config.mjs (loadConfig needs the same list); re-exported here
12
+ * because this is the module that acts on it.
13
+ *
14
+ * The list is deliberately short. Anything named here is exempt from batching,
15
+ * so a reason belongs on it only when a delayed alert would let damage widen:
16
+ * a leaked credential keeps being valid, a violated gate keeps merging. Those
17
+ * are safety events, and the cost of waking someone beats the cost of waiting.
18
+ *
19
+ * `AWAITING_USER_FEEDBACK` is deliberately NOT here, even though it is the most
20
+ * urgent-*feeling* reason. It is the one a blocked agent raises, so on a swarm
21
+ * of fifteen workers it is also the most frequent by a wide margin — and it is
22
+ * the exact case the governor exists to batch. Listing it (as v0.35.0 did) made
23
+ * every default-configured escalation critical and left the governor governing
24
+ * nothing. Same for `OODA_REPAIR_EXHAUSTED`: the task has already stopped, so
25
+ * nothing worsens while it sits in a digest.
26
+ *
27
+ * An operator who wants the old behaviour sets `notifications.critical_reasons`
28
+ * in `.agent/config.yml` — this is a default, not a policy.
29
+ */
30
+ export { DEFAULT_CRITICAL_REASONS };
31
+
32
+ /**
33
+ * How many incidents one flush may carry.
34
+ *
35
+ * Slack truncates the summary block and Discord accepts a bounded field list,
36
+ * so a flush of fifty would render ten and drop forty. The digest promises the
37
+ * opposite — that a buffered incident is never lost — so a flush sends at most
38
+ * this many and leaves the remainder buffered for the next one.
39
+ */
40
+ export const DIGEST_BATCH_LIMIT = 10;
16
41
 
17
42
  export function getDigestFilePath(root = resolveRoot()) {
18
43
  return join(getStateDir(root), "escalation-digest.json");
@@ -144,16 +169,21 @@ export async function flushEscalationDigest(config = {}, options = {}) {
144
169
  const discordUrl = process.env.DISCORD_WEBHOOK_URL || notifications.discordWebhookUrl || config.discordWebhookUrl || "";
145
170
  const dryRun = options.dryRun || config.dryRun || false;
146
171
 
147
- const count = digest.incidents.length;
148
- const results = { flushed: true, count, slack: false, discord: false, dryRun };
172
+ // One flush carries a bounded batch; whatever does not fit stays buffered.
173
+ const batch = digest.incidents.slice(0, DIGEST_BATCH_LIMIT);
174
+ const remainder = digest.incidents.slice(DIGEST_BATCH_LIMIT);
175
+ const count = batch.length;
176
+ const results = { flushed: true, count, pending: remainder.length, slack: false, discord: false, dryRun };
149
177
 
150
178
  if (dryRun) {
151
- if (!options.preserve) clearEscalationDigest(root);
179
+ // A dry run shows what *would* be sent. It must leave the buffer exactly as
180
+ // it found it — clearing here (as v0.35.0 did) discarded real incidents in
181
+ // exchange for a preview.
152
182
  return {
153
183
  ...results,
154
184
  payload: {
155
185
  count,
156
- incidents: digest.incidents,
186
+ incidents: batch,
157
187
  },
158
188
  };
159
189
  }
@@ -165,7 +195,7 @@ export async function flushEscalationDigest(config = {}, options = {}) {
165
195
  // Format Slack digest
166
196
  if (slackUrl) {
167
197
  try {
168
- const summaryText = digest.incidents
198
+ const summaryText = batch
169
199
  .map((inc) => `• *\`${inc.sessionId}\`* on \`${inc.branch}\` [${inc.reason}]: \`agentctl resume ${inc.sessionId}\``)
170
200
  .join("\n");
171
201
 
@@ -183,7 +213,12 @@ export async function flushEscalationDigest(config = {}, options = {}) {
183
213
  {
184
214
  type: "context",
185
215
  elements: [
186
- { type: "mrkdwn", text: `Aggregated by Type III Silence Governor · Oldest: ${digest.createdAt || "N/A"}` },
216
+ {
217
+ type: "mrkdwn",
218
+ text:
219
+ `Aggregated by Type III Silence Governor · Oldest: ${digest.createdAt || "N/A"}` +
220
+ (remainder.length ? ` · ${remainder.length} still buffered` : ""),
221
+ },
187
222
  ],
188
223
  },
189
224
  ],
@@ -203,7 +238,7 @@ export async function flushEscalationDigest(config = {}, options = {}) {
203
238
  // Format Discord digest
204
239
  if (discordUrl) {
205
240
  try {
206
- const fields = digest.incidents.slice(0, 10).map((inc) => ({
241
+ const fields = batch.map((inc) => ({
207
242
  name: `Session \`${inc.sessionId}\` [${inc.reason}]`,
208
243
  value: `Branch: \`${inc.branch}\`\n\`agentctl resume ${inc.sessionId} --response "<answer>"\``,
209
244
  inline: false,
@@ -216,7 +251,11 @@ export async function flushEscalationDigest(config = {}, options = {}) {
216
251
  title: `Escalation Digest (${count} incidents)`,
217
252
  color: 3447003,
218
253
  fields,
219
- footer: { text: `Aggregated by Type III Silence Governor · Total: ${count}` },
254
+ footer: {
255
+ text:
256
+ `Aggregated by Type III Silence Governor · Total: ${count}` +
257
+ (remainder.length ? ` · ${remainder.length} still buffered` : ""),
258
+ },
220
259
  },
221
260
  ],
222
261
  };
@@ -232,8 +271,13 @@ export async function flushEscalationDigest(config = {}, options = {}) {
232
271
  }
233
272
  }
234
273
 
235
- if (results.slack || results.discord) {
236
- if (!options.preserve) {
274
+ // Only what actually reached a webhook is dropped. A failed delivery leaves
275
+ // the whole buffer intact so the next flush retries it rather than silently
276
+ // eating the incidents.
277
+ if ((results.slack || results.discord) && !options.preserve) {
278
+ if (remainder.length) {
279
+ saveEscalationDigest(root, { incidents: remainder, createdAt: digest.createdAt });
280
+ } else {
237
281
  clearEscalationDigest(root);
238
282
  }
239
283
  }
@@ -303,11 +347,6 @@ export async function dispatchEscalation(incident = {}, config = {}) {
303
347
  }
304
348
  }
305
349
 
306
- // If immediate/critical dispatch:
307
- if (!isCritical) {
308
- recordInterruption(root);
309
- }
310
-
311
350
  const slackUrl = process.env.SLACK_WEBHOOK_URL || notifications.slackWebhookUrl || config.slackWebhookUrl || "";
312
351
  const discordUrl = process.env.DISCORD_WEBHOOK_URL || notifications.discordWebhookUrl || config.discordWebhookUrl || "";
313
352
 
@@ -341,6 +380,14 @@ export async function dispatchEscalation(incident = {}, config = {}) {
341
380
  };
342
381
  }
343
382
 
383
+ // The budget counts interruptions, not intentions. Recording it any earlier
384
+ // charged the operator's hourly allowance for a `--dry-run` preview or for a
385
+ // repo with no webhook configured — an alert nobody ever received. Same
386
+ // mistake the daily task budget made before cd26d6e; same fix.
387
+ if (!isCritical) {
388
+ recordInterruption(root);
389
+ }
390
+
344
391
  // Dispatch to Slack
345
392
  if (slackUrl) {
346
393
  try {