jules-orchestrator-kit 0.38.2 → 0.41.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agent/prompts/Bolt.md +10 -7
- package/.agent/prompts/Janitor.md +6 -6
- package/.agent/prompts/Task_Template.md +6 -5
- package/.agent/rules/jules-protocol.md +1 -1
- package/JULES_RULES_TEMPLATE.md +12 -8
- package/README.md +45 -8
- package/bin/agentctl.mjs +297 -9
- package/index.mjs +15 -1
- package/package.json +1 -1
- package/scripts/ci-scope-guard.mjs +186 -0
- package/scripts/jules-merge-swarm.mjs +13 -3
- package/src/assertions.mjs +579 -0
- package/src/budget.mjs +48 -6
- package/src/config.mjs +69 -6
- package/src/dag-engine.mjs +22 -1
- package/src/engine.mjs +133 -23
- package/src/evidence.mjs +111 -2
- package/src/execution-envelope.mjs +23 -7
- package/src/flaky-ledger.mjs +4 -1
- package/src/ops/command-registry.mjs +32 -0
- package/src/ops/doctor-planner.mjs +9 -3
- package/src/ops/pr-harvest.mjs +349 -0
- package/src/provider.mjs +491 -25
- package/src/review-repair.mjs +85 -7
- package/src/risk.mjs +134 -26
- package/src/role-resolver.mjs +54 -2
- package/src/router.mjs +64 -6
- package/src/state.mjs +15 -3
- package/src/task-optimizer.mjs +9 -1
- package/src/telemetry.mjs +65 -0
- package/src/web-templates.mjs +372 -7
- package/src/wizard-init.mjs +13 -3
- package/src/wizard-task.mjs +76 -6
package/src/web-templates.mjs
CHANGED
|
@@ -14,7 +14,8 @@ export const WEB_TEMPLATES = {
|
|
|
14
14
|
"Check for un-optimized dynamic imports and excessive JavaScript bundle size.",
|
|
15
15
|
"Verify that all newly introduced images have explicit width/height and loading='lazy' decoding='async'.",
|
|
16
16
|
"Ensure font preloading uses fetchpriority='high' and prevents Cumulative Layout Shift (CLS).",
|
|
17
|
-
"Verify that critical CSS is not blocked by third-party analytics or non-critical scripts."
|
|
17
|
+
"Verify that critical CSS is not blocked by third-party analytics or non-critical scripts.",
|
|
18
|
+
"Ensure server:defer / dynamic deferrals are NOT applied to static marketing prose or static landing components."
|
|
18
19
|
],
|
|
19
20
|
defaultParams: {
|
|
20
21
|
lcpMaxMs: 1200,
|
|
@@ -41,8 +42,9 @@ export const WEB_TEMPLATES = {
|
|
|
41
42
|
### Required Architectural Actions:
|
|
42
43
|
1. Eliminate unused CSS and JavaScript chunks on '${page}'.
|
|
43
44
|
2. Preload above-the-fold hero images / fonts using \`<link rel="preload">\` with correct \`fetchpriority\`.
|
|
44
|
-
3. Wrap below-the-fold heavy components in
|
|
45
|
-
4. Ensure zero content shifts during font swaps or dynamic component mounting
|
|
45
|
+
3. Wrap below-the-fold heavy dynamic components in lazy-loading.
|
|
46
|
+
4. Ensure zero content shifts during font swaps or dynamic component mounting.
|
|
47
|
+
5. Invariant: Do NOT apply server-deferral / streaming to static marketing text; reserve for heavy dynamic data.`;
|
|
46
48
|
}
|
|
47
49
|
},
|
|
48
50
|
|
|
@@ -131,7 +133,8 @@ export const WEB_TEMPLATES = {
|
|
|
131
133
|
"Verify that Playwright tests use strict, accessible locators (getByRole, getByLabel, getByText) rather than brittle CSS selectors.",
|
|
132
134
|
"Ensure zero hard-coded arbitrary wait timeouts (e.g. page.waitForTimeout); use web assertions with automatic retries.",
|
|
133
135
|
"Check that mobile viewports (375px) do not introduce horizontal scrollbars (document.body.scrollWidth > window.innerWidth).",
|
|
134
|
-
"Confirm snapshot assertions use appropriate pixel threshold tolerance to avoid flaky anti-aliasing failures."
|
|
136
|
+
"Confirm snapshot assertions use appropriate pixel threshold tolerance to avoid flaky anti-aliasing failures.",
|
|
137
|
+
"Ensure tests execute cleanly in headless CI environments without requiring active X11/Wayland display servers."
|
|
135
138
|
],
|
|
136
139
|
defaultParams: {
|
|
137
140
|
viewports: "Mobile (375x667), Tablet (768x1024), Desktop (1440x900)",
|
|
@@ -146,18 +149,54 @@ export const WEB_TEMPLATES = {
|
|
|
146
149
|
|
|
147
150
|
### Test Harness & Assertion Criteria:
|
|
148
151
|
1. **Multi-Viewport Coverage**: Test and capture snapshots across: ${viewports}.
|
|
149
|
-
2. **
|
|
152
|
+
2. **Headless Sandbox Invariant**: Ensure all Playwright runs specify headless execution compatible with remote CI sandboxes.
|
|
153
|
+
3. **Responsive Invariants**:
|
|
150
154
|
- Zero horizontal overflow on mobile viewports (\`overflow-x\` containment).
|
|
151
155
|
- Tap target sizes for mobile touch buttons >= 44x44px.
|
|
152
156
|
- Hamburger / collapsible navigation expands and closes with correct ARIA attributes.
|
|
153
|
-
|
|
157
|
+
4. **Resilient Locators**:
|
|
154
158
|
- Use user-facing accessible locators (\`page.getByRole('button', { name: /submit/i })\`).
|
|
155
159
|
- Never use arbitrary \`waitForTimeout(3000)\` sleeps; use \`expect(locator).toBeVisible()\`.
|
|
156
|
-
|
|
160
|
+
5. **Visual Snapshots**:
|
|
157
161
|
- Verify visual snapshots pass with \`expect(page).toHaveScreenshot()\`.`;
|
|
158
162
|
}
|
|
159
163
|
},
|
|
160
164
|
|
|
165
|
+
"agent-dead-code-audit": {
|
|
166
|
+
id: "agent-dead-code-audit",
|
|
167
|
+
name: "Dead Code Audit & Safe Removal Protocol",
|
|
168
|
+
description: "Audit unused exports and dead code with the Audit-First Principle to prevent deleting dynamic runtime dependencies.",
|
|
169
|
+
defaultVerifyCmd: "npm test",
|
|
170
|
+
category: "Refactoring & Audit",
|
|
171
|
+
criticFocus: [
|
|
172
|
+
"Verify that flagged 'unused' exports are not dynamically imported at runtime or registered in dynamic router/plugin maps.",
|
|
173
|
+
"Ensure the audit generates a structured report (.agent/reports/dead-code-audit.md) with confidence ratings before applying destructive file deletions.",
|
|
174
|
+
"Check that zero core library entry points or framework-specific file-based routes are accidentally removed.",
|
|
175
|
+
"Confirm all surviving test suites and typechecks pass with 0 errors after any proposed removal."
|
|
176
|
+
],
|
|
177
|
+
defaultParams: {
|
|
178
|
+
targetScope: "apps/ or src/",
|
|
179
|
+
toolName: "Knip / ts-prune"
|
|
180
|
+
},
|
|
181
|
+
generatePrompt: (params = {}) => {
|
|
182
|
+
const scope = params.targetScope || "src/";
|
|
183
|
+
const tool = params.toolName || "Knip";
|
|
184
|
+
const customGoal = params.goal ? `\n- **Target Focus**: ${params.goal}` : "";
|
|
185
|
+
|
|
186
|
+
return `Perform an Audit-First dead code inspection and safe cleanup for ${scope} using ${tool}.${customGoal}
|
|
187
|
+
|
|
188
|
+
### Audit-First Safe Refactoring Invariants:
|
|
189
|
+
1. **Report Before Delete (Audit-First Principle)**:
|
|
190
|
+
- Generate a markdown audit report at \`.agent/reports/dead-code-audit.md\` detailing suspected unused files/exports and confidence levels (High / Medium / Low).
|
|
191
|
+
- DO NOT delete files or exports with dynamic runtime references (e.g. Astro/Next.js dynamic routes, plugin registries, CMS schemas).
|
|
192
|
+
2. **Conservative Scope**:
|
|
193
|
+
- Only remove files verified to have 0 dynamic or static references.
|
|
194
|
+
- When in doubt, document the finding in the audit report rather than deleting.
|
|
195
|
+
3. **Verification**:
|
|
196
|
+
- Run typecheck and unit tests to ensure zero broken call sites.`;
|
|
197
|
+
}
|
|
198
|
+
},
|
|
199
|
+
|
|
161
200
|
"web-flaky-heal": {
|
|
162
201
|
id: "web-flaky-heal",
|
|
163
202
|
name: "Playwright / Async Flakiness Auto-Healer",
|
|
@@ -320,6 +359,332 @@ Add a repository-local test (no network access) that:
|
|
|
320
359
|
|
|
321
360
|
The test must fail on a hand-broken fixture before you consider it done.`;
|
|
322
361
|
}
|
|
362
|
+
},
|
|
363
|
+
|
|
364
|
+
"agent-qa-mutation": {
|
|
365
|
+
id: "agent-qa-mutation",
|
|
366
|
+
name: "Agent Test Quality & Mutation Falsification",
|
|
367
|
+
description: "Audit agent-authored tests by deliberately mutating code under test to prove assertions fail on real defects.",
|
|
368
|
+
defaultVerifyCmd: "npm test",
|
|
369
|
+
category: "Quality & Testing",
|
|
370
|
+
criticFocus: [
|
|
371
|
+
"Ensure tests assert actual returned values and behaviors, not trivial shape checks (e.g. toBeDefined, assertTrue, status == 200).",
|
|
372
|
+
"Verify that no test mocks the unit or function under test (mocking the subject under test measures only the mock).",
|
|
373
|
+
"Confirm that every newly added test was proven to fail (turn red) when the underlying production logic was deliberately mutated.",
|
|
374
|
+
"Ensure collected test count matches actual test count, and reconcile any skipped or deleted tautological tests."
|
|
375
|
+
],
|
|
376
|
+
defaultParams: {
|
|
377
|
+
targetTestDir: "test/",
|
|
378
|
+
mutationStrategy: "invert conditions, drop return fields, alter numeric constants"
|
|
379
|
+
},
|
|
380
|
+
generatePrompt: (params = {}) => {
|
|
381
|
+
const testDir = params.targetTestDir || "test/";
|
|
382
|
+
const strategy = params.mutationStrategy || "invert conditions, drop return fields, alter numeric constants";
|
|
383
|
+
const customGoal = params.goal ? `\n- **Target Focus**: ${params.goal}` : "";
|
|
384
|
+
|
|
385
|
+
return `Audit and falsify agent-authored tests in '${testDir}'.${customGoal}
|
|
386
|
+
|
|
387
|
+
### Mutation & Assertion Quality Invariants:
|
|
388
|
+
1. **Prove Each Test Can Fail (Mutation Falsification)**:
|
|
389
|
+
- For every test in scope, deliberately introduce a defect into the code under test (${strategy}), run only that test, and confirm it turns red.
|
|
390
|
+
- Immediately revert every deliberate code mutation before proceeding to the next test.
|
|
391
|
+
- A test nobody has seen fail is an unverified assumption; delete or rewrite tests that remain green under deliberate mutation.
|
|
392
|
+
2. **Eliminate Vacuous & Tautological Checks**:
|
|
393
|
+
- Strip tests that mock the subject under test (measuring the mock rather than the implementation).
|
|
394
|
+
- Replace shallow shape assertions (\`toBeDefined()\`, \`assertTrue(result)\`, \`is not None\`) with precise value assertions derived from requirements.
|
|
395
|
+
3. **Reconcile Collected vs Passed Counts**:
|
|
396
|
+
- Compare test runner collected test count against actual test function count to detect uncollected or silently ignored test files.
|
|
397
|
+
- Document every deleted tautological test with its specific failure mode.`;
|
|
398
|
+
}
|
|
399
|
+
},
|
|
400
|
+
|
|
401
|
+
"agent-ci-falsify": {
|
|
402
|
+
id: "agent-ci-falsify",
|
|
403
|
+
name: "CI Pipeline Falsification & Exit Code Guard",
|
|
404
|
+
description: "Audit CI pipeline steps to eliminate swallowed exit codes, vacuous passes, or skipped checks.",
|
|
405
|
+
defaultVerifyCmd: "npm test",
|
|
406
|
+
category: "CI & Infrastructure",
|
|
407
|
+
criticFocus: [
|
|
408
|
+
"Verify no command discards exit codes via un-guarded pipes (e.g. \`cmd | tee out.txt\` without pipefail) or \`|| true\`.",
|
|
409
|
+
"Check that test runners fail when matching 0 test files instead of exiting cleanly with code 0.",
|
|
410
|
+
"Ensure CI matrix legs and path filters (\`paths:\`, \`if:\`) have not silently excluded required verification checks.",
|
|
411
|
+
"Confirm every CI check prints explicit item counts/coverage rather than bare unverified pass status."
|
|
412
|
+
],
|
|
413
|
+
defaultParams: {
|
|
414
|
+
workflowPath: ".github/workflows/"
|
|
415
|
+
},
|
|
416
|
+
generatePrompt: (params = {}) => {
|
|
417
|
+
const workflowPath = params.workflowPath || ".github/workflows/";
|
|
418
|
+
const customGoal = params.goal ? `\n- **Target Focus**: ${params.goal}` : "";
|
|
419
|
+
|
|
420
|
+
return `Audit and harden CI/CD workflows in '${workflowPath}' against silent passes and swallowed exit codes.${customGoal}
|
|
421
|
+
|
|
422
|
+
### Pipeline Integrity Invariants:
|
|
423
|
+
1. **Zero Swallowed Exit Codes**:
|
|
424
|
+
- Eliminate \`|| true\`, \`continue-on-error: true\` (unless justified in an adjacent comment), and \`set +e\` workarounds.
|
|
425
|
+
- Ensure all shell pipes preserve non-zero exit statuses (e.g. \`set -o pipefail\` in bash steps).
|
|
426
|
+
2. **Falsifiable Step Execution**:
|
|
427
|
+
- Verify test discovery patterns do not exit 0 on empty matches (e.g. runners reporting '0 tests collected' must fail).
|
|
428
|
+
- Ensure every check step logs explicit processed item counts (tests executed, files linted, types checked).
|
|
429
|
+
3. **Trigger & Filter Integrity**:
|
|
430
|
+
- Audit \`paths:\` and \`if:\` conditional expressions to ensure no critical security, typecheck, or test jobs are bypassed.`;
|
|
431
|
+
}
|
|
432
|
+
},
|
|
433
|
+
|
|
434
|
+
"agent-service-isolate": {
|
|
435
|
+
id: "agent-service-isolate",
|
|
436
|
+
name: "Cold Sandbox Test Isolation & Service Decoupling",
|
|
437
|
+
description: "Decouple test suites from live databases, external network services, or background daemons at existing architecture seams.",
|
|
438
|
+
defaultVerifyCmd: "npm test",
|
|
439
|
+
category: "Testing & Architecture",
|
|
440
|
+
criticFocus: [
|
|
441
|
+
"Verify default test runner executes cleanly from cold with zero live databases, Docker daemons, or network connections.",
|
|
442
|
+
"Ensure assertions are never weakened or deleted to simulate passing service calls.",
|
|
443
|
+
"Confirm tests that genuinely require live external services are explicitly marked/tagged and documented.",
|
|
444
|
+
"Verify test isolation seams use existing repository abstractions/interfaces rather than brute monkey-patching."
|
|
445
|
+
],
|
|
446
|
+
defaultParams: {
|
|
447
|
+
targetServices: "databases, caches, third-party network APIs"
|
|
448
|
+
},
|
|
449
|
+
generatePrompt: (params = {}) => {
|
|
450
|
+
const services = params.targetServices || "databases, caches, third-party network APIs";
|
|
451
|
+
const customGoal = params.goal ? `\n- **Target Focus**: ${params.goal}` : "";
|
|
452
|
+
|
|
453
|
+
return `Decouple test suite from external service dependencies (${services}) for reliable sandbox execution.${customGoal}
|
|
454
|
+
|
|
455
|
+
### Sandbox Decoupling Invariants:
|
|
456
|
+
1. **Zero Live Service Requirement for Default Suite**:
|
|
457
|
+
- The default test command must pass from cold with zero background daemons, databases, or external network connectivity.
|
|
458
|
+
- Fake external dependencies at existing boundary seams and repository interfaces.
|
|
459
|
+
2. **No Test Assertion Weakening**:
|
|
460
|
+
- Never weaken assertions or delete checks to simulate service responses; retain exact expectation semantics.
|
|
461
|
+
3. **Explicit Isolation Markers**:
|
|
462
|
+
- Categorize and tag integration tests requiring live services with explicit test runner markers (e.g. \`@integration\`, \`[live-service]\`).
|
|
463
|
+
- Provide a separate, documented command for executing live integration suites.`;
|
|
464
|
+
}
|
|
465
|
+
},
|
|
466
|
+
|
|
467
|
+
"agent-error-paths": {
|
|
468
|
+
id: "agent-error-paths",
|
|
469
|
+
name: "Error Path & Failure Recovery Stress Test",
|
|
470
|
+
description: "Exercise unexecuted error paths, catch blocks, and retry loops by intentionally inducing real failure conditions.",
|
|
471
|
+
defaultVerifyCmd: "npm test",
|
|
472
|
+
category: "Resilience & Security",
|
|
473
|
+
criticFocus: [
|
|
474
|
+
"Verify broad \`catch (e)\` or \`except Exception\` blocks do not swallow fatal runtime bugs or misspellings.",
|
|
475
|
+
"Confirm retry loops are bounded, backoff properly, and only operate on strictly idempotent actions.",
|
|
476
|
+
"Ensure error fallbacks fail in the safe/restrictive direction (e.g. denying permissions if auth check is unreachable).",
|
|
477
|
+
"Check that error messages clearly state the failed component, input, and actionable remedy."
|
|
478
|
+
],
|
|
479
|
+
defaultParams: {
|
|
480
|
+
targetModules: "src/"
|
|
481
|
+
},
|
|
482
|
+
generatePrompt: (params = {}) => {
|
|
483
|
+
const modules = params.targetModules || "src/";
|
|
484
|
+
const customGoal = params.goal ? `\n- **Target Focus**: ${params.goal}` : "";
|
|
485
|
+
|
|
486
|
+
return `Audit, exercise, and harden error recovery paths and catch handlers in '${modules}'.${customGoal}
|
|
487
|
+
|
|
488
|
+
### Resilience & Error Handling Invariants:
|
|
489
|
+
1. **Execute Every Catch Block Under Real Failure Conditions**:
|
|
490
|
+
- Test error handlers by inducing real failure conditions (closed socket, invalid schema, full disk, revoked token) rather than only reasoning about them.
|
|
491
|
+
- Narrow overly broad \`catch (e)\` / \`except Exception\` blocks to catch only expected operational error types.
|
|
492
|
+
2. **Safe Fallback Direction**:
|
|
493
|
+
- Ensure fallback branches fail in the restrictive/safe direction (e.g. permission checks must deny access on error, never allow).
|
|
494
|
+
3. **Bounded Idempotent Retries**:
|
|
495
|
+
- Verify all retry mechanisms enforce maximum attempt caps, exponential backoff with jitter, and only retry idempotent operations.
|
|
496
|
+
4. **Actionable Error Telemetry**:
|
|
497
|
+
- Ensure logged error messages specify the failed subsystem, input context, and actionable diagnostic guidance.
|
|
498
|
+
5. **Standalone Schema & Validation Testing**:
|
|
499
|
+
- When asserting schema validation error messages, export and test standalone validation schemas (e.g. \`schema.safeParse()\`) directly in unit tests rather than relying on heavyweight framework action mocks that may mask validation errors.`;
|
|
500
|
+
}
|
|
501
|
+
},
|
|
502
|
+
|
|
503
|
+
"agent-security-audit": {
|
|
504
|
+
id: "agent-security-audit",
|
|
505
|
+
name: "Agent-Authored Code Security & Permission Audit",
|
|
506
|
+
description: "Audit agent changes for TLS validation bypasses, hardcoded test credentials in repo history, and excessive workflow permissions.",
|
|
507
|
+
defaultVerifyCmd: "npm test",
|
|
508
|
+
category: "Security & Permissions",
|
|
509
|
+
criticFocus: [
|
|
510
|
+
"Verify zero TLS certificate verification bypasses (\`rejectUnauthorized: false\`, \`NODE_TLS_REJECT_UNAUTHORIZED=0\`, \`verify=False\`).",
|
|
511
|
+
"Check that no synthetic API keys, tokens, or private keys are committed in fixtures, configs, or git history.",
|
|
512
|
+
"Ensure GitHub Actions workflows and tokens are scoped to minimum necessary permissions (never \`permissions: write-all\`).",
|
|
513
|
+
"Confirm all string interpolations in SQL queries, shell commands, or HTML templates are properly escaped or parameterized."
|
|
514
|
+
],
|
|
515
|
+
defaultParams: {
|
|
516
|
+
diffRange: "HEAD~1..HEAD"
|
|
517
|
+
},
|
|
518
|
+
generatePrompt: (params = {}) => {
|
|
519
|
+
const diffRange = params.diffRange || "HEAD~1..HEAD";
|
|
520
|
+
const customGoal = params.goal ? `\n- **Target Focus**: ${params.goal}` : "";
|
|
521
|
+
|
|
522
|
+
return `Conduct a targeted security audit of agent-authored modifications in diff '${diffRange}'.${customGoal}
|
|
523
|
+
|
|
524
|
+
### Agent Code Security Invariants:
|
|
525
|
+
1. **Zero Transport & TLS Bypasses**:
|
|
526
|
+
- Strictly prohibit disabling TLS validation (\`rejectUnauthorized: false\`, \`NODE_TLS_REJECT_UNAUTHORIZED=0\`, \`verify=False\`, \`--no-check-certificate\`).
|
|
527
|
+
2. **Credential & Secret Scrubbing**:
|
|
528
|
+
- Verify zero synthetic or real API tokens, passwords, or private keys in test fixtures, examples, or commit history.
|
|
529
|
+
3. **Least Privilege CI Permissions**:
|
|
530
|
+
- Ensure CI workflows declare minimum explicit permissions; flag any \`permissions: write-all\` or unconstrained token scopes.
|
|
531
|
+
4. **Injection Prevention**:
|
|
532
|
+
- Verify zero un-sanitized string concatenations in database queries, child process executions, file path resolutions, or HTML rendering.`;
|
|
533
|
+
}
|
|
534
|
+
},
|
|
535
|
+
|
|
536
|
+
"deep-debug": {
|
|
537
|
+
id: "deep-debug",
|
|
538
|
+
name: "Deep Debug & Concurrency Race Condition Resolver",
|
|
539
|
+
description: "Deep root-cause exploration, async call-graph mapping, and deterministic reproduction of race conditions.",
|
|
540
|
+
defaultVerifyCmd: "npm test",
|
|
541
|
+
category: "Deep Think",
|
|
542
|
+
criticFocus: [
|
|
543
|
+
"Verify that async race conditions or timing bugs are deterministically reproduced with a failing test before applying fixes.",
|
|
544
|
+
"Check for lock contention, unhandled promise rejections, and memory retention in async state buffers.",
|
|
545
|
+
"Confirm mutation falsification: inverting the fix logic turns the reproduction test red.",
|
|
546
|
+
"Ensure zero new third-party dependencies and complete cross-platform timing parity (Linux/macOS/Windows)."
|
|
547
|
+
],
|
|
548
|
+
defaultParams: {
|
|
549
|
+
targetPath: "src/",
|
|
550
|
+
issueSummary: "state desynchronization or race condition"
|
|
551
|
+
},
|
|
552
|
+
generatePrompt: (params = {}) => {
|
|
553
|
+
const target = params.targetPath || "src/";
|
|
554
|
+
const issue = params.issueSummary || params.goal || "state desynchronization or race condition";
|
|
555
|
+
const customGoal = params.goal && params.goal !== issue ? `\n- **Target Focus**: ${params.goal}` : "";
|
|
556
|
+
|
|
557
|
+
return `Diagnose and surgically resolve concurrency/state race condition in '${target}': ${issue}.${customGoal}
|
|
558
|
+
|
|
559
|
+
### Deep Debug Protocol (Exploration & Oracle First):
|
|
560
|
+
1. **Silent Discovery & Call-Graph Tracing**:
|
|
561
|
+
- Trace the lifecycle, state transitions, and async call graph across '${target}'.
|
|
562
|
+
- Map out all concurrent read/write paths, shared buffers, or event-loop ticks where state desync occurs.
|
|
563
|
+
- Do NOT write production code until the root-cause hypothesis is verified.
|
|
564
|
+
2. **Deterministic Reproduction Oracle**:
|
|
565
|
+
- Formulate a deterministic test in the test suite reproducing the race condition under load or async delay.
|
|
566
|
+
- Verify the reproduction test FAILS cleanly (RED) against current code.
|
|
567
|
+
3. **Surgical Implementation & Positive Perimeter**:
|
|
568
|
+
- Implement the minimal, zero-dependency fix using native runtime built-ins only.
|
|
569
|
+
- Confine modifications strictly to the identified faulty module and its test suite.
|
|
570
|
+
4. **Mutation Falsification**:
|
|
571
|
+
- Invert the boolean or lock condition in your fix to prove the reproduction test turns RED (falsifiability proof).`;
|
|
572
|
+
}
|
|
573
|
+
},
|
|
574
|
+
|
|
575
|
+
"deep-feature": {
|
|
576
|
+
id: "deep-feature",
|
|
577
|
+
name: "Deep Feature TDD & Oracle Architecture",
|
|
578
|
+
description: "Oracle-first TDD architecture, contract discovery, and positive perimeter implementation.",
|
|
579
|
+
defaultVerifyCmd: "npm test",
|
|
580
|
+
category: "Deep Think",
|
|
581
|
+
criticFocus: [
|
|
582
|
+
"Ensure contract discovery aligns with existing repo error patterns ({ ok, code, error }) and JSDoc.",
|
|
583
|
+
"Verify boundary condition coverage (null/undefined, oversized buffers, empty inputs, proto-pollution).",
|
|
584
|
+
"Confirm positive operational perimeter: no edits outside designated target and test files.",
|
|
585
|
+
"Ensure clean type check, linting, and 100% test pass rate."
|
|
586
|
+
],
|
|
587
|
+
defaultParams: {
|
|
588
|
+
featureName: "New Module Feature",
|
|
589
|
+
targetModule: "src/"
|
|
590
|
+
},
|
|
591
|
+
generatePrompt: (params = {}) => {
|
|
592
|
+
const feature = params.featureName || params.goal || "New Module Feature";
|
|
593
|
+
const modulePath = params.targetModule || "src/";
|
|
594
|
+
const customGoal = params.goal && params.goal !== feature ? `\n- **Target Focus**: ${params.goal}` : "";
|
|
595
|
+
|
|
596
|
+
return `Implement '${feature}' in '${modulePath}' using Oracle-First TDD Architecture.${customGoal}
|
|
597
|
+
|
|
598
|
+
### Deep Feature TDD Protocol:
|
|
599
|
+
1. **Contract & Interface Discovery**:
|
|
600
|
+
- Inspect adjacent modules in '${modulePath}' to match conventions, error schemas, and signatures.
|
|
601
|
+
- Define the public API contract before writing code.
|
|
602
|
+
2. **TDD Oracle Formulation (Red Phase)**:
|
|
603
|
+
- Author comprehensive unit tests covering happy paths, boundary conditions, and invalid inputs.
|
|
604
|
+
- Run the test suite and confirm that newly added tests fail with expected assertions (RED).
|
|
605
|
+
3. **Surgical Implementation (Green Phase)**:
|
|
606
|
+
- Implement the feature using native built-in modules only (zero new third-party dependencies).
|
|
607
|
+
- Normalize all filesystem paths for cross-platform support.
|
|
608
|
+
4. **Adversarial Critic Review**:
|
|
609
|
+
- Verify memory allocations, recursion depth, and stream backpressure.
|
|
610
|
+
- Ensure 100% clean test passes with 0 lint errors across all supported platforms.`;
|
|
611
|
+
}
|
|
612
|
+
},
|
|
613
|
+
|
|
614
|
+
"deep-optimize": {
|
|
615
|
+
id: "deep-optimize",
|
|
616
|
+
name: "Deep Benchmark-Gated Performance Optimization",
|
|
617
|
+
description: "Profiling-first optimization with 5-run median benchmark gate (>=20% throughput or >=30% RAM improvement).",
|
|
618
|
+
defaultVerifyCmd: "npm test",
|
|
619
|
+
category: "Deep Think",
|
|
620
|
+
criticFocus: [
|
|
621
|
+
"Verify 5-run median benchmark comparison before and after the change.",
|
|
622
|
+
"Ensure byte-for-byte behavioral parity and zero regressions in the existing test suite.",
|
|
623
|
+
"Check for memory leaks, recursion stack limits, and catastrophic regex backtracking (O(N^2)).",
|
|
624
|
+
"Validate that hot loop optimizations do not sacrifice readability or error handling."
|
|
625
|
+
],
|
|
626
|
+
defaultParams: {
|
|
627
|
+
targetModule: "src/",
|
|
628
|
+
metricTarget: ">=20% throughput improvement or >=30% memory reduction"
|
|
629
|
+
},
|
|
630
|
+
generatePrompt: (params = {}) => {
|
|
631
|
+
const target = params.targetModule || "src/";
|
|
632
|
+
const metric = params.metricTarget || ">=20% throughput improvement or >=30% memory reduction";
|
|
633
|
+
const customGoal = params.goal ? `\n- **Target Focus**: ${params.goal}` : "";
|
|
634
|
+
|
|
635
|
+
return `Optimize runtime execution time and memory footprint of '${target}'.${customGoal}
|
|
636
|
+
|
|
637
|
+
### Benchmark-Gated Optimization Protocol:
|
|
638
|
+
1. **Profiling & Baseline Measurement**:
|
|
639
|
+
- Trace hot paths, object allocations inside loops, and redundant I/O in '${target}'.
|
|
640
|
+
- Establish a reproducible benchmark measuring execution time (median of 5 runs) and heap memory.
|
|
641
|
+
2. **Algorithmic Refactoring**:
|
|
642
|
+
- Eliminate redundant cloning, regex re-compilations, synchronous disk roundtrips, or O(N^2) lookups.
|
|
643
|
+
- Maintain 100% byte-for-byte behavioral parity with the existing implementation.
|
|
644
|
+
3. **Benchmark Gate Verification**:
|
|
645
|
+
- Target Metric: ${metric}.
|
|
646
|
+
- Execute 5 consecutive benchmark runs to confirm statistically significant gains over baseline.
|
|
647
|
+
4. **Behavioral Integrity Check**:
|
|
648
|
+
- Run the full test suite to guarantee zero regression or semantic drift.`;
|
|
649
|
+
}
|
|
650
|
+
},
|
|
651
|
+
|
|
652
|
+
"deep-harden": {
|
|
653
|
+
id: "deep-harden",
|
|
654
|
+
name: "Deep Adversarial Hardening & Chaos Mutation",
|
|
655
|
+
description: "Threat modeling, atomic I/O (safeAtomicWrite), chaos/fuzz injection, and structured error invariant gates.",
|
|
656
|
+
defaultVerifyCmd: "npm test",
|
|
657
|
+
category: "Deep Think",
|
|
658
|
+
criticFocus: [
|
|
659
|
+
"Verify TOCTOU safety and atomic file operations (temporary file + rename) for state persistence.",
|
|
660
|
+
"Check resistance to unicode smuggling, escape code stripping, and path traversal (<traversal>).",
|
|
661
|
+
"Ensure zero empty catch blocks or optimistic defaults returned from corrupt state.",
|
|
662
|
+
"Confirm mutation falsification: deliberately invalidating security checks triggers instant test failure."
|
|
663
|
+
],
|
|
664
|
+
defaultParams: {
|
|
665
|
+
targetSubsystem: "src/",
|
|
666
|
+
threatFocus: "malformed inputs, torn writes, unhandled runtime exceptions"
|
|
667
|
+
},
|
|
668
|
+
generatePrompt: (params = {}) => {
|
|
669
|
+
const target = params.targetSubsystem || "src/";
|
|
670
|
+
const threat = params.threatFocus || "malformed inputs, torn writes, unhandled runtime exceptions";
|
|
671
|
+
const customGoal = params.goal ? `\n- **Target Focus**: ${params.goal}` : "";
|
|
672
|
+
|
|
673
|
+
return `Adversarially harden '${target}' against ${threat}.${customGoal}
|
|
674
|
+
|
|
675
|
+
### Adversarial Hardening Protocol:
|
|
676
|
+
1. **Threat Modeling & Vulnerability Tracing**:
|
|
677
|
+
- Audit atomic I/O: Ensure file writes use safe atomic writes (temp file + rename) to prevent torn state on crash.
|
|
678
|
+
- Audit sanitization: Verify resistance against Unicode smuggling, ANSI escape injections, and traversal paths.
|
|
679
|
+
- Audit error propagation: Replace silent catch blocks with clinical, structured error objects.
|
|
680
|
+
2. **Chaos & Fuzz Test Authorship**:
|
|
681
|
+
- Author stress tests feeding proto-pollution payloads, oversized binary buffers, and simulated SIGKILL interruptions.
|
|
682
|
+
- Verify that all failure conditions emit structured error codes rather than crashing.
|
|
683
|
+
3. **Mutation Falsification Gate**:
|
|
684
|
+
- Invert security checks deliberately to prove the new tests fail immediately (kill the mutant).
|
|
685
|
+
4. **Regression Verification**:
|
|
686
|
+
- Ensure 100% of existing and new tests pass cleanly with zero errors.`;
|
|
687
|
+
}
|
|
323
688
|
}
|
|
324
689
|
};
|
|
325
690
|
|
package/src/wizard-init.mjs
CHANGED
|
@@ -116,7 +116,11 @@ export const BUILTIN_PRESETS = [
|
|
|
116
116
|
*/
|
|
117
117
|
export function planInit(root = process.cwd(), options = {}) {
|
|
118
118
|
const oracle = detectStackOracles(root);
|
|
119
|
-
|
|
119
|
+
// Scaffolding `pro` for a caller who never stated a plan hands a free account
|
|
120
|
+
// a 100-task budget and 8 concurrent workers it does not have. The same
|
|
121
|
+
// reasoning makes FALLBACK_TIER conservative in loadConfig(); the wizard has
|
|
122
|
+
// to agree with it or the manifest guards a ceiling the runtime does not.
|
|
123
|
+
const tierName = options.tier || FALLBACK_TIER;
|
|
120
124
|
// An unrecognised name resolves the same way loadConfig() resolves it, so the
|
|
121
125
|
// scaffolded limits always match what the runtime will later enforce.
|
|
122
126
|
const limits = TIER_PROFILES[tierName] || TIER_PROFILES[FALLBACK_TIER];
|
|
@@ -233,7 +237,7 @@ export async function runInitWizard(root = process.cwd(), options = {}) {
|
|
|
233
237
|
throw new Error("Non-interactive init requires explicit options or allowDefaults: true");
|
|
234
238
|
}
|
|
235
239
|
|
|
236
|
-
let selectedTier = options.tier || existingConfig.tier ||
|
|
240
|
+
let selectedTier = options.tier || existingConfig.tier || FALLBACK_TIER;
|
|
237
241
|
let testCmd = options.testCmd;
|
|
238
242
|
let buildCmd = options.buildCmd;
|
|
239
243
|
let selectedPresets = options.presets;
|
|
@@ -290,12 +294,18 @@ export async function runInitWizard(root = process.cwd(), options = {}) {
|
|
|
290
294
|
}
|
|
291
295
|
}
|
|
292
296
|
|
|
297
|
+
// `...options` first, for the same reason as in wizard-task.mjs: spreading it
|
|
298
|
+
// last let a caller-supplied `tier` overwrite the plan the user picked from
|
|
299
|
+
// the menu, so selecting Free or Ultra silently wrote whatever the CLI had
|
|
300
|
+
// passed. `selectedTier` is seeded from `options.tier`, so an explicit
|
|
301
|
+
// `--tier` still wins — it just wins by seeding the menu instead of by
|
|
302
|
+
// discarding the answer.
|
|
293
303
|
const plan = planInit(root, {
|
|
304
|
+
...options,
|
|
294
305
|
tier: selectedTier,
|
|
295
306
|
testCmd,
|
|
296
307
|
buildCmd,
|
|
297
308
|
presets: selectedPresets,
|
|
298
|
-
...options,
|
|
299
309
|
});
|
|
300
310
|
|
|
301
311
|
// Atomic write plan
|
package/src/wizard-task.mjs
CHANGED
|
@@ -9,15 +9,78 @@ import { select, input, confirm, spinner, isTTY } from "./tui.mjs";
|
|
|
9
9
|
import { scorePromptFalsifiability } from "./task-optimizer.mjs";
|
|
10
10
|
import { getWebTemplate, synthesizeWebEnvelope } from "./web-templates.mjs";
|
|
11
11
|
|
|
12
|
-
|
|
12
|
+
/**
|
|
13
|
+
* Maximum protected paths to name in a prompt before summarising.
|
|
14
|
+
*
|
|
15
|
+
* The list is there to steer the agent, not to be exhaustive — the gate is what
|
|
16
|
+
* enforces it. Past ~12 entries the footer starts crowding the task itself,
|
|
17
|
+
* which is the attention-drift failure `.agent/rules/jules-protocol.md` rule 16
|
|
18
|
+
* warns about.
|
|
19
|
+
*/
|
|
20
|
+
const FOOTER_PROTECTED_LIMIT = 12;
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Builds the hard-constraints footer appended to every dispatched task.
|
|
24
|
+
*
|
|
25
|
+
* The protected-path line is derived from the repository's resolved scope
|
|
26
|
+
* rather than written as a literal. It used to read "Do NOT modify
|
|
27
|
+
* package.json, pnpm-lock.yaml, tsconfig.json" for every project, which was
|
|
28
|
+
* both wrong and misleading in a Rust, Go, Python or PHP repo — while
|
|
29
|
+
* `BUILTIN_PROTECT` in config.mjs already listed `Cargo.toml`, `go.mod`,
|
|
30
|
+
* `pyproject.toml` and `composer.json` for the gate. The kit knew the right
|
|
31
|
+
* answer and told the agent a different one, so the agent could edit a file the
|
|
32
|
+
* gate would then reject, burning a repair turn on an avoidable violation.
|
|
33
|
+
*
|
|
34
|
+
* @param {object} [config] - Loaded config; `scope.protect`/`scope.deny` drive the output.
|
|
35
|
+
* @param {object} [opts]
|
|
36
|
+
* @param {string} [opts.baseBranch] - Branch to rebase onto before opening the PR.
|
|
37
|
+
* @param {number} [opts.diffKb] - Diff payload ceiling in KB.
|
|
38
|
+
* @returns {string}
|
|
39
|
+
*/
|
|
40
|
+
export function buildGuardrailFooter(config = {}, opts = {}) {
|
|
41
|
+
const baseBranch = opts.baseBranch || config.baseBranch || "main";
|
|
42
|
+
const diffKb = opts.diffKb || config.limits?.diffKb || 75;
|
|
43
|
+
|
|
44
|
+
const scope = config.scope || {};
|
|
45
|
+
|
|
46
|
+
// Key material and git internals are enforced by the gate but pointless to
|
|
47
|
+
// name here: no agent was going to edit `id_rsa`, and each entry spends
|
|
48
|
+
// footer budget that the build manifests actually need.
|
|
49
|
+
const NOT_WORTH_NAMING = /^(\.git\/|\*\*\/\.env|\*\*\/\*\.(pem|key|p12|pfx)$|\*\*\/id_rsa|\*\*\/\.npmrc|\*\*\/\.netrc|\*\.(pem|key)$|id_rsa)/;
|
|
50
|
+
|
|
51
|
+
// `protect` comes first because that is where the stack's build manifests
|
|
52
|
+
// live — `Cargo.toml`, `go.mod`, `pyproject.toml`, `composer.json` — and
|
|
53
|
+
// those are the files an agent actually reaches for and must be warned off.
|
|
54
|
+
const paths = [...new Set(
|
|
55
|
+
[...(scope.protect || []), ...(scope.deny || [])]
|
|
56
|
+
.filter((p) => typeof p === "string" && p.trim() && !NOT_WORTH_NAMING.test(p))
|
|
57
|
+
.map((p) => p.replace(/^\*\*\//, ""))
|
|
58
|
+
)];
|
|
59
|
+
|
|
60
|
+
const shown = paths.slice(0, FOOTER_PROTECTED_LIMIT);
|
|
61
|
+
const remainder = paths.length - shown.length;
|
|
62
|
+
const protectedLine = shown.length
|
|
63
|
+
? `- Do NOT modify these protected paths: ${shown.join(", ")}${remainder > 0 ? `, and ${remainder} more (run \`agentctl gate\` to see the full set)` : ""}.`
|
|
64
|
+
: "- Do NOT modify build configuration, lockfiles, or CI workflow files.";
|
|
65
|
+
|
|
66
|
+
return `
|
|
13
67
|
---
|
|
14
68
|
HARD CONSTRAINTS:
|
|
15
|
-
|
|
16
|
-
- Diff Payload Governor: Keep total diff payload under
|
|
69
|
+
${protectedLine}
|
|
70
|
+
- Diff Payload Governor: Keep total diff payload under ${diffKb} KB (\`git diff | wc -c\`).
|
|
17
71
|
- Falsifiable & Evidence-Based: Attach full terminal verification output to PR. Never weaken assertions or delete failing tests to force a pass.
|
|
18
72
|
- Read-Before-Write: Inspect existing symbol signatures, definitions, and call sites before making edits.
|
|
19
|
-
-
|
|
73
|
+
- Remove any scratch files you created for debugging before submitting. Do not delete files that are part of the project.
|
|
74
|
+
- BEFORE opening the PR: Run \`git fetch origin ${baseBranch} && git rebase origin/${baseBranch}\`, then re-verify.
|
|
20
75
|
`;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Stack-neutral fallback for callers with no config in hand.
|
|
80
|
+
* Prefer `buildGuardrailFooter(config)`, which names the repository's real
|
|
81
|
+
* protected paths instead of guessing at an ecosystem.
|
|
82
|
+
*/
|
|
83
|
+
export const GUARDRAIL_FOOTER = buildGuardrailFooter();
|
|
21
84
|
|
|
22
85
|
const TRIVIAL_ORACLES = new Set(["true", "echo", ":", "false", "exit 0", "exit 1", "echo ok"]);
|
|
23
86
|
|
|
@@ -139,7 +202,7 @@ ${rawPrompt}
|
|
|
139
202
|
[VERIFICATION ORACLE]
|
|
140
203
|
Test/Verification Command: ${verifyCmd || "(None)"}
|
|
141
204
|
|
|
142
|
-
${
|
|
205
|
+
${buildGuardrailFooter(config)}`;
|
|
143
206
|
|
|
144
207
|
const promptAnalysis = scorePromptFalsifiability(rawPrompt, { rootDir: root, verifyCmd });
|
|
145
208
|
|
|
@@ -245,14 +308,21 @@ export async function runTaskCreateWizard(root = process.cwd(), options = {}) {
|
|
|
245
308
|
repoless = await confirm("Dispatch in Repoless mode (no repo source context)?", false, options);
|
|
246
309
|
}
|
|
247
310
|
|
|
311
|
+
// `...options` must come first. It used to come last, and because the CLI
|
|
312
|
+
// builds its options object from parseArgs — every key present, every unpassed
|
|
313
|
+
// flag `undefined` — spreading it afterwards overwrote each answer the user
|
|
314
|
+
// had just typed with `undefined`. `agentctl task create` then died on
|
|
315
|
+
// "Task prompt cannot be empty" no matter what was entered. The locals below
|
|
316
|
+
// are all seeded from `options`, so putting them last preserves flag values
|
|
317
|
+
// while letting an interactive answer win.
|
|
248
318
|
const plan = planTaskCreate(root, {
|
|
319
|
+
...options,
|
|
249
320
|
title,
|
|
250
321
|
prompt: promptText,
|
|
251
322
|
verifyCmd,
|
|
252
323
|
autoPr,
|
|
253
324
|
requirePlanApproval,
|
|
254
325
|
repoless,
|
|
255
|
-
...options,
|
|
256
326
|
});
|
|
257
327
|
|
|
258
328
|
// Perform Gate Preflight
|