jules-orchestrator-kit 0.41.0 → 0.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -10
- package/bin/agentctl.mjs +267 -34
- package/bin/init.js +12 -86
- package/package.json +1 -1
- package/src/engine.mjs +51 -1
- package/src/git.mjs +20 -1
- package/src/scaffold.mjs +283 -0
- package/src/security.mjs +164 -24
- package/src/task-optimizer.mjs +1 -1
package/src/engine.mjs
CHANGED
|
@@ -357,7 +357,22 @@ export async function gate(opts = {}) {
|
|
|
357
357
|
const runs = readVerifyRuns(root, stage.cmd);
|
|
358
358
|
flakyVerdictResult = flakyVerdict(runs);
|
|
359
359
|
if (flakyVerdictResult.verdict === "QUARANTINED") {
|
|
360
|
-
phases.push({
|
|
360
|
+
phases.push({
|
|
361
|
+
phase: "verify",
|
|
362
|
+
ok: false,
|
|
363
|
+
testResult,
|
|
364
|
+
buildResult,
|
|
365
|
+
failure: {
|
|
366
|
+
stageId: stage.id || "unit",
|
|
367
|
+
command: stage.cmd || null,
|
|
368
|
+
exitCode: res.status ?? null,
|
|
369
|
+
stdout: stdoutRedacted,
|
|
370
|
+
stderr: stderrRedacted,
|
|
371
|
+
diagnostics: [`Quarantined as flaky: ${flakyVerdictResult.reason || "alternating pass/fail across recent runs"}`],
|
|
372
|
+
},
|
|
373
|
+
flakyVerdict: flakyVerdictResult,
|
|
374
|
+
executionRecords,
|
|
375
|
+
});
|
|
361
376
|
appendTelemetry(root, "gate_phase", { phase: "verify", ok: false, quarantined: true });
|
|
362
377
|
appendTelemetry(root, "gate_finished", { ok: false, code: 8 });
|
|
363
378
|
return { ok: false, code: 8, phases, flakyVerdict: flakyVerdictResult };
|
|
@@ -413,6 +428,31 @@ export async function gate(opts = {}) {
|
|
|
413
428
|
|
|
414
429
|
const verifyOk = !failingCmd && !testTampered;
|
|
415
430
|
|
|
431
|
+
// What actually broke. Without this the verify phase reported `ok: false` and
|
|
432
|
+
// nothing else — not the stage, not the exit code, not a line of output — so
|
|
433
|
+
// the one gate failure the operator is expected to fix themselves was the
|
|
434
|
+
// only one that told them nothing about how. Both streams are already
|
|
435
|
+
// redacted at the point they were captured.
|
|
436
|
+
const verifyFailure = failingCmd
|
|
437
|
+
? {
|
|
438
|
+
stageId: failingCmd.stageId || failingCmd.phase || "verify",
|
|
439
|
+
command: failingCmd.command || null,
|
|
440
|
+
exitCode: failingCmd.status ?? null,
|
|
441
|
+
stdout: failingCmd.stdout || "",
|
|
442
|
+
stderr: failingCmd.stderr || "",
|
|
443
|
+
diagnostics: failingCmd.diagnostics || [],
|
|
444
|
+
}
|
|
445
|
+
: testTampered
|
|
446
|
+
? {
|
|
447
|
+
stageId: "test-integrity",
|
|
448
|
+
command: null,
|
|
449
|
+
exitCode: null,
|
|
450
|
+
stdout: "",
|
|
451
|
+
stderr: `Test files changed during the run (${preTestHash.slice(0, 12)} → ${postTestHash.slice(0, 12)}). evidence.strictTestLock treats a passing suite that the diff also rewrote as unproven.`,
|
|
452
|
+
diagnostics: [],
|
|
453
|
+
}
|
|
454
|
+
: null;
|
|
455
|
+
|
|
416
456
|
// Generate & persist Evidence Manifest
|
|
417
457
|
const evidenceManifest = generateEvidenceManifest(root, {
|
|
418
458
|
taskId: opts.taskId,
|
|
@@ -449,6 +489,7 @@ export async function gate(opts = {}) {
|
|
|
449
489
|
testResult,
|
|
450
490
|
buildResult,
|
|
451
491
|
serverResult,
|
|
492
|
+
failure: verifyFailure,
|
|
452
493
|
executionRecords,
|
|
453
494
|
testIntegrity: {
|
|
454
495
|
preTestHash,
|
|
@@ -834,6 +875,15 @@ export async function dispatch(task = {}, opts = {}) {
|
|
|
834
875
|
const roleObj = resolveRolePrompt(root, task.role);
|
|
835
876
|
if (roleObj) {
|
|
836
877
|
cleanPrompt = `${roleObj.content}\n\n${cleanPrompt}`.trim();
|
|
878
|
+
} else {
|
|
879
|
+
// A role reaching here comes from a task envelope or an internal
|
|
880
|
+
// synthesis rather than a typed flag, so the dispatch still proceeds with
|
|
881
|
+
// a generic agent — failing an automated heal swarm over a missing
|
|
882
|
+
// prompt file helps nobody. It must not proceed *silently* though: the
|
|
883
|
+
// caller asked for a specialist and is not getting one.
|
|
884
|
+
console.warn(
|
|
885
|
+
`⚠️ Role '${task.role}' has no prompt in .agent/prompts/ — dispatching without specialist context. Run 'agentctl init' to scaffold the shipped roles.`
|
|
886
|
+
);
|
|
837
887
|
}
|
|
838
888
|
}
|
|
839
889
|
|
package/src/git.mjs
CHANGED
|
@@ -27,6 +27,9 @@ export function runCmd(command, opts = {}) {
|
|
|
27
27
|
let args = [];
|
|
28
28
|
let useShell = false;
|
|
29
29
|
let shellCmd = "";
|
|
30
|
+
// True when `args` came from splitting a whitespace-separated string, which
|
|
31
|
+
// means no element can itself contain whitespace. See the Windows note below.
|
|
32
|
+
let tokenized = false;
|
|
30
33
|
|
|
31
34
|
if (Array.isArray(command)) {
|
|
32
35
|
binary = command[0];
|
|
@@ -40,9 +43,25 @@ export function runCmd(command, opts = {}) {
|
|
|
40
43
|
const tokens = trimmed.split(/\s+/).filter(Boolean);
|
|
41
44
|
binary = tokens[0] || "";
|
|
42
45
|
args = tokens.slice(1);
|
|
46
|
+
tokenized = true;
|
|
43
47
|
}
|
|
44
48
|
}
|
|
45
49
|
|
|
50
|
+
// Every package-manager entry point on Windows is a `.cmd` shim, and
|
|
51
|
+
// execFileSync cannot spawn one directly. `npm test` — the kit's own default
|
|
52
|
+
// verify command — therefore failed with `spawnSync npm ENOENT` on every
|
|
53
|
+
// Windows install, and the gate reported it as a plain non-zero verification
|
|
54
|
+
// rather than as an environment problem.
|
|
55
|
+
//
|
|
56
|
+
// Node's `shell: true` rebuilds the command line as `[file, ...args].join(" ")`
|
|
57
|
+
// and passes it to cmd.exe verbatim. That reconstruction is normally lossy —
|
|
58
|
+
// it does not quote an argument containing a space — but it is exact here,
|
|
59
|
+
// because these tokens were produced by splitting on whitespace in the first
|
|
60
|
+
// place. It is also only reached when the command contains no shell
|
|
61
|
+
// metacharacter, since those take the execSync branch above. An array command
|
|
62
|
+
// is excluded: its elements may legitimately contain spaces.
|
|
63
|
+
const winShim = tokenized && process.platform === "win32";
|
|
64
|
+
|
|
46
65
|
if (!binary && !useShell) {
|
|
47
66
|
if (opts.ignoreError) return { status: 0, stdout: "", stderr: "" };
|
|
48
67
|
throw new GateError("Empty command provided");
|
|
@@ -61,7 +80,7 @@ export function runCmd(command, opts = {}) {
|
|
|
61
80
|
: execFileSync(binary, args, {
|
|
62
81
|
cwd,
|
|
63
82
|
encoding: "utf-8",
|
|
64
|
-
shell:
|
|
83
|
+
shell: winShim,
|
|
65
84
|
stdio: ["ignore", "pipe", "pipe"],
|
|
66
85
|
env: opts.env || process.env,
|
|
67
86
|
timeout,
|
package/src/scaffold.mjs
ADDED
|
@@ -0,0 +1,283 @@
|
|
|
1
|
+
import { existsSync, mkdirSync, readdirSync, copyFileSync, readFileSync, appendFileSync, writeFileSync } from "node:fs";
|
|
2
|
+
import { join, dirname } from "node:path";
|
|
3
|
+
import { fileURLToPath } from "node:url";
|
|
4
|
+
import { detectPolyglotStack, detectEdgeRuntime } from "./stack-detector.mjs";
|
|
5
|
+
|
|
6
|
+
const KIT_ROOT = join(dirname(fileURLToPath(import.meta.url)), "..");
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Paths the kit writes at runtime and must never hand to its own gate.
|
|
10
|
+
*
|
|
11
|
+
* Without these, `agentctl init` left every ledger, evidence manifest and
|
|
12
|
+
* telemetry line untracked in the working tree. The gate audits that tree, so
|
|
13
|
+
* the kit's own bookkeeping showed up as a diff the agent was accused of
|
|
14
|
+
* making: first as scope violations against `.agent/config.yml`, then — once
|
|
15
|
+
* enough evidence files accumulated — as a CRITICAL secret verdict. A new user
|
|
16
|
+
* met both before dispatching a single task.
|
|
17
|
+
*/
|
|
18
|
+
export const RUNTIME_GITIGNORE_ENTRIES = [
|
|
19
|
+
".env",
|
|
20
|
+
".agent/history/",
|
|
21
|
+
".agent/state/",
|
|
22
|
+
".agent/evidence/",
|
|
23
|
+
".agent/handovers/",
|
|
24
|
+
".agent/jules-queue/.state/",
|
|
25
|
+
".agent/jules-queue/failed/",
|
|
26
|
+
".agent/jules-queue/.processing/",
|
|
27
|
+
".agent/jules-queue/completed/",
|
|
28
|
+
// A queued envelope is work-in-flight, not source. `.agent/jules-queue/**` is
|
|
29
|
+
// on the gate's deny list, so tracking the envelopes means the first gate
|
|
30
|
+
// after `task create` rejects the tree for the file `task create` just wrote.
|
|
31
|
+
// The negation has to follow the pattern it re-includes.
|
|
32
|
+
".agent/jules-queue/*.md",
|
|
33
|
+
"!.agent/jules-queue/README.md",
|
|
34
|
+
];
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Ensure `.gitignore` lists every runtime path in {@link RUNTIME_GITIGNORE_ENTRIES}.
|
|
38
|
+
*
|
|
39
|
+
* @param {string} root
|
|
40
|
+
* @returns {string[]} Entries newly appended (empty when already covered).
|
|
41
|
+
*/
|
|
42
|
+
export function ensureGitignore(root) {
|
|
43
|
+
const gitignorePath = join(root, ".gitignore");
|
|
44
|
+
const current = existsSync(gitignorePath) ? readFileSync(gitignorePath, "utf-8") : "";
|
|
45
|
+
const lines = new Set(current.split("\n").map((l) => l.trim()));
|
|
46
|
+
const missing = RUNTIME_GITIGNORE_ENTRIES.filter((e) => !lines.has(e));
|
|
47
|
+
if (missing.length === 0) return [];
|
|
48
|
+
|
|
49
|
+
const prefix = current && !current.endsWith("\n") ? "\n" : "";
|
|
50
|
+
appendFileSync(
|
|
51
|
+
gitignorePath,
|
|
52
|
+
`${prefix}\n# Jules Orchestrator runtime state & credentials\n${missing.join("\n")}\n`,
|
|
53
|
+
"utf-8"
|
|
54
|
+
);
|
|
55
|
+
return missing;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Copy every file from a directory shipped in the package into the target repo.
|
|
60
|
+
*
|
|
61
|
+
* @param {string} srcDir
|
|
62
|
+
* @param {string} destDir
|
|
63
|
+
* @param {boolean} force
|
|
64
|
+
* @returns {number} Files written.
|
|
65
|
+
*/
|
|
66
|
+
function copyDir(srcDir, destDir, force) {
|
|
67
|
+
if (!existsSync(srcDir) || srcDir === destDir) return 0;
|
|
68
|
+
mkdirSync(destDir, { recursive: true });
|
|
69
|
+
let written = 0;
|
|
70
|
+
for (const file of readdirSync(srcDir)) {
|
|
71
|
+
const src = join(srcDir, file);
|
|
72
|
+
const dest = join(destDir, file);
|
|
73
|
+
if (!existsSync(dest) || force) {
|
|
74
|
+
copyFileSync(src, dest);
|
|
75
|
+
written++;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
return written;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Generate stack-tailored contract templates (SPEC.md, CONSTRAINTS.md, DESIGN.md).
|
|
83
|
+
*
|
|
84
|
+
* @param {string} root - Project root
|
|
85
|
+
* @param {{ force?: boolean }} [options]
|
|
86
|
+
* @returns {string[]} Created contract file names
|
|
87
|
+
*/
|
|
88
|
+
export function scaffoldContracts(root = process.cwd(), options = {}) {
|
|
89
|
+
const force = Boolean(options.force);
|
|
90
|
+
const created = [];
|
|
91
|
+
|
|
92
|
+
const specPath = join(root, "SPEC.md");
|
|
93
|
+
if (!existsSync(specPath) || force) {
|
|
94
|
+
const specContent = `# SPEC — System & Product Contract
|
|
95
|
+
|
|
96
|
+
## Product
|
|
97
|
+
- Brief 1–2 sentence description of the system and target users.
|
|
98
|
+
|
|
99
|
+
## Core Loop
|
|
100
|
+
- Step-by-step lifecycle from input/event to final response/artifact.
|
|
101
|
+
|
|
102
|
+
## Goals
|
|
103
|
+
- [Goal 1: Core invariant functionality that MUST work]
|
|
104
|
+
- [Goal 2: Performance, throughput, or latency targets]
|
|
105
|
+
- [Goal 3: Test coverage & reliability criteria]
|
|
106
|
+
|
|
107
|
+
## Non-Goals
|
|
108
|
+
- [Explicit out-of-scope feature or abstraction]
|
|
109
|
+
- [Out-of-scope third-party dependencies or integrations]
|
|
110
|
+
|
|
111
|
+
## Definition of Done
|
|
112
|
+
- All verification test suites pass cleanly with 0 errors
|
|
113
|
+
- Zero security vulnerabilities and zero leaked secrets
|
|
114
|
+
- Diff size stays within the configured limit (<= 75 KB)
|
|
115
|
+
`;
|
|
116
|
+
writeFileSync(specPath, specContent, "utf-8");
|
|
117
|
+
created.push("SPEC.md");
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
const constraintsPath = join(root, "CONSTRAINTS.md");
|
|
121
|
+
if (!existsSync(constraintsPath) || force) {
|
|
122
|
+
const edgeInfo = detectEdgeRuntime(root);
|
|
123
|
+
const stackInfo = detectPolyglotStack(root);
|
|
124
|
+
|
|
125
|
+
let constraintsContent = "";
|
|
126
|
+
if (edgeInfo.edgePlatform === "cloudflare") {
|
|
127
|
+
constraintsContent = `# CONSTRAINTS — Cloudflare Workers / workerd
|
|
128
|
+
|
|
129
|
+
## Runtime Invariants
|
|
130
|
+
- Zero unbundled \`node:*\` imports in \`src/\` (rely on standard Web APIs or polyfilled modules)
|
|
131
|
+
- Bundle size: Max 10 MB total
|
|
132
|
+
- RAM memory limit: 128 MB
|
|
133
|
+
- Database & KV: Batch operations via \`db.batch([])\`, zero unbounded query loops
|
|
134
|
+
- Static content: Zero client-side JavaScript for static content
|
|
135
|
+
`;
|
|
136
|
+
} else if (stackInfo.stack === "cargo") {
|
|
137
|
+
constraintsContent = `# CONSTRAINTS — Rust Architecture
|
|
138
|
+
|
|
139
|
+
## Runtime & Safety Invariants
|
|
140
|
+
- Zero \`unsafe\` blocks unless explicitly audited and documented
|
|
141
|
+
- No unhandled \`.unwrap()\` or \`.expect()\` in production/request-handling code paths
|
|
142
|
+
- Clippy compliance: \`cargo clippy -- -D warnings\` must pass with 0 warnings
|
|
143
|
+
- Strict error propagation using \`Result<T, E>\` / \`thiserror\` / \`anyhow\`
|
|
144
|
+
`;
|
|
145
|
+
} else if (stackInfo.stack === "go") {
|
|
146
|
+
constraintsContent = `# CONSTRAINTS — Go Architecture
|
|
147
|
+
|
|
148
|
+
## Runtime & Safety Invariants
|
|
149
|
+
- Deterministic builds: \`CGO_ENABLED=0\`
|
|
150
|
+
- Explicit error handling: Never discard \`err\` returns (\`_ = err\` is strictly prohibited)
|
|
151
|
+
- Data race free: \`go test -race ./...\` must pass with 0 failures
|
|
152
|
+
- Strict struct tagging and deterministic serialization
|
|
153
|
+
`;
|
|
154
|
+
} else if (["python", "poetry", "uv", "pipenv"].includes(stackInfo.stack)) {
|
|
155
|
+
constraintsContent = `# CONSTRAINTS — Python Architecture
|
|
156
|
+
|
|
157
|
+
## Runtime & Safety Invariants
|
|
158
|
+
- Strict type annotations on all function signatures (\`mypy --strict\` passes)
|
|
159
|
+
- Zero unpinned dependencies in production manifests
|
|
160
|
+
- Lint & format cleanly with \`ruff\` or \`flake8\`/\`black\`
|
|
161
|
+
- Pytest suite passes 100% cleanly
|
|
162
|
+
`;
|
|
163
|
+
} else {
|
|
164
|
+
constraintsContent = `# CONSTRAINTS — Technical Invariants
|
|
165
|
+
|
|
166
|
+
## Architecture & Code Quality
|
|
167
|
+
- Zero third-party runtime dependencies in core orchestration/shared packages
|
|
168
|
+
- Diff Payload Budget: Keep diffs under 75 KB to prevent truncation
|
|
169
|
+
- Strict Test Lock: Never weaken assertions or delete failing tests to force green status
|
|
170
|
+
- Cross-Platform: Normalize all filesystem paths to POSIX slashes (\`/\`)
|
|
171
|
+
`;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
writeFileSync(constraintsPath, constraintsContent, "utf-8");
|
|
175
|
+
created.push("CONSTRAINTS.md");
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
// If web/UI stack is detected (Astro, Next, Svelte, Vue, React, Tailwind), scaffold DESIGN.md
|
|
179
|
+
const isWeb = existsSync(join(root, "astro.config.mjs")) ||
|
|
180
|
+
existsSync(join(root, "next.config.js")) ||
|
|
181
|
+
existsSync(join(root, "next.config.mjs")) ||
|
|
182
|
+
existsSync(join(root, "svelte.config.js")) ||
|
|
183
|
+
existsSync(join(root, "tailwind.config.js")) ||
|
|
184
|
+
existsSync(join(root, "tailwind.config.mjs")) ||
|
|
185
|
+
existsSync(join(root, "tailwind.config.ts"));
|
|
186
|
+
|
|
187
|
+
const designPath = join(root, "DESIGN.md");
|
|
188
|
+
if (isWeb && (!existsSync(designPath) || force)) {
|
|
189
|
+
const designContent = `# DESIGN — Visual Tokens & Design System
|
|
190
|
+
|
|
191
|
+
## Typography
|
|
192
|
+
- Headings: Clean sans-serif / geometric font
|
|
193
|
+
- Body: Readable system font / sans-serif
|
|
194
|
+
|
|
195
|
+
## Tokens (@theme)
|
|
196
|
+
- Consistent spacing scale (4px, 8px, 16px, 24px, 32px, 48px)
|
|
197
|
+
- Strict color tokens for surfaces, text, and borders
|
|
198
|
+
|
|
199
|
+
## Hard UI Rules
|
|
200
|
+
- Zero client-side JS for purely static content
|
|
201
|
+
- Accessible contrast ratios (WCAG AA minimum 4.5:1)
|
|
202
|
+
- Explicit hover, focus-visible, and active states on all interactive elements
|
|
203
|
+
`;
|
|
204
|
+
writeFileSync(designPath, designContent, "utf-8");
|
|
205
|
+
created.push("DESIGN.md");
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
return created;
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
/**
|
|
212
|
+
* Scaffold the repository assets the CLI's documented features depend on.
|
|
213
|
+
*
|
|
214
|
+
* This is the single source of truth for both entry points. `agentctl init`
|
|
215
|
+
* and `jules-init` used to scaffold different things: only the latter wrote
|
|
216
|
+
* AGENTS.md, the role prompts and the guardrails, while the README's quickstart
|
|
217
|
+
* pointed at the former. Anyone following the quickstart got a Jules that never
|
|
218
|
+
* saw the protocol and a `--role` flag with nothing to resolve against.
|
|
219
|
+
*
|
|
220
|
+
* Existing files are preserved unless `force` is set — re-running init is a
|
|
221
|
+
* routine way to pick up new presets and must not overwrite local edits.
|
|
222
|
+
*
|
|
223
|
+
* @param {string} [root=process.cwd()]
|
|
224
|
+
* @param {{ force?: boolean, contracts?: boolean }} [options]
|
|
225
|
+
* @returns {{ created: string[], gitignore: string[] }}
|
|
226
|
+
*/
|
|
227
|
+
export function scaffoldRepoAssets(root = process.cwd(), options = {}) {
|
|
228
|
+
const force = Boolean(options.force);
|
|
229
|
+
const created = [];
|
|
230
|
+
|
|
231
|
+
const agentDir = join(root, ".agent");
|
|
232
|
+
const queueDir = join(agentDir, "jules-queue");
|
|
233
|
+
for (const d of [agentDir, queueDir, join(queueDir, "completed"), join(agentDir, "rules"), join(agentDir, "prompts"), join(agentDir, "workflows")]) {
|
|
234
|
+
if (!existsSync(d)) mkdirSync(d, { recursive: true });
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
// AGENTS.md is how the agent learns the protocol at all, so an existing file
|
|
238
|
+
// is appended to rather than replaced: repositories that already brief their
|
|
239
|
+
// agents must not lose that briefing to a scaffolding step.
|
|
240
|
+
const agentsFile = join(root, "AGENTS.md");
|
|
241
|
+
const template = join(KIT_ROOT, "JULES_RULES_TEMPLATE.md");
|
|
242
|
+
if (existsSync(template) && template !== agentsFile) {
|
|
243
|
+
if (!existsSync(agentsFile) || force) {
|
|
244
|
+
copyFileSync(template, agentsFile);
|
|
245
|
+
created.push("AGENTS.md");
|
|
246
|
+
} else if (!readFileSync(agentsFile, "utf-8").includes("<MCP_DIRECTIVE>")) {
|
|
247
|
+
appendFileSync(agentsFile, `\n\n---\n\n${readFileSync(template, "utf-8")}`, "utf-8");
|
|
248
|
+
created.push("AGENTS.md (appended)");
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
if (copyDir(join(KIT_ROOT, ".agent/prompts"), join(agentDir, "prompts"), force) > 0) {
|
|
253
|
+
created.push(".agent/prompts/ (Overseer, Bolt, Sentinel, Janitor)");
|
|
254
|
+
}
|
|
255
|
+
if (copyDir(join(KIT_ROOT, ".agent/rules"), join(agentDir, "rules"), force) > 0) {
|
|
256
|
+
created.push(".agent/rules/");
|
|
257
|
+
}
|
|
258
|
+
if (copyDir(join(KIT_ROOT, ".agent/workflows"), join(agentDir, "workflows"), force) > 0) {
|
|
259
|
+
created.push(".agent/workflows/");
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
// Scaffold contract documents (SPEC.md, CONSTRAINTS.md, DESIGN.md)
|
|
263
|
+
if (options.contracts !== false) {
|
|
264
|
+
const contracts = scaffoldContracts(root, options);
|
|
265
|
+
created.push(...contracts);
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
// isTaskFile() skips README.md, so the queue can carry its own explanation
|
|
269
|
+
// without the runner mistaking it for a task envelope.
|
|
270
|
+
const queueReadme = join(queueDir, "README.md");
|
|
271
|
+
if (!existsSync(queueReadme)) {
|
|
272
|
+
writeFileSync(
|
|
273
|
+
queueReadme,
|
|
274
|
+
"# Task Queue\n\nEach `TASK-*.md` here is one queued task envelope.\n\n" +
|
|
275
|
+
"- `agentctl task create` writes them\n- `agentctl queue` dispatches them\n" +
|
|
276
|
+
"- Dispatched envelopes move to `completed/`; failures stay put for a re-run\n",
|
|
277
|
+
"utf-8"
|
|
278
|
+
);
|
|
279
|
+
created.push(".agent/jules-queue/README.md");
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
return { created, gitignore: ensureGitignore(root) };
|
|
283
|
+
}
|
package/src/security.mjs
CHANGED
|
@@ -454,8 +454,26 @@ function secretScanVariants(addedLines) {
|
|
|
454
454
|
// get encoded into a single CI variable.
|
|
455
455
|
const BASE64_CANDIDATE = /[A-Za-z0-9+/\-_]{20,}={0,2}/g;
|
|
456
456
|
|
|
457
|
+
// Budgets the decoder spends before it gives up and reports `capped`.
|
|
458
|
+
//
|
|
459
|
+
// The count that matters is payloads *retained* — blobs that decoded to text
|
|
460
|
+
// and so could be carrying a credential. Counting every token that merely
|
|
461
|
+
// matches the base64 alphabet instead made a digest indistinguishable from a
|
|
462
|
+
// payload: a sha256 hex string is 64 characters of that alphabet, decodes to
|
|
463
|
+
// binary, gets discarded, and used to consume a slot anyway. Any diff holding
|
|
464
|
+
// 65 hashes — every lockfile bump — then tripped the cap and failed closed as
|
|
465
|
+
// a CRITICAL credential leak with no credential anywhere in it.
|
|
457
466
|
const BASE64_MAX_CANDIDATES = 64;
|
|
467
|
+
const BASE64_MAX_TOKENS_EXAMINED = 8192;
|
|
458
468
|
const BASE64_MAX_DECODED_BYTES = 64 * 1024;
|
|
469
|
+
// Per-blob ceiling, so one oversized payload cannot spend the whole budget and
|
|
470
|
+
// starve the blobs after it. The trade-off is deliberate: a credential buried
|
|
471
|
+
// past 8 KB inside a single blob is missed, where the old code caught it only
|
|
472
|
+
// by refusing to decode and then failing the entire diff closed. That refusal
|
|
473
|
+
// fired on every checked-in base64 asset, and a gate that cries wolf on
|
|
474
|
+
// ordinary input gets switched off. The cleartext scanners still run over the
|
|
475
|
+
// raw diff regardless.
|
|
476
|
+
const BASE64_MAX_BLOB_BYTES = 8 * 1024;
|
|
459
477
|
|
|
460
478
|
/**
|
|
461
479
|
* Share of characters that are printable ASCII (plus tab/newline/return).
|
|
@@ -480,14 +498,15 @@ function printableRatio(str) {
|
|
|
480
498
|
function decodeBase64Blobs(text, onDecoded) {
|
|
481
499
|
if (!text) return { decoded: [], capped: false };
|
|
482
500
|
const decoded = [];
|
|
483
|
-
let
|
|
501
|
+
let examined = 0;
|
|
502
|
+
let retained = 0;
|
|
484
503
|
let bytes = 0;
|
|
485
504
|
let capped = false;
|
|
486
505
|
|
|
487
506
|
BASE64_CANDIDATE.lastIndex = 0;
|
|
488
507
|
let match;
|
|
489
508
|
while ((match = BASE64_CANDIDATE.exec(text)) !== null) {
|
|
490
|
-
if (
|
|
509
|
+
if (examined++ >= BASE64_MAX_TOKENS_EXAMINED) {
|
|
491
510
|
capped = true;
|
|
492
511
|
break;
|
|
493
512
|
}
|
|
@@ -497,10 +516,17 @@ function decodeBase64Blobs(text, onDecoded) {
|
|
|
497
516
|
stdBlob += "=";
|
|
498
517
|
}
|
|
499
518
|
|
|
500
|
-
|
|
519
|
+
// An oversized blob is decoded up to a bounded prefix rather than skipped
|
|
520
|
+
// outright. Base64 decodes in independent 4-character groups, so a prefix
|
|
521
|
+
// is exact, and a credential near the head of a large payload still
|
|
522
|
+
// surfaces — where skipping used to hide it and then blame the whole diff.
|
|
523
|
+
const budget = Math.min(BASE64_MAX_BLOB_BYTES, BASE64_MAX_DECODED_BYTES - bytes);
|
|
524
|
+
if (budget <= 0) {
|
|
501
525
|
capped = true;
|
|
502
|
-
|
|
526
|
+
break;
|
|
503
527
|
}
|
|
528
|
+
const maxChars = Math.floor(budget / 3) * 4;
|
|
529
|
+
if (stdBlob.length > maxChars) stdBlob = stdBlob.slice(0, maxChars);
|
|
504
530
|
|
|
505
531
|
let plain;
|
|
506
532
|
try {
|
|
@@ -509,8 +535,16 @@ function decodeBase64Blobs(text, onDecoded) {
|
|
|
509
535
|
continue;
|
|
510
536
|
}
|
|
511
537
|
bytes += plain.length;
|
|
538
|
+
|
|
539
|
+
// A blob that decodes to binary has been examined and cleared. It is not a
|
|
540
|
+
// blind spot, so it must not spend a payload slot.
|
|
512
541
|
if (printableRatio(plain) < 0.9) continue;
|
|
513
542
|
|
|
543
|
+
if (retained++ >= BASE64_MAX_CANDIDATES) {
|
|
544
|
+
capped = true;
|
|
545
|
+
break;
|
|
546
|
+
}
|
|
547
|
+
|
|
514
548
|
decoded.push(plain);
|
|
515
549
|
if (onDecoded) onDecoded(plain, rawBlob);
|
|
516
550
|
|
|
@@ -543,35 +577,141 @@ export function hasEncodedSecret(text) {
|
|
|
543
577
|
return result.decoded.some((plain) => hasHighConfidenceSecret(plain));
|
|
544
578
|
}
|
|
545
579
|
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
580
|
+
/**
|
|
581
|
+
* Group the added lines of a unified diff by the file they belong to.
|
|
582
|
+
*
|
|
583
|
+
* Line numbers come from the `@@` hunk headers and count the post-image, so a
|
|
584
|
+
* reported number matches what an editor shows after the change is applied.
|
|
585
|
+
* Both the file and the number are best-effort: a fragment with no headers —
|
|
586
|
+
* the shape `wizard-task.mjs` synthesises from a prompt — yields one anonymous
|
|
587
|
+
* segment, which is exactly the old whole-diff behaviour.
|
|
588
|
+
*
|
|
589
|
+
* @param {string} diffText
|
|
590
|
+
* @returns {Array<{ file: string|null, lines: Array<{ text: string, no: number|null }> }>}
|
|
591
|
+
*/
|
|
592
|
+
function splitDiffByFile(diffText) {
|
|
593
|
+
const byFile = new Map();
|
|
594
|
+
let current = null;
|
|
595
|
+
let lineNo = null;
|
|
596
|
+
|
|
597
|
+
const select = (file) => {
|
|
598
|
+
if (!byFile.has(file)) byFile.set(file, { file, lines: [] });
|
|
599
|
+
current = byFile.get(file);
|
|
600
|
+
};
|
|
553
601
|
|
|
602
|
+
for (const line of diffText.split("\n")) {
|
|
603
|
+
if (line.startsWith("+++")) {
|
|
604
|
+
const name = line.slice(3).split("\t")[0].trim().replace(/^b\//, "");
|
|
605
|
+
select(name && name !== "/dev/null" ? name : null);
|
|
606
|
+
lineNo = null;
|
|
607
|
+
continue;
|
|
608
|
+
}
|
|
609
|
+
const hunk = /^@@ -\d+(?:,\d+)? \+(\d+)/.exec(line);
|
|
610
|
+
if (hunk) {
|
|
611
|
+
lineNo = Number(hunk[1]);
|
|
612
|
+
continue;
|
|
613
|
+
}
|
|
614
|
+
if (line.startsWith("+")) {
|
|
615
|
+
if (!current) select(null);
|
|
616
|
+
current.lines.push({ text: line.slice(1), no: lineNo });
|
|
617
|
+
if (lineNo !== null) lineNo++;
|
|
618
|
+
} else if (lineNo !== null && !line.startsWith("-") && !line.startsWith("\\")) {
|
|
619
|
+
lineNo++; // A context line advances the post-image just as an added one does.
|
|
620
|
+
}
|
|
621
|
+
}
|
|
622
|
+
|
|
623
|
+
return [...byFile.values()].filter((s) => s.lines.length > 0);
|
|
624
|
+
}
|
|
625
|
+
|
|
626
|
+
/**
|
|
627
|
+
* Classify a block of added lines. Returns the single most severe finding, or
|
|
628
|
+
* null when the block is clean.
|
|
629
|
+
*
|
|
630
|
+
* @param {string} addedLines
|
|
631
|
+
* @returns {{ severity: string, type: string, description: string, encoded: boolean }|null}
|
|
632
|
+
*/
|
|
633
|
+
function classifyAddedLines(addedLines) {
|
|
554
634
|
const { all: variants, normalized } = secretScanVariants(addedLines);
|
|
555
|
-
|
|
635
|
+
if (variants.some((v) => hasHighConfidenceSecret(v))) {
|
|
636
|
+
return { severity: "CRITICAL", type: "HIGH_CONFIDENCE_SECRET", encoded: false, description: "High-confidence secret pattern detected in added diff lines" };
|
|
637
|
+
}
|
|
556
638
|
// Only worth decoding when nothing was found in the clear, and only against
|
|
557
639
|
// the fully-normalised text: decoding is the expensive step, and the
|
|
558
640
|
// intermediate variants differ from it in ways base64 blobs do not care about.
|
|
559
|
-
|
|
560
|
-
const hasLow = !hasHigh && !hasEncoded && variants.some((v) => hasLowConfidenceSecret(v));
|
|
561
|
-
const findings = [];
|
|
562
|
-
|
|
563
|
-
if (hasHigh) {
|
|
564
|
-
findings.push({ severity: "CRITICAL", type: "HIGH_CONFIDENCE_SECRET", description: "High-confidence secret pattern detected in added diff lines" });
|
|
565
|
-
} else if (hasEncoded) {
|
|
641
|
+
if (hasEncodedSecret(normalized)) {
|
|
566
642
|
// Same type as the cleartext case: every gate that blocks on
|
|
567
643
|
// HIGH_CONFIDENCE_SECRET should block on this too, and a new type would
|
|
568
644
|
// have silently passed through the ones not updated. The description
|
|
569
645
|
// carries the difference the operator needs.
|
|
570
|
-
|
|
571
|
-
}
|
|
572
|
-
|
|
646
|
+
return { severity: "CRITICAL", type: "HIGH_CONFIDENCE_SECRET", encoded: true, description: "High-confidence secret pattern detected inside a base64-encoded value on an added diff line" };
|
|
647
|
+
}
|
|
648
|
+
if (variants.some((v) => hasLowConfidenceSecret(v))) {
|
|
649
|
+
return { severity: "HIGH", type: "LOW_CONFIDENCE_SECRET", encoded: false, description: "Low-confidence secret or authorization token detected in added diff lines" };
|
|
650
|
+
}
|
|
651
|
+
return null;
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
/**
|
|
655
|
+
* Narrow a segment-level finding to the line that produced it.
|
|
656
|
+
*
|
|
657
|
+
* Only runs on a segment that has already been flagged, so the extra pass costs
|
|
658
|
+
* nothing on a clean diff. Returns null when no single line reproduces the
|
|
659
|
+
* verdict — a credential split across a concatenation belongs to the block, not
|
|
660
|
+
* to either half of it, and guessing one of them would point the operator at an
|
|
661
|
+
* innocent line.
|
|
662
|
+
*
|
|
663
|
+
* @param {Array<{ text: string, no: number|null }>} lines
|
|
664
|
+
* @param {string} type
|
|
665
|
+
* @returns {number|null}
|
|
666
|
+
*/
|
|
667
|
+
function locateFindingLine(lines, type) {
|
|
668
|
+
for (const line of lines) {
|
|
669
|
+
if (line.no === null) continue;
|
|
670
|
+
const hit = classifyAddedLines(line.text);
|
|
671
|
+
if (hit && hit.type === type) return line.no;
|
|
672
|
+
}
|
|
673
|
+
return null;
|
|
674
|
+
}
|
|
675
|
+
|
|
676
|
+
export function scanDiff(diffTextStr = "", options = {}) {
|
|
677
|
+
if (!diffTextStr) return { ok: true, findings: [] };
|
|
678
|
+
|
|
679
|
+
const segments = splitDiffByFile(diffTextStr);
|
|
680
|
+
const findings = [];
|
|
681
|
+
|
|
682
|
+
for (const segment of segments) {
|
|
683
|
+
const hit = classifyAddedLines(segment.lines.map((l) => l.text).join("\n"));
|
|
684
|
+
if (!hit) continue;
|
|
685
|
+
const line = segment.file ? locateFindingLine(segment.lines, hit.type) : null;
|
|
686
|
+
const at = segment.file ? ` (${segment.file}${line ? `:${line}` : ""})` : "";
|
|
687
|
+
findings.push({
|
|
688
|
+
severity: hit.severity,
|
|
689
|
+
type: hit.type,
|
|
690
|
+
file: segment.file,
|
|
691
|
+
line,
|
|
692
|
+
description: `${hit.description}${at}`,
|
|
693
|
+
});
|
|
694
|
+
}
|
|
695
|
+
|
|
696
|
+
// Scanning per file loses anything that only matches across a file boundary,
|
|
697
|
+
// which the previous whole-diff join happened to catch. Rather than trade
|
|
698
|
+
// detection for attribution, fall back to the joined text when every file
|
|
699
|
+
// came back clean — the cost lands only on diffs with nothing to report.
|
|
700
|
+
if (findings.length === 0 && segments.length > 1) {
|
|
701
|
+
const hit = classifyAddedLines(segments.flatMap((s) => s.lines.map((l) => l.text)).join("\n"));
|
|
702
|
+
if (hit) {
|
|
703
|
+
findings.push({
|
|
704
|
+
severity: hit.severity,
|
|
705
|
+
type: hit.type,
|
|
706
|
+
file: null,
|
|
707
|
+
line: null,
|
|
708
|
+
description: `${hit.description} (spanning more than one file)`,
|
|
709
|
+
});
|
|
710
|
+
}
|
|
573
711
|
}
|
|
574
712
|
|
|
713
|
+
const secretsOk = findings.length === 0;
|
|
714
|
+
|
|
575
715
|
const edgeRes = checkEdgeRuntimeImports(diffTextStr, options);
|
|
576
716
|
if (!edgeRes.ok) {
|
|
577
717
|
for (const v of edgeRes.violations) {
|
|
@@ -583,12 +723,12 @@ export function scanDiff(diffTextStr = "", options = {}) {
|
|
|
583
723
|
const crossPkgRes = checkCrossPackageImports(diffTextStr, root, options);
|
|
584
724
|
if (!crossPkgRes.ok) {
|
|
585
725
|
for (const v of crossPkgRes.violations) {
|
|
586
|
-
findings.push({ severity: "HIGH", type: "CROSS_PACKAGE_BOUNDARY_VIOLATION", description: v.reason });
|
|
726
|
+
findings.push({ severity: "HIGH", type: "CROSS_PACKAGE_BOUNDARY_VIOLATION", file: v.file ?? null, description: v.reason });
|
|
587
727
|
}
|
|
588
728
|
}
|
|
589
729
|
|
|
590
730
|
return {
|
|
591
|
-
ok:
|
|
731
|
+
ok: secretsOk && edgeRes.ok && crossPkgRes.ok,
|
|
592
732
|
findings,
|
|
593
733
|
};
|
|
594
734
|
}
|
package/src/task-optimizer.mjs
CHANGED
|
@@ -280,7 +280,7 @@ export function scorePromptFalsifiability(promptText, options = {}) {
|
|
|
280
280
|
} else if (!verifyCmd) {
|
|
281
281
|
score -= 15;
|
|
282
282
|
issues.push({ type: "MISSING_ORACLE", message: "No automated test or build verification command specified.", penalty: 15 });
|
|
283
|
-
suggestions.push("Specify a verification command using --verify (e.g. 'npm test').");
|
|
283
|
+
suggestions.push("Specify a verification command using --verify-cmd (e.g. 'npm test').");
|
|
284
284
|
}
|
|
285
285
|
|
|
286
286
|
// Final score clamping & letter grade assignment
|