ccqa 1.39.0 → 1.40.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -4
- package/dist/bin/ccqa.mjs +748 -543
- package/dist/hub-client/index.d.mts +56 -1
- package/dist/hub-client/index.mjs +6 -0
- package/dist/package.json +1 -1
- package/dist/runtime/test-helpers.mjs +1 -1
- package/dist/{spawn-ab-CRIVfWpw.mjs → spawn-ab-Bm34WBui.mjs} +3 -1
- package/package.json +1 -1
package/dist/bin/ccqa.mjs
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import { HubApiError, createHubClient, hubRequest } from "../hub-client/index.mjs";
|
|
3
3
|
import { t as EVIDENCE_DIR_ENV } from "../evidence-constants-C425F7ZG.mjs";
|
|
4
|
-
import { a as formatAgentBrowserUnavailableMessage, i as assertAgentBrowserAvailable, n as spawnAB, o as pathWithAgentBrowserShim, r as AgentBrowserUnavailableError, s as resolveAgentBrowserBin$1, t as sleepSync } from "../spawn-ab-
|
|
4
|
+
import { a as formatAgentBrowserUnavailableMessage, i as assertAgentBrowserAvailable, n as spawnAB, o as pathWithAgentBrowserShim, r as AgentBrowserUnavailableError, s as resolveAgentBrowserBin$1, t as sleepSync } from "../spawn-ab-Bm34WBui.mjs";
|
|
5
5
|
import { createRequire } from "node:module";
|
|
6
6
|
import { Command } from "commander";
|
|
7
7
|
import { accessSync, appendFileSync, createWriteStream, existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
|
@@ -19,6 +19,7 @@ import { promisify } from "node:util";
|
|
|
19
19
|
import { createInterface } from "node:readline/promises";
|
|
20
20
|
import { createServer } from "node:http";
|
|
21
21
|
import { gunzipSync, gzipSync } from "node:zlib";
|
|
22
|
+
import { setTimeout as setTimeout$1 } from "node:timers/promises";
|
|
22
23
|
import { createInterface as createInterface$1 } from "node:readline";
|
|
23
24
|
import { createServer as createServer$1 } from "node:net";
|
|
24
25
|
//#region src/run/report-constants.ts
|
|
@@ -1615,10 +1616,17 @@ function warnOnceIfNativeBinaryMissing() {
|
|
|
1615
1616
|
const missing = missingNativeBinaryPackage();
|
|
1616
1617
|
if (missing) warn(missingNativeBinaryMessage(missing));
|
|
1617
1618
|
}
|
|
1619
|
+
/** Whole minutes read best, but the ceiling is set in ms and may be seconds. */
|
|
1620
|
+
function formatDuration$1(ms) {
|
|
1621
|
+
if (ms < 6e4 || ms % 6e4 !== 0) return `${Math.round(ms / 1e3)}s`;
|
|
1622
|
+
const minutes = ms / 6e4;
|
|
1623
|
+
return `${minutes} minute${minutes === 1 ? "" : "s"}`;
|
|
1624
|
+
}
|
|
1618
1625
|
async function invokeClaudeStreaming(options, onEvent) {
|
|
1619
|
-
const { prompt, systemPrompt, allowedTools, disableBuiltinTools = false, disableThinking = false, mcpServers, maxTurns, env, model, cwd, onAbAction, onAbActionFailed, silenceBashLog = false, envScrubMap = [], relaxAbConstraints = false } = options;
|
|
1626
|
+
const { prompt, systemPrompt, allowedTools, disableBuiltinTools = false, disableThinking = false, mcpServers, maxTurns, timeoutMs, env, model, cwd, onAbAction, onAbActionFailed, silenceBashLog = false, envScrubMap = [], relaxAbConstraints = false } = options;
|
|
1620
1627
|
const resolvedModel = resolveModel(model);
|
|
1621
1628
|
const mergedEnv = buildInvocationEnv(env);
|
|
1629
|
+
const abortController = new AbortController();
|
|
1622
1630
|
let lastAbToolUseId = null;
|
|
1623
1631
|
const claimAbToolUse = (toolUseId) => {
|
|
1624
1632
|
if (toolUseId !== lastAbToolUseId) return false;
|
|
@@ -1631,6 +1639,7 @@ async function invokeClaudeStreaming(options, onEvent) {
|
|
|
1631
1639
|
allowedTools: allowedTools ?? ["Bash(*)"],
|
|
1632
1640
|
permissionMode: "bypassPermissions",
|
|
1633
1641
|
allowDangerouslySkipPermissions: true,
|
|
1642
|
+
abortController,
|
|
1634
1643
|
...resolvedModel ? { model: resolvedModel } : {},
|
|
1635
1644
|
...cwd ? { cwd } : {},
|
|
1636
1645
|
...mergedEnv ? { env: mergedEnv } : {},
|
|
@@ -1695,7 +1704,10 @@ async function invokeClaudeStreaming(options, onEvent) {
|
|
|
1695
1704
|
} : void 0
|
|
1696
1705
|
};
|
|
1697
1706
|
warnOnceIfNativeBinaryMissing();
|
|
1707
|
+
const capTimer = timeoutMs === void 0 ? null : setTimeout(() => abortController.abort(), timeoutMs);
|
|
1708
|
+
capTimer?.unref?.();
|
|
1698
1709
|
let result = "";
|
|
1710
|
+
let answered = false;
|
|
1699
1711
|
let isError = false;
|
|
1700
1712
|
let errorDetail = null;
|
|
1701
1713
|
let cost = {
|
|
@@ -1720,6 +1732,7 @@ async function invokeClaudeStreaming(options, onEvent) {
|
|
|
1720
1732
|
}
|
|
1721
1733
|
}
|
|
1722
1734
|
if (msg.type === "result") {
|
|
1735
|
+
answered = true;
|
|
1723
1736
|
isError = msg.is_error ?? false;
|
|
1724
1737
|
if (msg.subtype === "success") result = msg.result;
|
|
1725
1738
|
else {
|
|
@@ -1733,6 +1746,13 @@ async function invokeClaudeStreaming(options, onEvent) {
|
|
|
1733
1746
|
isError = true;
|
|
1734
1747
|
errorDetail = err instanceof Error ? err.message : String(err);
|
|
1735
1748
|
if (!result) result = errorDetail;
|
|
1749
|
+
} finally {
|
|
1750
|
+
if (capTimer) clearTimeout(capTimer);
|
|
1751
|
+
}
|
|
1752
|
+
if (abortController.signal.aborted && timeoutMs !== void 0 && !answered) {
|
|
1753
|
+
isError = true;
|
|
1754
|
+
errorDetail = `stopped after ${formatDuration$1(timeoutMs)} (host time limit)`;
|
|
1755
|
+
result = errorDetail;
|
|
1736
1756
|
}
|
|
1737
1757
|
tallyInvocation(cost);
|
|
1738
1758
|
return {
|
|
@@ -4130,10 +4150,11 @@ const execFileP = promisify(execFile);
|
|
|
4130
4150
|
* `.ccqa/`. Changes outside `cwd` are kept under their repo-root path and
|
|
4131
4151
|
* flagged `outsideCwd` (see `ChangedFile`) rather than dropped.
|
|
4132
4152
|
*
|
|
4133
|
-
* `detectRenames` defaults on.
|
|
4134
|
-
* `
|
|
4135
|
-
* only the destination, so a file's
|
|
4136
|
-
*
|
|
4153
|
+
* `detectRenames` defaults on. Callers whose diff is held against past facts
|
|
4154
|
+
* — `ccqa hub deploy record` and spec selection — turn it off: with rename
|
|
4155
|
+
* detection, a rename is one entry naming only the destination, so a file's
|
|
4156
|
+
* old path would silently drop out — off, it appears as a delete plus an add
|
|
4157
|
+
* and both paths are kept.
|
|
4137
4158
|
*/
|
|
4138
4159
|
async function getChangedFilesBetween(base, head, cwd, options = {}) {
|
|
4139
4160
|
return diffNameStatus(`${base}..${head}`, cwd, options.detectRenames ?? true);
|
|
@@ -7340,7 +7361,9 @@ async function closeMeasurement(collector, ref, coverageDir) {
|
|
|
7340
7361
|
* Every failure here is otherwise silent and identical to success: a root that
|
|
7341
7362
|
* does not exist, or does not contain the project, sends every relative source
|
|
7342
7363
|
* outside it, and the run reports a smaller file set with no error at all —
|
|
7343
|
-
* the answer this measurement exists to prevent.
|
|
7364
|
+
* the answer this measurement exists to prevent. Spec selection
|
|
7365
|
+
* (src/select/analyze.ts) resolves the same key through this function too, so
|
|
7366
|
+
* both sides of an intersection agree on what the root means.
|
|
7344
7367
|
*/
|
|
7345
7368
|
async function resolveRoot(cwd, declared) {
|
|
7346
7369
|
if (declared === void 0) return void 0;
|
|
@@ -10487,16 +10510,8 @@ async function removeTempStateDir(statePath) {
|
|
|
10487
10510
|
* than throwing; the caller decides whether an un-restored session is fatal.
|
|
10488
10511
|
*/
|
|
10489
10512
|
function loadStateIntoSession(sessionName, statePath) {
|
|
10490
|
-
const boot =
|
|
10491
|
-
|
|
10492
|
-
sessionName,
|
|
10493
|
-
"open",
|
|
10494
|
-
"about:blank"
|
|
10495
|
-
]);
|
|
10496
|
-
if (boot.status !== 0) return {
|
|
10497
|
-
ok: false,
|
|
10498
|
-
error: (boot.stderr || boot.stdout || `open exited ${boot.status}`).trim()
|
|
10499
|
-
};
|
|
10513
|
+
const boot = bootSession(sessionName);
|
|
10514
|
+
if (!boot.ok) return boot;
|
|
10500
10515
|
const load = spawnAB([
|
|
10501
10516
|
"--session",
|
|
10502
10517
|
sessionName,
|
|
@@ -10506,7 +10521,26 @@ function loadStateIntoSession(sessionName, statePath) {
|
|
|
10506
10521
|
]);
|
|
10507
10522
|
if (load.status !== 0) return {
|
|
10508
10523
|
ok: false,
|
|
10509
|
-
error: (load.stderr || load.stdout || `state load exited ${load.status}`).trim()
|
|
10524
|
+
error: (load.stderr || load.stdout || `state load exited ${load.status}`).trim(),
|
|
10525
|
+
wedged: load.wedged
|
|
10526
|
+
};
|
|
10527
|
+
return { ok: true };
|
|
10528
|
+
}
|
|
10529
|
+
/**
|
|
10530
|
+
* Bring up the session's daemon and browser without navigating, so whatever
|
|
10531
|
+
* the caller attaches next lands on the session rather than racing a page load.
|
|
10532
|
+
*/
|
|
10533
|
+
function bootSession(sessionName) {
|
|
10534
|
+
const boot = spawnAB([
|
|
10535
|
+
"--session",
|
|
10536
|
+
sessionName,
|
|
10537
|
+
"open",
|
|
10538
|
+
"about:blank"
|
|
10539
|
+
]);
|
|
10540
|
+
if (boot.status !== 0) return {
|
|
10541
|
+
ok: false,
|
|
10542
|
+
error: (boot.stderr || boot.stdout || `open exited ${boot.status}`).trim(),
|
|
10543
|
+
wedged: boot.wedged
|
|
10510
10544
|
};
|
|
10511
10545
|
return { ok: true };
|
|
10512
10546
|
}
|
|
@@ -10633,8 +10667,7 @@ function verifySessionRestores(statePath, verifyUrl) {
|
|
|
10633
10667
|
* false "unhealthy" would re-inject the saved state and wipe auth the spec
|
|
10634
10668
|
* acquired live — breaking a spec that was fine. Missing a same-origin-ish
|
|
10635
10669
|
* sign-in wall just leaves that step failing as before (no regression), so the
|
|
10636
|
-
* asymmetry favours only firing on a provably dead daemon.
|
|
10637
|
-
* required (it's the re-anchor target for recovery) but no longer compared here.
|
|
10670
|
+
* asymmetry favours only firing on a provably dead daemon.
|
|
10638
10671
|
*/
|
|
10639
10672
|
function checkLiveSessionHealth(sessionName) {
|
|
10640
10673
|
const probe = spawnAB([
|
|
@@ -10645,30 +10678,30 @@ function checkLiveSessionHealth(sessionName) {
|
|
|
10645
10678
|
]);
|
|
10646
10679
|
if (probe.status !== 0) return {
|
|
10647
10680
|
healthy: false,
|
|
10681
|
+
kind: probe.wedged === true ? "unresponsive" : "errored",
|
|
10648
10682
|
reason: (probe.stderr || probe.stdout || `probe exited ${probe.status}`).trim()
|
|
10649
10683
|
};
|
|
10650
10684
|
const href = unwrapEvalString(probe.stdout);
|
|
10651
10685
|
if (!href || href === "about:blank" || href.startsWith("chrome://") || href.startsWith("chrome-error://")) return {
|
|
10652
10686
|
healthy: false,
|
|
10687
|
+
kind: "blank",
|
|
10653
10688
|
reason: `blank/absent page (${href || "empty"})`
|
|
10654
10689
|
};
|
|
10655
10690
|
return { healthy: true };
|
|
10656
10691
|
}
|
|
10657
10692
|
/**
|
|
10658
|
-
*
|
|
10659
|
-
*
|
|
10660
|
-
*
|
|
10661
|
-
*
|
|
10662
|
-
*
|
|
10663
|
-
*
|
|
10664
|
-
*
|
|
10665
|
-
* injection result; the trailing `open` is best-effort (a failed nav still
|
|
10666
|
-
* leaves the state attached for the model's own next navigation).
|
|
10693
|
+
* Put a live session back on its feet after {@link checkLiveSessionHealth}
|
|
10694
|
+
* found it broken: boot its browser, re-attach the saved state if it has one,
|
|
10695
|
+
* and anchor on `verifyUrl` so a retrying model resumes from a signed-in page.
|
|
10696
|
+
*
|
|
10697
|
+
* An unresponsive daemon must already have been killed by the caller — every
|
|
10698
|
+
* command below goes through the socket it is ignoring. The trailing `open` is
|
|
10699
|
+
* best-effort; a failed nav still leaves a usable session behind.
|
|
10667
10700
|
*/
|
|
10668
10701
|
function recoverLiveSession(sessionName, statePath, verifyUrl) {
|
|
10669
|
-
const
|
|
10670
|
-
if (!
|
|
10671
|
-
spawnAB([
|
|
10702
|
+
const restored = statePath ? loadStateIntoSession(sessionName, statePath) : bootSession(sessionName);
|
|
10703
|
+
if (!restored.ok) return restored;
|
|
10704
|
+
if (verifyUrl) spawnAB([
|
|
10672
10705
|
"--session",
|
|
10673
10706
|
sessionName,
|
|
10674
10707
|
"open",
|
|
@@ -10835,159 +10868,170 @@ function capDeployPaths(paths) {
|
|
|
10835
10868
|
return paths.slice(0, MAX_SENT_CHANGED_PATHS);
|
|
10836
10869
|
}
|
|
10837
10870
|
//#endregion
|
|
10838
|
-
//#region src/
|
|
10871
|
+
//#region src/config/project-config.ts
|
|
10839
10872
|
/**
|
|
10840
|
-
*
|
|
10873
|
+
* Loader for the consumer project's `.ccqa/config.yaml` — per-target
|
|
10874
|
+
* generation settings (default target, output dirs, reusable code resources,
|
|
10875
|
+
* generation conventions).
|
|
10841
10876
|
*
|
|
10842
|
-
*
|
|
10843
|
-
*
|
|
10844
|
-
*
|
|
10845
|
-
* a claim that has to be earned, and `unknown` is always available instead.
|
|
10877
|
+
* This module only validates and holds the config. `path` / `guides` /
|
|
10878
|
+
* `examples` entries may be glob patterns; they are kept verbatim here and
|
|
10879
|
+
* expanded by the generation engine, which owns size limits and warnings.
|
|
10846
10880
|
*/
|
|
10847
|
-
function buildSelectSystemPrompt() {
|
|
10848
|
-
return `You decide which end-to-end test specs have to be re-run after a set of source changes.
|
|
10849
|
-
|
|
10850
|
-
You are given the files that changed between two commits, and an inventory of every test spec: what each one does, step by step. Return one verdict per spec.
|
|
10851
|
-
|
|
10852
|
-
## The three verdicts
|
|
10853
|
-
|
|
10854
|
-
- **needed** — at least one changed file plausibly affects what this spec verifies. Name the file(s) in \`touchedBy\`.
|
|
10855
|
-
- **notNeeded** — you have accounted for the changed files and none of them reach what this spec does.
|
|
10856
|
-
- **unknown** — you cannot tell.
|
|
10857
|
-
|
|
10858
|
-
## Why \`notNeeded\` is the only answer that can hurt
|
|
10859
|
-
|
|
10860
|
-
Re-running a spec that did not need it costs a few minutes of CI. NOT re-running a spec that needed it lets a regression reach users with the suite still green — the failure mode this tool exists to prevent.
|
|
10861
|
-
|
|
10862
|
-
So the two answers are not symmetric:
|
|
10863
|
-
|
|
10864
|
-
- \`needed\` and \`unknown\` are both safe. The caller runs them.
|
|
10865
|
-
- \`notNeeded\` is a **positive claim**. Make it only when you have actually looked at what the spec exercises and at what changed, and can say the two do not meet.
|
|
10866
|
-
|
|
10867
|
-
When you cannot make that claim — a file whose purpose you cannot infer, a change whose blast radius you cannot bound, a spec whose steps are missing or unclear — answer \`unknown\`. That is a correct answer, not a failure. Guessing \`notNeeded\` is the one thing that causes damage.
|
|
10868
|
-
|
|
10869
|
-
**Do not over-correct.** Marking every spec \`needed\` throws away the entire point: the caller ends up running the full suite on every change. When a change is confined to an area that a spec demonstrably never touches, say \`notNeeded\` and say why. Both a reflexive "run everything" and a careless "skip it" are wrong; judge each spec on the evidence.
|
|
10870
|
-
|
|
10871
|
-
## How to judge
|
|
10872
|
-
|
|
10873
|
-
1. Read the changed paths. Most are self-describing — a path naming a screen, a component, a handler, or a route tells you what it belongs to.
|
|
10874
|
-
2. For a path whose purpose is not obvious from its name, use \`Read\` or \`Grep\` to find out what it does before judging. Prefer this over guessing. Stay focused: this is a routing decision, not a code review.
|
|
10875
|
-
3. Match against what each spec actually does — the screens its steps open, the controls they drive, the strings they assert on.
|
|
10876
|
-
4. Remember indirect reach: shared layout, navigation, authentication, permission checks, and data-access code are touched by specs that never mention them. A change to a sign-in path can break every spec that has to sign in first.
|
|
10877
|
-
|
|
10878
|
-
## What does not affect any spec
|
|
10879
|
-
|
|
10880
|
-
Treat these as irrelevant unless something specific says otherwise: documentation and Markdown, changes to test files of the product's own unit-test suite, lockfile-only churn, formatting-only changes, and code paths that only run in a build or tooling context.
|
|
10881
|
-
|
|
10882
|
-
## Output (STRICT)
|
|
10883
|
-
|
|
10884
|
-
Output ONE fenced \`\`\`json block, and nothing else outside it.
|
|
10885
|
-
|
|
10886
|
-
\`\`\`json
|
|
10887
|
-
{
|
|
10888
|
-
"specs": [
|
|
10889
|
-
{
|
|
10890
|
-
"spec": "<feature>/<spec>",
|
|
10891
|
-
"verdict": "needed" | "notNeeded" | "unknown",
|
|
10892
|
-
"reason": "<one sentence>",
|
|
10893
|
-
"touchedBy": ["<changed path>"]
|
|
10894
|
-
}
|
|
10895
|
-
]
|
|
10896
|
-
}
|
|
10897
|
-
\`\`\`
|
|
10898
|
-
|
|
10899
|
-
Rules for the output:
|
|
10900
|
-
|
|
10901
|
-
- Include **every** spec from the inventory, exactly once, using the key as it was given to you.
|
|
10902
|
-
- \`touchedBy\` is required for \`needed\` and must name paths from the changed-file list. Omit it otherwise.
|
|
10903
|
-
- \`reason\` for \`notNeeded\` must say what you checked, not just "unrelated".
|
|
10904
|
-
- \`reason\` for \`unknown\` must name what you could not determine.
|
|
10905
|
-
`;
|
|
10906
|
-
}
|
|
10907
|
-
/** Cap on how many changed paths are spelled out before the list is summarised. */
|
|
10908
|
-
const MAX_LISTED_PATHS = 400;
|
|
10909
|
-
function buildSelectPrompt(input) {
|
|
10910
|
-
return `## Changed files (${input.base} → ${input.head})
|
|
10911
|
-
|
|
10912
|
-
${formatChangedFiles(input.changed)}
|
|
10913
|
-
|
|
10914
|
-
## Test spec inventory
|
|
10915
|
-
|
|
10916
|
-
${input.specs.map(formatSpec).join("\n\n")}
|
|
10917
|
-
|
|
10918
|
-
## Task
|
|
10919
|
-
|
|
10920
|
-
Return one verdict for each of the ${input.specs.length} specs above. Clear a spec only when you can account for the changes; otherwise answer \`needed\` or \`unknown\`.
|
|
10921
|
-
`;
|
|
10922
|
-
}
|
|
10923
|
-
function formatChangedFiles(changed) {
|
|
10924
|
-
if (changed.length === 0) return "(none)";
|
|
10925
|
-
const listed = changed.slice(0, MAX_LISTED_PATHS);
|
|
10926
|
-
const lines = listed.map((f) => `- ${f.status.padEnd(8)} ${f.path}${f.outsideCwd ? " (outside the tested package)" : ""}`);
|
|
10927
|
-
if (changed.length > listed.length) lines.push(`- ... and ${changed.length - listed.length} more paths, not listed. You have NOT seen the full change set: do not answer \`notNeeded\` for a spec unless the listed paths alone rule it out.`);
|
|
10928
|
-
return lines.join("\n");
|
|
10929
|
-
}
|
|
10930
|
-
function formatSpec(spec) {
|
|
10931
|
-
const steps = spec.steps.map((s, i) => ` ${i + 1}. ${s}`).join("\n");
|
|
10932
|
-
return `### ${specKey(spec)}\n${spec.title}\n${steps}`;
|
|
10933
|
-
}
|
|
10934
|
-
//#endregion
|
|
10935
|
-
//#region src/select/types.ts
|
|
10936
10881
|
/**
|
|
10937
|
-
*
|
|
10938
|
-
*
|
|
10939
|
-
* `
|
|
10882
|
+
* An existing code asset the generated tests should reuse (import), in one of
|
|
10883
|
+
* two forms — exactly one of:
|
|
10884
|
+
* - `path`: code inside the consumer repo (literal path or glob pattern);
|
|
10885
|
+
* - `package`: an installed npm package (imported by name).
|
|
10886
|
+
* `description` tells the generator what the asset contains.
|
|
10940
10887
|
*/
|
|
10941
|
-
const
|
|
10942
|
-
|
|
10943
|
-
|
|
10944
|
-
|
|
10945
|
-
|
|
10946
|
-
|
|
10947
|
-
|
|
10948
|
-
|
|
10949
|
-
|
|
10950
|
-
|
|
10951
|
-
|
|
10952
|
-
|
|
10953
|
-
|
|
10954
|
-
|
|
10955
|
-
|
|
10956
|
-
|
|
10957
|
-
|
|
10958
|
-
|
|
10959
|
-
}
|
|
10888
|
+
const ResourceRefSchema = z.union([z.object({
|
|
10889
|
+
path: z.string().min(1),
|
|
10890
|
+
description: z.string().optional()
|
|
10891
|
+
}).strict(), z.object({
|
|
10892
|
+
package: z.string().min(1),
|
|
10893
|
+
description: z.string().optional()
|
|
10894
|
+
}).strict()], { error: "a resource must have exactly one of `path` (code in this repo) or `package` (installed npm package), plus an optional `description`" });
|
|
10895
|
+
/**
|
|
10896
|
+
* How generated code should be written, as guide inputs to the prompt (never
|
|
10897
|
+
* imported as code): `guides` are convention documents, `examples` are
|
|
10898
|
+
* existing tests whose style to imitate. Entries may be glob patterns.
|
|
10899
|
+
*/
|
|
10900
|
+
const ConventionsSchema = z.object({
|
|
10901
|
+
guides: z.array(z.string().min(1)).default([]),
|
|
10902
|
+
examples: z.array(z.string().min(1)).default([])
|
|
10903
|
+
}).strict();
|
|
10904
|
+
/**
|
|
10905
|
+
* Per-target settings. `outDir` (where generated tests are written) and
|
|
10906
|
+
* `runCommand` (how to execute them; `{files}` expands to the generated
|
|
10907
|
+
* paths, `{artifactsDir}` to the spec's report artifacts dir — see
|
|
10908
|
+
* src/targets/run-artifacts.ts) are optional at this layer because not every
|
|
10909
|
+
* target needs them — e.g. agent-browser stores its output in the spec
|
|
10910
|
+
* directory. A target that requires either must validate its presence itself.
|
|
10911
|
+
*/
|
|
10912
|
+
const TargetConfigSchema = z.object({
|
|
10913
|
+
outDir: z.string().min(1).optional(),
|
|
10914
|
+
runCommand: z.string().min(1).optional(),
|
|
10915
|
+
resources: z.array(ResourceRefSchema).default([]),
|
|
10916
|
+
conventions: ConventionsSchema.default({
|
|
10917
|
+
guides: [],
|
|
10918
|
+
examples: []
|
|
10919
|
+
})
|
|
10920
|
+
}).strict();
|
|
10960
10921
|
/**
|
|
10961
|
-
*
|
|
10962
|
-
*
|
|
10922
|
+
* Specs that must not run at the same time, grouped by the thing they share.
|
|
10923
|
+
*
|
|
10924
|
+
* The key names the shared thing (a chat channel, a seeded account, a tenant);
|
|
10925
|
+
* the list names the specs that write to it. `ccqa run` never runs two members
|
|
10926
|
+
* of one group concurrently, and specs sharing no group still run in parallel.
|
|
10963
10927
|
*
|
|
10964
|
-
*
|
|
10965
|
-
*
|
|
10966
|
-
*
|
|
10967
|
-
* spec's answer in the same reply. `touchedBy` keeps the same tolerance one
|
|
10968
|
-
* level down — a non-string element is dropped, not fatal to the entry.
|
|
10928
|
+
* Kept here rather than on each spec so there is one place to read the whole
|
|
10929
|
+
* picture, and so a mistyped member is a spec key that does not resolve —
|
|
10930
|
+
* caught — rather than a resource name that silently matches nothing.
|
|
10969
10931
|
*/
|
|
10970
|
-
const
|
|
10971
|
-
|
|
10972
|
-
|
|
10973
|
-
|
|
10974
|
-
|
|
10975
|
-
|
|
10976
|
-
|
|
10977
|
-
|
|
10932
|
+
const SerialGroupsSchema = z.record(z.string().regex(/^[a-z0-9][a-z0-9._-]*$/i, "serial group name must be a slug (letters, digits, '.', '_', '-')"), z.array(z.string().min(1)).min(1));
|
|
10933
|
+
/**
|
|
10934
|
+
* Which specs act as which external identity, for the flows whose requests
|
|
10935
|
+
* cannot carry a spec id at all.
|
|
10936
|
+
*
|
|
10937
|
+
* A chat platform's webhook is sent by the platform, not the browser, so no
|
|
10938
|
+
* cookie rides along and everything the flow reaches would be unattributed.
|
|
10939
|
+
* What the request does carry is who caused it, and if only one spec is allowed
|
|
10940
|
+
* to act as that identity at a time, "who" plus "when" is enough.
|
|
10941
|
+
*
|
|
10942
|
+
* ```yaml
|
|
10943
|
+
* coverage:
|
|
10944
|
+
* actors:
|
|
10945
|
+
* slack: # the preset's tag prefix
|
|
10946
|
+
* ${TEST_USER_ID}: [chat/create-item, chat/resolve-item]
|
|
10947
|
+
* ```
|
|
10948
|
+
*
|
|
10949
|
+
* The provider name is the prefix the matching preset stamps, and the key is an
|
|
10950
|
+
* identity expression the run's variables resolve. Only the unexpanded text is
|
|
10951
|
+
* ever displayed or used as a lock key, so the identity itself stays out of
|
|
10952
|
+
* reports and the hub.
|
|
10953
|
+
*/
|
|
10954
|
+
const CoverageActorsSchema = z.record(z.string().regex(/^[a-z0-9][a-z0-9._-]*$/i, "actor provider must be a slug (letters, digits, '.', '_', '-')"), z.record(z.string().min(1), z.array(z.string().min(1)).min(1)));
|
|
10955
|
+
/**
|
|
10956
|
+
* Settings for `ccqa run --coverage`, which measures what each spec actually
|
|
10957
|
+
* reached in the application under test.
|
|
10958
|
+
*/
|
|
10959
|
+
const CoverageConfigSchema = z.object({
|
|
10960
|
+
instrumentedOrigins: z.array(z.string().min(1)).min(1),
|
|
10961
|
+
sink: z.string().min(1).default("http://127.0.0.1:4757"),
|
|
10962
|
+
projectRoot: z.string().min(1).optional(),
|
|
10963
|
+
include: z.array(z.string().min(1)).optional(),
|
|
10964
|
+
actors: CoverageActorsSchema.default({})
|
|
10965
|
+
}).strict();
|
|
10966
|
+
/**
|
|
10967
|
+
* Top-level `.ccqa/config.yaml` schema. `defaultTarget` is used by specs
|
|
10968
|
+
* with no `target:` of their own. Both defaults make a missing config file
|
|
10969
|
+
* equivalent to "agent-browser only, no extra settings".
|
|
10970
|
+
*/
|
|
10971
|
+
const ProjectConfigSchema = z.object({
|
|
10972
|
+
defaultTarget: TargetIdSchema.default(AGENT_BROWSER_TARGET),
|
|
10973
|
+
targets: z.record(TargetIdSchema, TargetConfigSchema).default({}),
|
|
10974
|
+
serialGroups: SerialGroupsSchema.default({}),
|
|
10975
|
+
coverage: CoverageConfigSchema.optional()
|
|
10976
|
+
}).strict();
|
|
10977
|
+
/** Config file location, relative to the project root (`--cwd`). */
|
|
10978
|
+
const PROJECT_CONFIG_PATH = ".ccqa/config.yaml";
|
|
10979
|
+
/**
|
|
10980
|
+
* Load `<cwd>/.ccqa/config.yaml`. A missing file yields the defaults (an
|
|
10981
|
+
* empty file too); a present but broken file is an error — never silently
|
|
10982
|
+
* fall back when the user wrote a config.
|
|
10983
|
+
*/
|
|
10984
|
+
async function loadProjectConfig(cwd) {
|
|
10985
|
+
let content;
|
|
10986
|
+
try {
|
|
10987
|
+
content = await readFile(join(cwd, PROJECT_CONFIG_PATH), "utf8");
|
|
10988
|
+
} catch (e) {
|
|
10989
|
+
if (e.code === "ENOENT") return ProjectConfigSchema.parse({});
|
|
10990
|
+
throw e;
|
|
10991
|
+
}
|
|
10992
|
+
return parseProjectConfig(content);
|
|
10993
|
+
}
|
|
10994
|
+
/** Parse config YAML. Schema rejections are rewritten with actionable messages. */
|
|
10995
|
+
function parseProjectConfig(content, source = PROJECT_CONFIG_PATH) {
|
|
10996
|
+
let raw;
|
|
10997
|
+
try {
|
|
10998
|
+
raw = parse(content);
|
|
10999
|
+
} catch (e) {
|
|
11000
|
+
throw new Error(`Failed to parse YAML (${source}): ${e.message}`);
|
|
11001
|
+
}
|
|
11002
|
+
try {
|
|
11003
|
+
return ProjectConfigSchema.parse(raw ?? {});
|
|
11004
|
+
} catch (e) {
|
|
11005
|
+
throw enrichZodError(e, source);
|
|
11006
|
+
}
|
|
11007
|
+
}
|
|
11008
|
+
/** Flatten a ZodError into one `Invalid <source>:` message, path per line. */
|
|
11009
|
+
function enrichZodError(error, source) {
|
|
11010
|
+
if (!(error instanceof ZodError)) return error;
|
|
11011
|
+
const lines = [`Invalid ${source}:`];
|
|
11012
|
+
for (const issue of error.issues) {
|
|
11013
|
+
const path = issue.path.join(".") || "(root)";
|
|
11014
|
+
const message = issue.code === "invalid_key" && issue.issues[0] ? issue.issues[0].message : issue.message;
|
|
11015
|
+
lines.push(` - ${path}: ${message}`);
|
|
11016
|
+
}
|
|
11017
|
+
return new Error(lines.join("\n"));
|
|
11018
|
+
}
|
|
10978
11019
|
//#endregion
|
|
10979
11020
|
//#region src/select/analyze.ts
|
|
10980
11021
|
/**
|
|
10981
11022
|
* Decide which specs a change set reaches.
|
|
10982
11023
|
*
|
|
10983
|
-
* Two passes, in this order and for this reason: what
|
|
10984
|
-
*
|
|
10985
|
-
*
|
|
10986
|
-
* spec must re-run —
|
|
10987
|
-
*
|
|
11024
|
+
* Two passes, in this order and for this reason: what a change to ccqa's own
|
|
11025
|
+
* tree settles is settled first, and only the remainder is held against
|
|
11026
|
+
* measured reach. A change to a spec's own file, or to a block it includes,
|
|
11027
|
+
* means that spec must re-run — reach cannot see the test's own definition,
|
|
11028
|
+
* so no measurement is consulted for it. Everything else intersects the diff
|
|
11029
|
+
* with the files the spec's last measured run actually reached (ADR-0024);
|
|
11030
|
+
* a spec with no measurement stays `unknown`, because an unmeasured edge is
|
|
11031
|
+
* not an unreached one.
|
|
10988
11032
|
*/
|
|
10989
11033
|
async function selectSpecs(input) {
|
|
10990
|
-
const { changed, specs, cwd, base, head,
|
|
11034
|
+
const { changed, specs, cwd, base, head, edges } = input;
|
|
10991
11035
|
const { productChanges, mechanicallyNeeded } = partitionChanges(changed, specs);
|
|
10992
11036
|
const byInventoryKey = new Map(specs.map((s) => [specKey(s), s]));
|
|
10993
11037
|
const byKey = /* @__PURE__ */ new Map();
|
|
@@ -11011,13 +11055,11 @@ async function selectSpecs(input) {
|
|
|
11011
11055
|
source: "mechanical",
|
|
11012
11056
|
reason: "no file outside .ccqa/ changed in this range"
|
|
11013
11057
|
});
|
|
11014
|
-
else if (undecided.length > 0) for (const selection of await
|
|
11058
|
+
else if (undecided.length > 0) for (const selection of await judgeWithCoverage({
|
|
11059
|
+
pending: undecided,
|
|
11015
11060
|
productChanges,
|
|
11016
|
-
undecided,
|
|
11017
11061
|
cwd,
|
|
11018
|
-
|
|
11019
|
-
head,
|
|
11020
|
-
model
|
|
11062
|
+
edges
|
|
11021
11063
|
})) byKey.set(specKey(selection), selection);
|
|
11022
11064
|
return {
|
|
11023
11065
|
base,
|
|
@@ -11027,12 +11069,12 @@ async function selectSpecs(input) {
|
|
|
11027
11069
|
};
|
|
11028
11070
|
}
|
|
11029
11071
|
/**
|
|
11030
|
-
* Split the diff into the part
|
|
11031
|
-
* decides itself.
|
|
11072
|
+
* Split the diff into the part measured reach has to answer for and the part
|
|
11073
|
+
* that decides itself.
|
|
11032
11074
|
*
|
|
11033
11075
|
* `.ccqa/` paths are ccqa's own: a spec directory names the spec it belongs
|
|
11034
11076
|
* to, a block names the specs that include it. Product paths carry no such
|
|
11035
|
-
* mapping —
|
|
11077
|
+
* mapping — those are what the coverage intersection exists to answer.
|
|
11036
11078
|
*/
|
|
11037
11079
|
function partitionChanges(changed, specs) {
|
|
11038
11080
|
const productChanges = [];
|
|
@@ -11079,118 +11121,214 @@ function isCcqaPath(path) {
|
|
|
11079
11121
|
return /(?:^|\/)\.ccqa\//.test(path);
|
|
11080
11122
|
}
|
|
11081
11123
|
/**
|
|
11082
|
-
*
|
|
11083
|
-
*
|
|
11084
|
-
*
|
|
11085
|
-
*
|
|
11086
|
-
*
|
|
11087
|
-
|
|
11088
|
-
|
|
11124
|
+
* Hold each undecided spec's last measured reach against the diff.
|
|
11125
|
+
*
|
|
11126
|
+
* Three outcomes, and only the middle one is a positive claim: no
|
|
11127
|
+
* measurement means `unknown` (an unmeasured edge is not an unreached one —
|
|
11128
|
+
* the absence of evidence runs the spec); a non-empty intersection means
|
|
11129
|
+
* `needed`, with the intersecting paths as the reason; an empty one means
|
|
11130
|
+
* `notNeeded` — the measurement accounts for everything the spec reached,
|
|
11131
|
+
* and the diff missed all of it. Changes outside the measured root fall out
|
|
11132
|
+
* of the comparison entirely: the root is the declared boundary of what
|
|
11133
|
+
* measurement governs, so what lies beyond it clears specs quietly — one
|
|
11134
|
+
* warning names the dropped paths, because a root configured too narrow
|
|
11135
|
+
* looks exactly like this and hides real reach (see docs/coverage.md).
|
|
11136
|
+
*/
|
|
11137
|
+
async function judgeWithCoverage(input) {
|
|
11138
|
+
const { pending, productChanges, cwd, edges } = input;
|
|
11139
|
+
const noMeasurement = "no measurement to consult: the hub holds no measured reach for this spec";
|
|
11140
|
+
if (edges.size === 0) return pending.map((s) => unknownSelection(s, noMeasurement));
|
|
11141
|
+
const measuredChanges = rerootChangesForCoverage(productChanges, await resolveCoverageRoots(productChanges, cwd));
|
|
11142
|
+
const dropped = productChanges.length - measuredChanges.length;
|
|
11143
|
+
if (dropped > 0) warn(`select-specs: ${dropped} of ${productChanges.length} changed files fall outside coverage.projectRoot and cannot be compared against measured reach`);
|
|
11144
|
+
return pending.map((spec) => {
|
|
11145
|
+
const edge = edges.get(specKey(spec));
|
|
11146
|
+
if (!edge) return unknownSelection(spec, noMeasurement);
|
|
11147
|
+
const touchedBy = measuredChanges.filter((c) => edge.files.has(c.measured)).map((c) => c.original);
|
|
11148
|
+
if (touchedBy.length > 0) return {
|
|
11149
|
+
featureName: spec.featureName,
|
|
11150
|
+
specName: spec.specName,
|
|
11151
|
+
verdict: "needed",
|
|
11152
|
+
source: "coverage",
|
|
11153
|
+
reason: "the change touches files this spec's last measured run reached",
|
|
11154
|
+
touchedBy
|
|
11155
|
+
};
|
|
11156
|
+
return {
|
|
11157
|
+
featureName: spec.featureName,
|
|
11158
|
+
specName: spec.specName,
|
|
11159
|
+
verdict: "notNeeded",
|
|
11160
|
+
source: "coverage",
|
|
11161
|
+
reason: "the spec's last measured run reached none of the changed files"
|
|
11162
|
+
};
|
|
11163
|
+
});
|
|
11164
|
+
}
|
|
11165
|
+
function unknownSelection(spec, reason) {
|
|
11166
|
+
return {
|
|
11167
|
+
featureName: spec.featureName,
|
|
11168
|
+
specName: spec.specName,
|
|
11169
|
+
verdict: "unknown",
|
|
11170
|
+
source: "coverage",
|
|
11171
|
+
reason
|
|
11172
|
+
};
|
|
11173
|
+
}
|
|
11089
11174
|
/**
|
|
11090
|
-
*
|
|
11091
|
-
*
|
|
11092
|
-
*
|
|
11093
|
-
*
|
|
11094
|
-
*
|
|
11095
|
-
*
|
|
11096
|
-
* degrades to running more than necessary — never to skipping something.
|
|
11175
|
+
* Re-root diff paths to the measurement's own base. The two sides must speak
|
|
11176
|
+
* the same paths or every intersection silently misses (ADR-0024): the diff
|
|
11177
|
+
* is cwd-relative (repo-root relative for `outsideCwd` entries) while
|
|
11178
|
+
* measured files are `coverage.projectRoot`-relative. A file resolving
|
|
11179
|
+
* outside the coverage root is dropped — the measurement drops those files
|
|
11180
|
+
* too, so it could never intersect an edge.
|
|
11097
11181
|
*/
|
|
11098
|
-
|
|
11099
|
-
const
|
|
11100
|
-
|
|
11101
|
-
|
|
11102
|
-
|
|
11103
|
-
const
|
|
11104
|
-
|
|
11105
|
-
|
|
11106
|
-
|
|
11107
|
-
|
|
11108
|
-
head
|
|
11109
|
-
}),
|
|
11110
|
-
systemPrompt: buildSelectSystemPrompt(),
|
|
11111
|
-
allowedTools: [
|
|
11112
|
-
"Read",
|
|
11113
|
-
"Grep",
|
|
11114
|
-
"Glob"
|
|
11115
|
-
],
|
|
11116
|
-
silenceBashLog: true,
|
|
11117
|
-
cwd,
|
|
11118
|
-
...model ? { model } : {}
|
|
11119
|
-
}, (_msg) => {});
|
|
11120
|
-
if (isError) {
|
|
11121
|
-
lastError = "the selection model returned an error";
|
|
11122
|
-
continue;
|
|
11123
|
-
}
|
|
11124
|
-
const json = extractJsonBlock(result);
|
|
11125
|
-
if (!json) {
|
|
11126
|
-
lastError = "the selection model returned no JSON block";
|
|
11127
|
-
continue;
|
|
11128
|
-
}
|
|
11129
|
-
try {
|
|
11130
|
-
parsed = JSON.parse(json);
|
|
11131
|
-
lastError = "";
|
|
11132
|
-
break;
|
|
11133
|
-
} catch (e) {
|
|
11134
|
-
lastError = `the selection model's JSON did not parse: ${e.message}`;
|
|
11135
|
-
}
|
|
11136
|
-
}
|
|
11137
|
-
if (lastError) return abandonSelection(undecided, `${lastError} (${MAX_ATTEMPTS} attempts)`);
|
|
11138
|
-
const changedPaths = new Set(productChanges.map((f) => f.path));
|
|
11139
|
-
const byUndecidedKey = new Map(undecided.map((s) => [specKey(s), s]));
|
|
11140
|
-
const answers = /* @__PURE__ */ new Map();
|
|
11141
|
-
for (const raw of readSpecArray(parsed)) {
|
|
11142
|
-
const spec = byUndecidedKey.get(raw.spec);
|
|
11143
|
-
if (!spec) continue;
|
|
11144
|
-
answers.set(raw.spec, {
|
|
11145
|
-
featureName: spec.featureName,
|
|
11146
|
-
specName: spec.specName,
|
|
11147
|
-
verdict: raw.verdict,
|
|
11148
|
-
source: "model",
|
|
11149
|
-
reason: raw.reason,
|
|
11150
|
-
...raw.verdict === "needed" ? { touchedBy: raw.touchedBy.filter((p) => changedPaths.has(p)) } : {}
|
|
11182
|
+
function rerootChangesForCoverage(changed, roots) {
|
|
11183
|
+
const out = [];
|
|
11184
|
+
for (const file of changed) {
|
|
11185
|
+
const base = file.outsideCwd ? roots.repoRoot : roots.cwd;
|
|
11186
|
+
if (base === null) continue;
|
|
11187
|
+
const measured = relative(roots.coverageRoot, resolve(base, file.path)).replaceAll("\\", "/");
|
|
11188
|
+
if (measured.startsWith("..")) continue;
|
|
11189
|
+
out.push({
|
|
11190
|
+
original: file.path,
|
|
11191
|
+
measured
|
|
11151
11192
|
});
|
|
11152
11193
|
}
|
|
11153
|
-
|
|
11154
|
-
if (missing.length > 0) warn(`select-specs: the model omitted ${missing.length} spec(s); treating them as unknown`);
|
|
11155
|
-
return [...answers.values(), ...allUnknown(missing, "the selection model did not return a verdict for this spec")];
|
|
11194
|
+
return out;
|
|
11156
11195
|
}
|
|
11157
11196
|
/**
|
|
11158
|
-
*
|
|
11159
|
-
*
|
|
11160
|
-
*
|
|
11161
|
-
*
|
|
11197
|
+
* The roots `rerootChangesForCoverage` needs: `coverage.projectRoot` from
|
|
11198
|
+
* `.ccqa/config.yaml` (defaults to cwd), and the git repo root — resolved
|
|
11199
|
+
* only when an `outsideCwd` entry exists to anchor.
|
|
11200
|
+
*
|
|
11201
|
+
* The projectRoot goes through the measurement's own `resolveRoot` — env refs
|
|
11202
|
+
* expanded, the directory verified to exist and contain cwd — and a config it
|
|
11203
|
+
* rejects fails here too. Resolving it any other way would silently re-root
|
|
11204
|
+
* every path somewhere the measurement never stored files under.
|
|
11162
11205
|
*/
|
|
11163
|
-
function
|
|
11164
|
-
const
|
|
11165
|
-
|
|
11166
|
-
|
|
11167
|
-
|
|
11168
|
-
|
|
11169
|
-
|
|
11206
|
+
async function resolveCoverageRoots(changed, cwd) {
|
|
11207
|
+
const coverageRoot = await resolveRoot(cwd, (await loadProjectConfig(cwd)).coverage?.projectRoot) ?? resolve(cwd);
|
|
11208
|
+
let repoRoot = null;
|
|
11209
|
+
if (changed.some((f) => f.outsideCwd)) try {
|
|
11210
|
+
const { stdout } = await execFileP("git", ["rev-parse", "--show-toplevel"], { cwd });
|
|
11211
|
+
repoRoot = stdout.trim();
|
|
11212
|
+
} catch {}
|
|
11213
|
+
return {
|
|
11214
|
+
cwd: resolve(cwd),
|
|
11215
|
+
repoRoot,
|
|
11216
|
+
coverageRoot
|
|
11217
|
+
};
|
|
11218
|
+
}
|
|
11219
|
+
//#endregion
|
|
11220
|
+
//#region src/select/coverage-edges.ts
|
|
11221
|
+
/**
|
|
11222
|
+
* How many recent hub runs are probed for report-row coverage. Bounded
|
|
11223
|
+
* because every probe downloads a whole report.json; past this many runs a
|
|
11224
|
+
* measurement is old enough that treating it as absent — `unknown`, so the
|
|
11225
|
+
* spec runs — is the safer answer anyway. The stream side carries its own
|
|
11226
|
+
* bound: the hub lists at most its newest twenty measured runs.
|
|
11227
|
+
*/
|
|
11228
|
+
const MAX_REPORT_RUNS = 20;
|
|
11229
|
+
/**
|
|
11230
|
+
* How old a measurement may be and still decide a spec. The same fourteen
|
|
11231
|
+
* days the stream store retains events for (`COVERAGE_RETENTION_DAYS`): past
|
|
11232
|
+
* it an edge is too stale to clear a spec with confidence, so it is not
|
|
11233
|
+
* adopted and the spec degrades to `unknown` — which runs.
|
|
11234
|
+
*/
|
|
11235
|
+
const EDGE_MAX_AGE_MS = 336 * 60 * 60 * 1e3;
|
|
11236
|
+
/**
|
|
11237
|
+
* Read every spec's most recent measured reach from the hub.
|
|
11238
|
+
*
|
|
11239
|
+
* Never throws: a hub that cannot be read yields an empty map (warned), which
|
|
11240
|
+
* the selection degrades to `unknown` across the board — the caller runs
|
|
11241
|
+
* those specs, so an unreadable hub costs runs, never a skipped regression.
|
|
11242
|
+
*/
|
|
11243
|
+
async function loadCoverageEdges(input) {
|
|
11244
|
+
const edges = /* @__PURE__ */ new Map();
|
|
11245
|
+
const freshAfter = Date.now() - EDGE_MAX_AGE_MS;
|
|
11246
|
+
const merge = (key, edge) => {
|
|
11247
|
+
if (edge.measuredAt < freshAfter) return;
|
|
11248
|
+
const existing = edges.get(key);
|
|
11249
|
+
if (!existing || edge.measuredAt > existing.measuredAt) edges.set(key, edge);
|
|
11250
|
+
};
|
|
11251
|
+
const results = await Promise.allSettled([collectStreamEdges(input, merge), collectReportEdges(input, merge)]);
|
|
11252
|
+
let skipped = 0;
|
|
11253
|
+
for (const result of results) if (result.status === "rejected") warn(`select-specs: could not read coverage measurements from the hub (${errMessage(result.reason)})`);
|
|
11254
|
+
else skipped += result.value;
|
|
11255
|
+
if (skipped > 0) warn(`select-specs: ${skipped} measured run(s) on the hub could not be read; their reach is treated as absent`);
|
|
11256
|
+
return edges;
|
|
11257
|
+
}
|
|
11258
|
+
/**
|
|
11259
|
+
* Edges from the coverage event stream. The plain read answers for the most
|
|
11260
|
+
* recently measured run and lists every run the stream retains; each older
|
|
11261
|
+
* run is then resolved individually. Returns how many runs could not be read.
|
|
11262
|
+
*/
|
|
11263
|
+
async function collectStreamEdges(input, merge) {
|
|
11264
|
+
const { hub, project } = input;
|
|
11265
|
+
const latest = await hub.getCoverage(project);
|
|
11266
|
+
ingestResolved(latest.resolved, merge);
|
|
11267
|
+
const older = latest.runIds.filter((runId) => runId !== latest.resolved?.runId);
|
|
11268
|
+
return (await Promise.allSettled(older.map(async (runId) => ingestResolved((await hub.getCoverage(project, { runId })).resolved, merge)))).filter((r) => r.status === "rejected").length;
|
|
11269
|
+
}
|
|
11270
|
+
function ingestResolved(resolved, merge) {
|
|
11271
|
+
if (!resolved) return;
|
|
11272
|
+
for (const spec of resolved.specs) {
|
|
11273
|
+
const key = stripRunIdPrefix(spec.specId, resolved.runId);
|
|
11274
|
+
if (key === null) continue;
|
|
11275
|
+
if (spec.files.length === 0) continue;
|
|
11276
|
+
merge(key, {
|
|
11277
|
+
files: new Set(spec.files),
|
|
11278
|
+
measuredAt: resolved.asOf
|
|
11279
|
+
});
|
|
11170
11280
|
}
|
|
11171
|
-
return out;
|
|
11172
11281
|
}
|
|
11173
11282
|
/**
|
|
11174
|
-
*
|
|
11175
|
-
*
|
|
11176
|
-
*
|
|
11177
|
-
* Without the warning the caller reports "N specs could not be decided" and
|
|
11178
|
-
* runs them all — which is the safe outcome, but indistinguishable from a
|
|
11179
|
-
* genuinely ambiguous diff. A wrong model name or an expired credential would
|
|
11180
|
-
* quietly cost a full suite run every time.
|
|
11283
|
+
* A stream specId is `<runId>.<feature>/<spec>` (src/coverage/session.ts).
|
|
11284
|
+
* The runId itself may contain `.`, so the known prefix is stripped by
|
|
11285
|
+
* length, never by splitting on the dot.
|
|
11181
11286
|
*/
|
|
11182
|
-
function
|
|
11183
|
-
|
|
11184
|
-
return allUnknown(specs, reason);
|
|
11287
|
+
function stripRunIdPrefix(specId, runId) {
|
|
11288
|
+
return specId.startsWith(`${runId}.`) ? specId.slice(runId.length + 1) : null;
|
|
11185
11289
|
}
|
|
11186
|
-
|
|
11187
|
-
|
|
11188
|
-
|
|
11189
|
-
|
|
11190
|
-
|
|
11191
|
-
|
|
11192
|
-
|
|
11193
|
-
|
|
11290
|
+
/**
|
|
11291
|
+
* The one slice of report.json this consumer reads. Parsed with its own
|
|
11292
|
+
* narrow schema rather than the full report schema so a report from another
|
|
11293
|
+
* ccqa version still yields its edges as long as this shape holds.
|
|
11294
|
+
*/
|
|
11295
|
+
const ReportCoverageRowsSchema = z.object({ results: z.array(z.object({
|
|
11296
|
+
feature: z.string(),
|
|
11297
|
+
spec: z.string(),
|
|
11298
|
+
coverage: z.object({ files: z.array(z.string()) }).optional()
|
|
11299
|
+
})) });
|
|
11300
|
+
/**
|
|
11301
|
+
* Edges from pushed run reports, newest first. Only `kind: run` runs are
|
|
11302
|
+
* probed — audits and recordings execute no specs, so they carry no reach —
|
|
11303
|
+
* and a still-`running` run is skipped: its rows are still arriving, so its
|
|
11304
|
+
* measurement is not settled. Returns how many reports could not be read.
|
|
11305
|
+
*/
|
|
11306
|
+
async function collectReportEdges(input, merge) {
|
|
11307
|
+
const { hub, project } = input;
|
|
11308
|
+
const eligible = (await hub.listRuns({
|
|
11309
|
+
project,
|
|
11310
|
+
kind: "run",
|
|
11311
|
+
limit: MAX_REPORT_RUNS
|
|
11312
|
+
})).flatMap((run) => {
|
|
11313
|
+
if (run.status === "running") return [];
|
|
11314
|
+
const measuredAt = Date.parse(run.createdAt);
|
|
11315
|
+
if (Number.isNaN(measuredAt)) return [];
|
|
11316
|
+
return [{
|
|
11317
|
+
id: run.id,
|
|
11318
|
+
measuredAt
|
|
11319
|
+
}];
|
|
11320
|
+
});
|
|
11321
|
+
return (await Promise.allSettled(eligible.map(async ({ id, measuredAt }) => {
|
|
11322
|
+
const parsed = ReportCoverageRowsSchema.safeParse(await hub.getReport(id));
|
|
11323
|
+
if (!parsed.success) return;
|
|
11324
|
+
for (const row of parsed.data.results) {
|
|
11325
|
+
if (!row.coverage || row.coverage.files.length === 0) continue;
|
|
11326
|
+
merge(`${row.feature}/${row.spec}`, {
|
|
11327
|
+
files: new Set(row.coverage.files),
|
|
11328
|
+
measuredAt
|
|
11329
|
+
});
|
|
11330
|
+
}
|
|
11331
|
+
}))).filter((r) => r.status === "rejected").length;
|
|
11194
11332
|
}
|
|
11195
11333
|
//#endregion
|
|
11196
11334
|
//#region src/select/inventory.ts
|
|
@@ -11243,6 +11381,68 @@ function describeStep(step) {
|
|
|
11243
11381
|
function oneLine$1(text) {
|
|
11244
11382
|
return text.trim().replace(/\s+/g, " ");
|
|
11245
11383
|
}
|
|
11384
|
+
function emptyDeployLog() {
|
|
11385
|
+
return {
|
|
11386
|
+
nextIndex: 0,
|
|
11387
|
+
entries: []
|
|
11388
|
+
};
|
|
11389
|
+
}
|
|
11390
|
+
/** Append `input` to `current`; the appended entry is always the last of `entries`. */
|
|
11391
|
+
function appendDeploy(current, input) {
|
|
11392
|
+
const log = current ?? emptyDeployLog();
|
|
11393
|
+
const head = log.entries[log.entries.length - 1];
|
|
11394
|
+
const gapBefore = head ? head.sha !== input.previousSha : log.nextIndex > 0;
|
|
11395
|
+
const entries = [...log.entries, {
|
|
11396
|
+
...input,
|
|
11397
|
+
index: log.nextIndex,
|
|
11398
|
+
changedPaths: input.changedPaths === null ? null : input.changedPaths.slice(0, 500),
|
|
11399
|
+
gapBefore
|
|
11400
|
+
}];
|
|
11401
|
+
if (entries.length > 200) {
|
|
11402
|
+
entries.splice(0, entries.length - 200);
|
|
11403
|
+
entries[0] = {
|
|
11404
|
+
...entries[0],
|
|
11405
|
+
gapBefore: true
|
|
11406
|
+
};
|
|
11407
|
+
}
|
|
11408
|
+
return {
|
|
11409
|
+
nextIndex: log.nextIndex + 1,
|
|
11410
|
+
entries
|
|
11411
|
+
};
|
|
11412
|
+
}
|
|
11413
|
+
/**
|
|
11414
|
+
* Fold one deploy's selection into the touch index.
|
|
11415
|
+
*
|
|
11416
|
+
* Only `needed` moves a position a verdict reads. An `unknown` records the
|
|
11417
|
+
* newest undecided position too, but that one is record-only (ADR-0023) —
|
|
11418
|
+
* freshness reads an undecided judgment as "did not reach". A `notNeeded`
|
|
11419
|
+
* writes neither — it is the absence of a marker at this position, which is
|
|
11420
|
+
* exactly what a later baseline comparison reads it as.
|
|
11421
|
+
*
|
|
11422
|
+
* Positions only ever advance. Deploys are folded in log order, so a spec
|
|
11423
|
+
* needed at #7 and cleared at #9 keeps `needed.index: 7`: a baseline at #5
|
|
11424
|
+
* must still see that #7 touched it.
|
|
11425
|
+
*/
|
|
11426
|
+
function foldTouchIndex(current, entry, selection) {
|
|
11427
|
+
const out = { ...current };
|
|
11428
|
+
for (const [key, decision] of Object.entries(selection)) {
|
|
11429
|
+
const previous = out[key] ?? {};
|
|
11430
|
+
if (decision.verdict === "needed") out[key] = {
|
|
11431
|
+
...previous,
|
|
11432
|
+
needed: {
|
|
11433
|
+
index: entry.index,
|
|
11434
|
+
sha: entry.sha,
|
|
11435
|
+
at: entry.at,
|
|
11436
|
+
matchedPaths: (decision.touchedBy ?? []).slice(0, 10)
|
|
11437
|
+
}
|
|
11438
|
+
};
|
|
11439
|
+
else if (decision.verdict === "unknown") out[key] = {
|
|
11440
|
+
...previous,
|
|
11441
|
+
undecidedIndex: entry.index
|
|
11442
|
+
};
|
|
11443
|
+
}
|
|
11444
|
+
return out;
|
|
11445
|
+
}
|
|
11246
11446
|
//#endregion
|
|
11247
11447
|
//#region src/cli/session.ts
|
|
11248
11448
|
const AB = resolveAgentBrowserBin$1();
|
|
@@ -11530,8 +11730,8 @@ const promptRm = new Command("rm").description("Delete a prompt from the hub.").
|
|
|
11530
11730
|
info(`deleted prompt "${name}" from the hub`);
|
|
11531
11731
|
}));
|
|
11532
11732
|
const promptCommand = new Command("prompt").description("Manage prompt assets (per-flow user/agent guidance, triage/audit user guidance, learned calibration prompts) stored on the hub (fetched automatically by `ccqa run` / `ccqa audit` at run time).").addCommand(promptPush).addCommand(promptLs).addCommand(promptRm);
|
|
11533
|
-
const deployRecord = new Command("record").description("Tell the hub what a deploy shipped, so it can answer which specs need a re-run (`ccqa run --only-hub-rerun-needed`). Run this from the deploy job, after the deploy succeeds. The changed paths are computed locally with a two-dot `git diff <previous> <sha>`; a job that has only curl and git can POST the same body directly (see docs/hub.md).").requiredOption("--profile <name>", "Environment this deploy landed in (e.g. 'stg'). Required: dev and stg sit at different commits, so the deploy log is per-profile.").requiredOption("--sha <sha>", "Commit that was deployed.").option("--previous <sha>", "Commit this deploy replaced. Omit it and the hub's current log head is used — the normal case, recording no discontinuity. Pass a sha that differs from the head and the hub records one (gapBefore) in the chain: use this for a first record with a real baseline, or to re-anchor a head that no longer matches reality. With no head and nothing passed, there's nothing to diff against: changedPaths is unset and the spec selection is skipped.").option("--ref <ref>", "Ref that was deployed (branch or tag). Recorded for display only.").option("--no-select-specs", "Record the deploy without deciding which specs it reaches. The entry then becomes a hole in the range — every spec behind it is assumed reached rather than being cleared, and nothing can fill it in later, since the hub has no checkout to diff.
|
|
11534
|
-
await
|
|
11733
|
+
const deployRecord = new Command("record").description("Tell the hub what a deploy shipped, so it can answer which specs need a re-run (`ccqa run --only-hub-rerun-needed`). Run this from the deploy job, after the deploy succeeds. The changed paths are computed locally with a two-dot `git diff <previous> <sha>`; a job that has only curl and git can POST the same body directly (see docs/hub.md).").requiredOption("--profile <name>", "Environment this deploy landed in (e.g. 'stg'). Required: dev and stg sit at different commits, so the deploy log is per-profile.").requiredOption("--sha <sha>", "Commit that was deployed.").option("--previous <sha>", "Commit this deploy replaced. Omit it and the hub's current log head is used — the normal case, recording no discontinuity. Pass a sha that differs from the head and the hub records one (gapBefore) in the chain: use this for a first record with a real baseline, or to re-anchor a head that no longer matches reality. With no head and nothing passed, there's nothing to diff against: changedPaths is unset and the spec selection is skipped.").option("--ref <ref>", "Ref that was deployed (branch or tag). Recorded for display only.").option("--no-select-specs", "Record the deploy without deciding which specs it reaches. The entry then becomes a hole in the range — every spec behind it is assumed reached rather than being cleared, and nothing can fill it in later, since the hub has no checkout to diff. The decision intersects the diff with measured coverage from the hub and calls no model, so there is rarely a reason to skip it.").option(...hubUrlOption).option(...hubTokenOption).option("--project <name>", "Project whose deploy log this entry joins. Defaults to the current directory's name.").option("--cwd <path>", "Directory the git diff and the default --project name are resolved against.").action(withHubErrors(async (opts) => {
|
|
11734
|
+
await runDeployRecord(opts);
|
|
11535
11735
|
}));
|
|
11536
11736
|
async function runDeployRecord(opts) {
|
|
11537
11737
|
const cwd = resolveCwd(opts.cwd);
|
|
@@ -11541,7 +11741,7 @@ async function runDeployRecord(opts) {
|
|
|
11541
11741
|
const runUrl = githubRunUrl();
|
|
11542
11742
|
const diff = previous === null ? null : await diffOrNull(previous, opts.sha, cwd);
|
|
11543
11743
|
const changedPaths = diff ? capDeployPaths(diff.map((f) => f.path)) : null;
|
|
11544
|
-
const selection = opts.selectSpecs !== false && previous !== null && diff !== null ? await selectionForDeploy(diff, previous, opts.sha, cwd
|
|
11744
|
+
const selection = opts.selectSpecs !== false && previous !== null && diff !== null ? await selectionForDeploy(hub, project, diff, previous, opts.sha, cwd) : void 0;
|
|
11545
11745
|
const entry = await hub.recordDeploy(project, opts.profile, {
|
|
11546
11746
|
sha: opts.sha,
|
|
11547
11747
|
previousSha: previous,
|
|
@@ -11569,11 +11769,13 @@ async function runDeployRecord(opts) {
|
|
|
11569
11769
|
* Takes the diff `deployRecord` already fetched for `changedPaths`, rather
|
|
11570
11770
|
* than diffing again — the decision needs the diff and the spec tree, and the
|
|
11571
11771
|
* hub has neither, but there's no reason to ask git for the same range twice.
|
|
11772
|
+
* The hub does hold the coverage measurements the verdicts rest on, so those
|
|
11773
|
+
* are read back through the same connection the entry is posted over.
|
|
11572
11774
|
* `undefined` on failure rather than a half-answer: the deploy is then
|
|
11573
11775
|
* recorded without a selection, and specs behind it read `unknown` instead of
|
|
11574
11776
|
* being cleared by a selection that isn't there.
|
|
11575
11777
|
*/
|
|
11576
|
-
async function selectionForDeploy(changed, previous, sha, cwd
|
|
11778
|
+
async function selectionForDeploy(hub, project, changed, previous, sha, cwd) {
|
|
11577
11779
|
try {
|
|
11578
11780
|
const specs = await loadSpecInventory(cwd);
|
|
11579
11781
|
if (specs.length === 0) return void 0;
|
|
@@ -11583,12 +11785,15 @@ async function selectionForDeploy(changed, previous, sha, cwd, model) {
|
|
|
11583
11785
|
cwd,
|
|
11584
11786
|
base: previous,
|
|
11585
11787
|
head: sha,
|
|
11586
|
-
|
|
11788
|
+
edges: await loadCoverageEdges({
|
|
11789
|
+
hub,
|
|
11790
|
+
project
|
|
11791
|
+
})
|
|
11587
11792
|
});
|
|
11588
11793
|
return Object.fromEntries(report.specs.map((s) => [specKey(s), {
|
|
11589
11794
|
verdict: s.verdict,
|
|
11590
11795
|
reason: s.reason,
|
|
11591
|
-
...s.touchedBy?.length ? { touchedBy: s.touchedBy } : {}
|
|
11796
|
+
...s.touchedBy?.length ? { touchedBy: s.touchedBy.slice(0, 10) } : {}
|
|
11592
11797
|
}]));
|
|
11593
11798
|
} catch (err) {
|
|
11594
11799
|
warn(`could not decide which specs this deploy reaches (${errMessage(err)}); recording the deploy without a selection`);
|
|
@@ -11876,6 +12081,146 @@ function buildLiveUserPrompt(step) {
|
|
|
11876
12081
|
return `Execute step ${step.id} and emit your STEP_RESULT verdict as instructed in the system prompt.`;
|
|
11877
12082
|
}
|
|
11878
12083
|
//#endregion
|
|
12084
|
+
//#region src/runtime/agent-browser-daemon.ts
|
|
12085
|
+
/**
|
|
12086
|
+
* Forcing a session's agent-browser daemon out when it has stopped answering.
|
|
12087
|
+
*
|
|
12088
|
+
* Every other way ccqa reaches a daemon is a command over that daemon's socket
|
|
12089
|
+
* — `close`, the only shutdown the CLI offers, included — so none of them work
|
|
12090
|
+
* on one that no longer reads it. The per-session pid file does not.
|
|
12091
|
+
*/
|
|
12092
|
+
/** Measured against agent-browser 0.26-0.34. */
|
|
12093
|
+
function agentBrowserRuntimeDir() {
|
|
12094
|
+
const explicit = process.env["AGENT_BROWSER_SOCKET_DIR"];
|
|
12095
|
+
const xdg = process.env["XDG_RUNTIME_DIR"];
|
|
12096
|
+
const base = explicit ?? (xdg ? join(xdg, "agent-browser") : join(homedir(), ".agent-browser"));
|
|
12097
|
+
const namespace = process.env["AGENT_BROWSER_NAMESPACE"];
|
|
12098
|
+
return namespace ? join(base, "namespaces", namespace, "run") : base;
|
|
12099
|
+
}
|
|
12100
|
+
/** The daemon pid agent-browser recorded for `sessionName`, if it wrote one. */
|
|
12101
|
+
function readDaemonPid(sessionName) {
|
|
12102
|
+
try {
|
|
12103
|
+
const pid = Number(readFileSync(join(agentBrowserRuntimeDir(), `${sessionName}.pid`), "utf8").trim());
|
|
12104
|
+
return Number.isInteger(pid) && pid > 1 ? pid : null;
|
|
12105
|
+
} catch {
|
|
12106
|
+
return null;
|
|
12107
|
+
}
|
|
12108
|
+
}
|
|
12109
|
+
function isAlive(pid) {
|
|
12110
|
+
try {
|
|
12111
|
+
process.kill(pid, 0);
|
|
12112
|
+
return true;
|
|
12113
|
+
} catch (e) {
|
|
12114
|
+
return e.code === "EPERM";
|
|
12115
|
+
}
|
|
12116
|
+
}
|
|
12117
|
+
/**
|
|
12118
|
+
* A pid file outlives an unclean exit, so its number may since have been handed
|
|
12119
|
+
* to something else. Where `ps` cannot answer — Windows has none — decline
|
|
12120
|
+
* rather than guess, which makes the kill a no-op instead of a hazard.
|
|
12121
|
+
*
|
|
12122
|
+
* Residual risk this cannot cover: a pid recycled onto *another session's*
|
|
12123
|
+
* daemon has identical argv and passes.
|
|
12124
|
+
*/
|
|
12125
|
+
function looksLikeAgentBrowser(pid) {
|
|
12126
|
+
const ps = spawnSync("ps", [
|
|
12127
|
+
"-p",
|
|
12128
|
+
String(pid),
|
|
12129
|
+
"-o",
|
|
12130
|
+
"args="
|
|
12131
|
+
], { encoding: "utf8" });
|
|
12132
|
+
return ps.status === 0 && ps.stdout.includes("agent-browser");
|
|
12133
|
+
}
|
|
12134
|
+
function childPids(parent) {
|
|
12135
|
+
const ps = spawnSync("ps", ["-eo", "pid=,ppid="], { encoding: "utf8" });
|
|
12136
|
+
if (ps.status !== 0) return [];
|
|
12137
|
+
const children = [];
|
|
12138
|
+
for (const line of ps.stdout.split("\n")) {
|
|
12139
|
+
const [pid, ppid] = line.trim().split(/\s+/).map(Number);
|
|
12140
|
+
if (pid && ppid === parent) children.push(pid);
|
|
12141
|
+
}
|
|
12142
|
+
return children;
|
|
12143
|
+
}
|
|
12144
|
+
/** Ramped so the common case — a daemon that exits at once — is not taxed. */
|
|
12145
|
+
const TERM_POLL_MS = [
|
|
12146
|
+
25,
|
|
12147
|
+
50,
|
|
12148
|
+
100,
|
|
12149
|
+
250,
|
|
12150
|
+
250,
|
|
12151
|
+
250,
|
|
12152
|
+
500,
|
|
12153
|
+
500,
|
|
12154
|
+
1e3,
|
|
12155
|
+
1e3,
|
|
12156
|
+
1e3
|
|
12157
|
+
];
|
|
12158
|
+
/** One phrase covering both outcomes, so callers log a single line. */
|
|
12159
|
+
function describeKill(kill) {
|
|
12160
|
+
return kill.killed ? `killed daemon pid ${kill.pid}` : `could not kill the daemon (${kill.reason})`;
|
|
12161
|
+
}
|
|
12162
|
+
/**
|
|
12163
|
+
* Stop the daemon serving `sessionName`. The session stays reusable —
|
|
12164
|
+
* agent-browser boots a fresh daemon under the same name and clears the stale
|
|
12165
|
+
* socket itself.
|
|
12166
|
+
*
|
|
12167
|
+
* SIGTERM is given time because a daemon that exits on its own signal takes its
|
|
12168
|
+
* browser down with it, while SIGKILL leaves that browser running with no owner.
|
|
12169
|
+
*/
|
|
12170
|
+
async function killSessionDaemon(sessionName) {
|
|
12171
|
+
const pid = readDaemonPid(sessionName);
|
|
12172
|
+
if (pid === null) return {
|
|
12173
|
+
killed: false,
|
|
12174
|
+
reason: "no pid file for this session"
|
|
12175
|
+
};
|
|
12176
|
+
if (!isAlive(pid)) return {
|
|
12177
|
+
killed: false,
|
|
12178
|
+
reason: `pid ${pid} is not running`
|
|
12179
|
+
};
|
|
12180
|
+
if (!looksLikeAgentBrowser(pid)) return {
|
|
12181
|
+
killed: false,
|
|
12182
|
+
reason: `pid ${pid} is not an agent-browser process`
|
|
12183
|
+
};
|
|
12184
|
+
try {
|
|
12185
|
+
process.kill(pid, "SIGTERM");
|
|
12186
|
+
} catch {
|
|
12187
|
+
return {
|
|
12188
|
+
killed: false,
|
|
12189
|
+
reason: `pid ${pid} could not be signalled`
|
|
12190
|
+
};
|
|
12191
|
+
}
|
|
12192
|
+
for (const wait of TERM_POLL_MS) {
|
|
12193
|
+
if (!isAlive(pid)) return {
|
|
12194
|
+
killed: true,
|
|
12195
|
+
pid
|
|
12196
|
+
};
|
|
12197
|
+
await setTimeout$1(wait);
|
|
12198
|
+
}
|
|
12199
|
+
if (!isAlive(pid)) return {
|
|
12200
|
+
killed: true,
|
|
12201
|
+
pid
|
|
12202
|
+
};
|
|
12203
|
+
const owned = childPids(pid);
|
|
12204
|
+
try {
|
|
12205
|
+
process.kill(pid, "SIGKILL");
|
|
12206
|
+
} catch {}
|
|
12207
|
+
for (const wait of TERM_POLL_MS) {
|
|
12208
|
+
if (!isAlive(pid)) break;
|
|
12209
|
+
await setTimeout$1(wait);
|
|
12210
|
+
}
|
|
12211
|
+
if (isAlive(pid)) return {
|
|
12212
|
+
killed: false,
|
|
12213
|
+
reason: `pid ${pid} survived SIGKILL`
|
|
12214
|
+
};
|
|
12215
|
+
for (const child of owned) if (isAlive(child)) try {
|
|
12216
|
+
process.kill(child, "SIGTERM");
|
|
12217
|
+
} catch {}
|
|
12218
|
+
return {
|
|
12219
|
+
killed: true,
|
|
12220
|
+
pid
|
|
12221
|
+
};
|
|
12222
|
+
}
|
|
12223
|
+
//#endregion
|
|
11879
12224
|
//#region src/runtime/live-result-parse.ts
|
|
11880
12225
|
const MAX_REASON_LEN = 2e3;
|
|
11881
12226
|
/** Parse a single STEP_RESULT line. Returns null on malformed input. */
|
|
@@ -11967,7 +12312,14 @@ async function runLiveExecutor(input) {
|
|
|
11967
12312
|
const retries = Math.max(0, input.retries ?? 0);
|
|
11968
12313
|
if (statePath) {
|
|
11969
12314
|
const injected = loadStateIntoSession(input.sessionName, statePath);
|
|
11970
|
-
if (!injected.ok
|
|
12315
|
+
if (!injected.ok && injected.wedged) {
|
|
12316
|
+
const kill = await killSessionDaemon(input.sessionName);
|
|
12317
|
+
warn(`session state restore failed: ${injected.error}; ${describeKill(kill)}`);
|
|
12318
|
+
if (kill.killed) {
|
|
12319
|
+
const retried = loadStateIntoSession(input.sessionName, statePath);
|
|
12320
|
+
if (!retried.ok) warn(`session state restore failed again: ${retried.error}`);
|
|
12321
|
+
}
|
|
12322
|
+
} else if (!injected.ok) warn(`session state restore failed: ${injected.error}`);
|
|
11971
12323
|
}
|
|
11972
12324
|
for (let i = 0; i < input.steps.length; i++) {
|
|
11973
12325
|
const step$1 = input.steps[i];
|
|
@@ -11988,15 +12340,18 @@ async function runLiveExecutor(input) {
|
|
|
11988
12340
|
for (;;) {
|
|
11989
12341
|
lastOutcome = await executeStepAttempt(step$1, paths, systemPrompt, userPrompt);
|
|
11990
12342
|
if (lastOutcome.status === "passed") break;
|
|
11991
|
-
if (!recoveredOnce
|
|
12343
|
+
if (!recoveredOnce) {
|
|
12344
|
+
recoveredOnce = true;
|
|
11992
12345
|
const health = checkLiveSessionHealth(input.sessionName);
|
|
11993
|
-
if (!health.healthy) {
|
|
11994
|
-
|
|
11995
|
-
|
|
11996
|
-
if (!
|
|
11997
|
-
|
|
11998
|
-
|
|
11999
|
-
|
|
12346
|
+
if (!health.healthy && (health.kind !== "blank" || statePath)) {
|
|
12347
|
+
const kill = health.kind === "unresponsive" ? await killSessionDaemon(input.sessionName) : null;
|
|
12348
|
+
warn(`session broken during ${step$1.id} (${health.reason}); ` + (kill ? describeKill(kill) : "re-injecting auth-state"));
|
|
12349
|
+
if (!kill || kill.killed) {
|
|
12350
|
+
const rec = recoverLiveSession(input.sessionName, statePath, verifyUrl);
|
|
12351
|
+
if (!rec.ok) warn(`session recovery failed: ${rec.error}`);
|
|
12352
|
+
attempt++;
|
|
12353
|
+
continue;
|
|
12354
|
+
}
|
|
12000
12355
|
}
|
|
12001
12356
|
}
|
|
12002
12357
|
if (attempt >= retries) break;
|
|
@@ -12040,7 +12395,8 @@ async function runLiveExecutor(input) {
|
|
|
12040
12395
|
systemPrompt,
|
|
12041
12396
|
model: input.model,
|
|
12042
12397
|
envScrubMap: input.envScrubMap,
|
|
12043
|
-
relaxAbConstraints: true
|
|
12398
|
+
relaxAbConstraints: true,
|
|
12399
|
+
timeoutMs: stepAttemptTimeoutMs()
|
|
12044
12400
|
}, (msg) => {
|
|
12045
12401
|
if (msg.type !== "assistant") return;
|
|
12046
12402
|
for (const block of msg.message.content ?? []) {
|
|
@@ -12096,6 +12452,18 @@ async function runLiveExecutor(input) {
|
|
|
12096
12452
|
* are the winning path — exactly the shortcut a later run should reuse.
|
|
12097
12453
|
*/
|
|
12098
12454
|
const MAX_LEARNED_COMMANDS = 15;
|
|
12455
|
+
/**
|
|
12456
|
+
* Wall-clock ceiling on one step attempt. The prompt's own ~3 minute wait
|
|
12457
|
+
* budget cannot end a step, because a model parked on a notification never
|
|
12458
|
+
* comes back to read it. Sized off measured runs: the longest passing step was
|
|
12459
|
+
* 4 minutes, against a wedged one that ran 15.
|
|
12460
|
+
*/
|
|
12461
|
+
const STEP_ATTEMPT_TIMEOUT_MS = 8 * 6e4;
|
|
12462
|
+
/** Env override so a slow environment can be tuned without a release. */
|
|
12463
|
+
function stepAttemptTimeoutMs() {
|
|
12464
|
+
const raw = Number(process.env["CCQA_LIVE_STEP_TIMEOUT_MS"]);
|
|
12465
|
+
return Number.isFinite(raw) && raw > 0 ? raw : STEP_ATTEMPT_TIMEOUT_MS;
|
|
12466
|
+
}
|
|
12099
12467
|
function emptyStepCost() {
|
|
12100
12468
|
return {
|
|
12101
12469
|
totalCostUsd: null,
|
|
@@ -12406,13 +12774,14 @@ function truncate(s, maxBytes) {
|
|
|
12406
12774
|
* Close an agent-browser session by name. Used before/after a `ccqa generate`
|
|
12407
12775
|
* run so a wedged daemon from a previous attempt can't hang the next one.
|
|
12408
12776
|
*
|
|
12409
|
-
* Always resolves; never throws. If the binary is missing
|
|
12410
|
-
* doesn't exist,
|
|
12411
|
-
*
|
|
12777
|
+
* Always resolves; never throws. If the binary is missing or the session
|
|
12778
|
+
* doesn't exist, we silently return — close is best-effort cleanup, not a
|
|
12779
|
+
* precondition.
|
|
12412
12780
|
*/
|
|
12413
12781
|
async function closeSession(sessionName) {
|
|
12414
12782
|
const abBin = resolveAgentBrowserBin();
|
|
12415
12783
|
if (!abBin) return;
|
|
12784
|
+
let unanswered = false;
|
|
12416
12785
|
await new Promise((resolve) => {
|
|
12417
12786
|
const child = spawn(process.execPath, [abBin, "close"], {
|
|
12418
12787
|
env: {
|
|
@@ -12422,6 +12791,7 @@ async function closeSession(sessionName) {
|
|
|
12422
12791
|
stdio: "ignore"
|
|
12423
12792
|
});
|
|
12424
12793
|
const timer = setTimeout(() => {
|
|
12794
|
+
unanswered = true;
|
|
12425
12795
|
child.kill("SIGTERM");
|
|
12426
12796
|
}, CLOSE_TIMEOUT_MS);
|
|
12427
12797
|
const finish = () => {
|
|
@@ -12431,6 +12801,7 @@ async function closeSession(sessionName) {
|
|
|
12431
12801
|
child.on("error", finish);
|
|
12432
12802
|
child.on("exit", finish);
|
|
12433
12803
|
});
|
|
12804
|
+
if (unanswered) await killSessionDaemon(sessionName);
|
|
12434
12805
|
}
|
|
12435
12806
|
//#endregion
|
|
12436
12807
|
//#region src/cli/run-live.ts
|
|
@@ -12851,155 +13222,6 @@ function oneLine(s) {
|
|
|
12851
13222
|
return s.replace(/\s+/g, " ").trim();
|
|
12852
13223
|
}
|
|
12853
13224
|
//#endregion
|
|
12854
|
-
//#region src/config/project-config.ts
|
|
12855
|
-
/**
|
|
12856
|
-
* Loader for the consumer project's `.ccqa/config.yaml` — per-target
|
|
12857
|
-
* generation settings (default target, output dirs, reusable code resources,
|
|
12858
|
-
* generation conventions).
|
|
12859
|
-
*
|
|
12860
|
-
* This module only validates and holds the config. `path` / `guides` /
|
|
12861
|
-
* `examples` entries may be glob patterns; they are kept verbatim here and
|
|
12862
|
-
* expanded by the generation engine, which owns size limits and warnings.
|
|
12863
|
-
*/
|
|
12864
|
-
/**
|
|
12865
|
-
* An existing code asset the generated tests should reuse (import), in one of
|
|
12866
|
-
* two forms — exactly one of:
|
|
12867
|
-
* - `path`: code inside the consumer repo (literal path or glob pattern);
|
|
12868
|
-
* - `package`: an installed npm package (imported by name).
|
|
12869
|
-
* `description` tells the generator what the asset contains.
|
|
12870
|
-
*/
|
|
12871
|
-
const ResourceRefSchema = z.union([z.object({
|
|
12872
|
-
path: z.string().min(1),
|
|
12873
|
-
description: z.string().optional()
|
|
12874
|
-
}).strict(), z.object({
|
|
12875
|
-
package: z.string().min(1),
|
|
12876
|
-
description: z.string().optional()
|
|
12877
|
-
}).strict()], { error: "a resource must have exactly one of `path` (code in this repo) or `package` (installed npm package), plus an optional `description`" });
|
|
12878
|
-
/**
|
|
12879
|
-
* How generated code should be written, as guide inputs to the prompt (never
|
|
12880
|
-
* imported as code): `guides` are convention documents, `examples` are
|
|
12881
|
-
* existing tests whose style to imitate. Entries may be glob patterns.
|
|
12882
|
-
*/
|
|
12883
|
-
const ConventionsSchema = z.object({
|
|
12884
|
-
guides: z.array(z.string().min(1)).default([]),
|
|
12885
|
-
examples: z.array(z.string().min(1)).default([])
|
|
12886
|
-
}).strict();
|
|
12887
|
-
/**
|
|
12888
|
-
* Per-target settings. `outDir` (where generated tests are written) and
|
|
12889
|
-
* `runCommand` (how to execute them; `{files}` expands to the generated
|
|
12890
|
-
* paths, `{artifactsDir}` to the spec's report artifacts dir — see
|
|
12891
|
-
* src/targets/run-artifacts.ts) are optional at this layer because not every
|
|
12892
|
-
* target needs them — e.g. agent-browser stores its output in the spec
|
|
12893
|
-
* directory. A target that requires either must validate its presence itself.
|
|
12894
|
-
*/
|
|
12895
|
-
const TargetConfigSchema = z.object({
|
|
12896
|
-
outDir: z.string().min(1).optional(),
|
|
12897
|
-
runCommand: z.string().min(1).optional(),
|
|
12898
|
-
resources: z.array(ResourceRefSchema).default([]),
|
|
12899
|
-
conventions: ConventionsSchema.default({
|
|
12900
|
-
guides: [],
|
|
12901
|
-
examples: []
|
|
12902
|
-
})
|
|
12903
|
-
}).strict();
|
|
12904
|
-
/**
|
|
12905
|
-
* Specs that must not run at the same time, grouped by the thing they share.
|
|
12906
|
-
*
|
|
12907
|
-
* The key names the shared thing (a chat channel, a seeded account, a tenant);
|
|
12908
|
-
* the list names the specs that write to it. `ccqa run` never runs two members
|
|
12909
|
-
* of one group concurrently, and specs sharing no group still run in parallel.
|
|
12910
|
-
*
|
|
12911
|
-
* Kept here rather than on each spec so there is one place to read the whole
|
|
12912
|
-
* picture, and so a mistyped member is a spec key that does not resolve —
|
|
12913
|
-
* caught — rather than a resource name that silently matches nothing.
|
|
12914
|
-
*/
|
|
12915
|
-
const SerialGroupsSchema = z.record(z.string().regex(/^[a-z0-9][a-z0-9._-]*$/i, "serial group name must be a slug (letters, digits, '.', '_', '-')"), z.array(z.string().min(1)).min(1));
|
|
12916
|
-
/**
|
|
12917
|
-
* Which specs act as which external identity, for the flows whose requests
|
|
12918
|
-
* cannot carry a spec id at all.
|
|
12919
|
-
*
|
|
12920
|
-
* A chat platform's webhook is sent by the platform, not the browser, so no
|
|
12921
|
-
* cookie rides along and everything the flow reaches would be unattributed.
|
|
12922
|
-
* What the request does carry is who caused it, and if only one spec is allowed
|
|
12923
|
-
* to act as that identity at a time, "who" plus "when" is enough.
|
|
12924
|
-
*
|
|
12925
|
-
* ```yaml
|
|
12926
|
-
* coverage:
|
|
12927
|
-
* actors:
|
|
12928
|
-
* slack: # the preset's tag prefix
|
|
12929
|
-
* ${TEST_USER_ID}: [chat/create-item, chat/resolve-item]
|
|
12930
|
-
* ```
|
|
12931
|
-
*
|
|
12932
|
-
* The provider name is the prefix the matching preset stamps, and the key is an
|
|
12933
|
-
* identity expression the run's variables resolve. Only the unexpanded text is
|
|
12934
|
-
* ever displayed or used as a lock key, so the identity itself stays out of
|
|
12935
|
-
* reports and the hub.
|
|
12936
|
-
*/
|
|
12937
|
-
const CoverageActorsSchema = z.record(z.string().regex(/^[a-z0-9][a-z0-9._-]*$/i, "actor provider must be a slug (letters, digits, '.', '_', '-')"), z.record(z.string().min(1), z.array(z.string().min(1)).min(1)));
|
|
12938
|
-
/**
|
|
12939
|
-
* Settings for `ccqa run --coverage`, which measures what each spec actually
|
|
12940
|
-
* reached in the application under test.
|
|
12941
|
-
*/
|
|
12942
|
-
const CoverageConfigSchema = z.object({
|
|
12943
|
-
instrumentedOrigins: z.array(z.string().min(1)).min(1),
|
|
12944
|
-
sink: z.string().min(1).default("http://127.0.0.1:4757"),
|
|
12945
|
-
projectRoot: z.string().min(1).optional(),
|
|
12946
|
-
include: z.array(z.string().min(1)).optional(),
|
|
12947
|
-
actors: CoverageActorsSchema.default({})
|
|
12948
|
-
}).strict();
|
|
12949
|
-
/**
|
|
12950
|
-
* Top-level `.ccqa/config.yaml` schema. `defaultTarget` is used by specs
|
|
12951
|
-
* with no `target:` of their own. Both defaults make a missing config file
|
|
12952
|
-
* equivalent to "agent-browser only, no extra settings".
|
|
12953
|
-
*/
|
|
12954
|
-
const ProjectConfigSchema = z.object({
|
|
12955
|
-
defaultTarget: TargetIdSchema.default(AGENT_BROWSER_TARGET),
|
|
12956
|
-
targets: z.record(TargetIdSchema, TargetConfigSchema).default({}),
|
|
12957
|
-
serialGroups: SerialGroupsSchema.default({}),
|
|
12958
|
-
coverage: CoverageConfigSchema.optional()
|
|
12959
|
-
}).strict();
|
|
12960
|
-
/** Config file location, relative to the project root (`--cwd`). */
|
|
12961
|
-
const PROJECT_CONFIG_PATH = ".ccqa/config.yaml";
|
|
12962
|
-
/**
|
|
12963
|
-
* Load `<cwd>/.ccqa/config.yaml`. A missing file yields the defaults (an
|
|
12964
|
-
* empty file too); a present but broken file is an error — never silently
|
|
12965
|
-
* fall back when the user wrote a config.
|
|
12966
|
-
*/
|
|
12967
|
-
async function loadProjectConfig(cwd) {
|
|
12968
|
-
let content;
|
|
12969
|
-
try {
|
|
12970
|
-
content = await readFile(join(cwd, PROJECT_CONFIG_PATH), "utf8");
|
|
12971
|
-
} catch (e) {
|
|
12972
|
-
if (e.code === "ENOENT") return ProjectConfigSchema.parse({});
|
|
12973
|
-
throw e;
|
|
12974
|
-
}
|
|
12975
|
-
return parseProjectConfig(content);
|
|
12976
|
-
}
|
|
12977
|
-
/** Parse config YAML. Schema rejections are rewritten with actionable messages. */
|
|
12978
|
-
function parseProjectConfig(content, source = PROJECT_CONFIG_PATH) {
|
|
12979
|
-
let raw;
|
|
12980
|
-
try {
|
|
12981
|
-
raw = parse(content);
|
|
12982
|
-
} catch (e) {
|
|
12983
|
-
throw new Error(`Failed to parse YAML (${source}): ${e.message}`);
|
|
12984
|
-
}
|
|
12985
|
-
try {
|
|
12986
|
-
return ProjectConfigSchema.parse(raw ?? {});
|
|
12987
|
-
} catch (e) {
|
|
12988
|
-
throw enrichZodError(e, source);
|
|
12989
|
-
}
|
|
12990
|
-
}
|
|
12991
|
-
/** Flatten a ZodError into one `Invalid <source>:` message, path per line. */
|
|
12992
|
-
function enrichZodError(error, source) {
|
|
12993
|
-
if (!(error instanceof ZodError)) return error;
|
|
12994
|
-
const lines = [`Invalid ${source}:`];
|
|
12995
|
-
for (const issue of error.issues) {
|
|
12996
|
-
const path = issue.path.join(".") || "(root)";
|
|
12997
|
-
const message = issue.code === "invalid_key" && issue.issues[0] ? issue.issues[0].message : issue.message;
|
|
12998
|
-
lines.push(` - ${path}: ${message}`);
|
|
12999
|
-
}
|
|
13000
|
-
return new Error(lines.join("\n"));
|
|
13001
|
-
}
|
|
13002
|
-
//#endregion
|
|
13003
13225
|
//#region src/coverage/inbox.ts
|
|
13004
13226
|
/**
|
|
13005
13227
|
* The run's side of the hub coverage inbox (ADR-0022). Under
|
|
@@ -15289,28 +15511,58 @@ function stripCodeFences(text) {
|
|
|
15289
15511
|
return m && m[1] !== void 0 ? m[1] : text;
|
|
15290
15512
|
}
|
|
15291
15513
|
//#endregion
|
|
15514
|
+
//#region src/select/types.ts
|
|
15515
|
+
/**
|
|
15516
|
+
* How the verdict was reached. Kept because the two sources have different
|
|
15517
|
+
* trust: `mechanical` is set arithmetic on paths and cannot be wrong;
|
|
15518
|
+
* `coverage` intersects the diff with the spec's last measured reach
|
|
15519
|
+
* (ADR-0024), which can only be wrong through staleness — and staleness
|
|
15520
|
+
* degrades to `unknown`, never to a guess.
|
|
15521
|
+
*/
|
|
15522
|
+
const SelectSourceSchema = z.enum(["mechanical", "coverage"]);
|
|
15523
|
+
const SpecSelectionSchema = z.object({
|
|
15524
|
+
featureName: z.string().min(1),
|
|
15525
|
+
specName: z.string().min(1),
|
|
15526
|
+
verdict: SelectVerdictSchema,
|
|
15527
|
+
source: SelectSourceSchema,
|
|
15528
|
+
reason: z.string(),
|
|
15529
|
+
touchedBy: z.array(z.string()).optional()
|
|
15530
|
+
});
|
|
15531
|
+
z.object({
|
|
15532
|
+
base: z.string(),
|
|
15533
|
+
head: z.string(),
|
|
15534
|
+
changedFiles: z.number().int().nonnegative(),
|
|
15535
|
+
specs: z.array(SpecSelectionSchema)
|
|
15536
|
+
});
|
|
15537
|
+
/** Specs the caller should actually run: everything not positively cleared. */
|
|
15538
|
+
function specsToRun(report) {
|
|
15539
|
+
return report.specs.filter((s) => s.verdict !== "notNeeded");
|
|
15540
|
+
}
|
|
15541
|
+
//#endregion
|
|
15292
15542
|
//#region src/cli/changed-specs.ts
|
|
15293
15543
|
/**
|
|
15294
15544
|
* Filter specs to those a range of commits reaches. Powers `ccqa run
|
|
15295
15545
|
* --only-affected-by <ref>`; `ccqa audit` uses the same call.
|
|
15296
15546
|
*
|
|
15297
|
-
* The decision is made by `ccqa select-specs`, which
|
|
15298
|
-
*
|
|
15299
|
-
*
|
|
15547
|
+
* The decision is made by `ccqa select-specs`, which intersects the diff with
|
|
15548
|
+
* each spec's last measured reach from the hub (ADR-0024). Deterministic and
|
|
15549
|
+
* free — no model call — and wrong in only one direction: a spec without a
|
|
15550
|
+
* measurement comes back `unknown` and runs.
|
|
15300
15551
|
*
|
|
15301
15552
|
* Specs come back `needed`, `notNeeded` or `unknown`; everything but
|
|
15302
|
-
* `notNeeded` runs. `unknown` is the selector saying it
|
|
15303
|
-
* the safe reading of that is to run the spec.
|
|
15553
|
+
* `notNeeded` runs. `unknown` is the selector saying it has no measurement to
|
|
15554
|
+
* consult, and the safe reading of that is to run the spec.
|
|
15304
15555
|
*/
|
|
15305
15556
|
async function collectChangedSpecs(specs, opts) {
|
|
15306
|
-
const { cwd, base,
|
|
15307
|
-
const
|
|
15557
|
+
const { cwd, base, hub, quiet, flagName } = opts;
|
|
15558
|
+
const flag = flagName ?? "--only-affected-by";
|
|
15559
|
+
const resolved = await resolveAnalysisBase(base, flag, cwd);
|
|
15308
15560
|
const meta$1 = (key, value) => {
|
|
15309
15561
|
if (!quiet) meta(key, value);
|
|
15310
15562
|
};
|
|
15311
15563
|
let changed;
|
|
15312
15564
|
try {
|
|
15313
|
-
changed = await getChangedFilesBetween(resolved.sha, "HEAD", cwd);
|
|
15565
|
+
changed = await getChangedFilesBetween(resolved.sha, "HEAD", cwd, { detectRenames: false });
|
|
15314
15566
|
} catch (e) {
|
|
15315
15567
|
throw new RunUsageError(`failed to run 'git diff' against ${resolved.ref}: ${e.message}`);
|
|
15316
15568
|
}
|
|
@@ -15326,13 +15578,16 @@ async function collectChangedSpecs(specs, opts) {
|
|
|
15326
15578
|
} catch (e) {
|
|
15327
15579
|
throw new RunUsageError(e.message);
|
|
15328
15580
|
}
|
|
15581
|
+
let edges = /* @__PURE__ */ new Map();
|
|
15582
|
+
if (hub) edges = await loadCoverageEdges(hub);
|
|
15583
|
+
else warn(`${flag}: no hub connection, so coverage measurements cannot be consulted — undecided specs will run`);
|
|
15329
15584
|
const report = await selectSpecs({
|
|
15330
15585
|
changed,
|
|
15331
15586
|
specs: inventory,
|
|
15332
15587
|
cwd,
|
|
15333
15588
|
base: resolved.sha,
|
|
15334
15589
|
head: "HEAD",
|
|
15335
|
-
|
|
15590
|
+
edges
|
|
15336
15591
|
});
|
|
15337
15592
|
const toRun = new Set(specsToRun(report).map(specKey));
|
|
15338
15593
|
const undecided = report.specs.filter((s) => s.verdict === "unknown").length;
|
|
@@ -15588,7 +15843,7 @@ async function executeRun(targets, opts) {
|
|
|
15588
15843
|
if (opts.onlyAffectedBy) specs = (await collectChangedSpecs(specs, {
|
|
15589
15844
|
cwd,
|
|
15590
15845
|
base: opts.onlyAffectedBy,
|
|
15591
|
-
|
|
15846
|
+
hub: hubCtx
|
|
15592
15847
|
})).specs;
|
|
15593
15848
|
meta("selected", `${specs.length} of ${before} spec${before === 1 ? "" : "s"}`);
|
|
15594
15849
|
if (specs.length === 0 && inProgress > 0) {
|
|
@@ -16501,7 +16756,7 @@ function installTeardownSignalHandlers(teardown, onSignal) {
|
|
|
16501
16756
|
}
|
|
16502
16757
|
//#endregion
|
|
16503
16758
|
//#region src/cli/run.ts
|
|
16504
|
-
const runCommand = addHubOptions(addProfileOption(addLanguageOption(new Command("run").argument("[targets...]", "Specs to run, space-separated: each '<feature>/<spec>', '<feature>', or omit for all. Duplicates are de-duped.").description("Run specs, on any target. Agent-browser specs replay the recorded test.spec.ts under vitest (default), or, with spec.yaml `mode: live`, have Claude drive agent-browser live per step. External-target specs (playwright, runn) run through the target's configured `runCommand`. A structured report (report.json + evidence) is always written; use --report-to-hub to also stream it to a hub.").optionsGroup("Which specs to run:").option("--only-affected-by <ref>", "Only specs `ccqa select-specs`
|
|
16759
|
+
const runCommand = addHubOptions(addProfileOption(addLanguageOption(new Command("run").argument("[targets...]", "Specs to run, space-separated: each '<feature>/<spec>', '<feature>', or omit for all. Duplicates are de-duped.").description("Run specs, on any target. Agent-browser specs replay the recorded test.spec.ts under vitest (default), or, with spec.yaml `mode: live`, have Claude drive agent-browser live per step. External-target specs (playwright, runn) run through the target's configured `runCommand`. A structured report (report.json + evidence) is always written; use --report-to-hub to also stream it to a hub.").optionsGroup("Which specs to run:").option("--only-affected-by <ref>", "Only specs `ccqa select-specs` decides the diff against <ref> reaches (e.g. origin/main), by intersecting it with each spec's measured coverage from the hub. In pull_request CI, pass $GITHUB_BASE_REF. Cannot be combined with an explicit spec id.").option("--only-hub-rerun-needed", "Only specs the hub answers `rerunNeeded` for: the audit cleared them, and their last result does not cover what is deployed — including every spec the deploy log cannot place, which is assumed reached rather than skipped. A spec whose audit has not caught up answers `inProgress`, and one the audit rejected or whose last run failed answers `needsRepair`; neither is taken, because running them races the audit or repairs nothing. No git diff involved. Requires a hub connection and --hub-profile.").option("--dry-run", "Print the specs this invocation would run, then exit 0 without executing anything and without writing a report. Works with every selection flag.").optionsGroup("How to run them:").option("--concurrency <n>", "Run up to N specs in parallel within each phase (deterministic / external-target / live), never across phases. Default 1 (sequential). Specs in the same `serialGroups` entry of .ccqa/config.yaml still take turns. Live specs each get an isolated agent-browser session; high values spawn many headed Chrome instances.", parseConcurrency$1, 1).option("-m, --model <name>", "Claude model alias ('sonnet'|'opus'|'haiku') or full ID. Overrides CCQA_MODEL.").option("--live-step-retry <n>", "(live only) Retry each failed step up to N more times before recording failure. This retries a step, not the whole spec — see --on-fail-explain-rerun for that.", (raw) => {
|
|
16505
16760
|
const n = Number(raw);
|
|
16506
16761
|
if (!Number.isFinite(n) || n < 0 || Math.floor(n) !== n) throw new Error(`--live-step-retry must be a non-negative integer, got "${raw}"`);
|
|
16507
16762
|
return n;
|
|
@@ -17220,7 +17475,7 @@ function runValidationAction(action, sessionName, envOverrides = {}) {
|
|
|
17220
17475
|
};
|
|
17221
17476
|
}
|
|
17222
17477
|
let result = spawnAB(built);
|
|
17223
|
-
if (result.status !== 0 &&
|
|
17478
|
+
if (result.status !== 0 && result.wedged === true) result = spawnAB(built);
|
|
17224
17479
|
if (result.status === 0) return {
|
|
17225
17480
|
skipped: false,
|
|
17226
17481
|
ok: true,
|
|
@@ -17351,10 +17606,6 @@ function rescueLostSteps(actions, kept, dropped, opts) {
|
|
|
17351
17606
|
rescuedSteps
|
|
17352
17607
|
};
|
|
17353
17608
|
}
|
|
17354
|
-
/** Did this agent-browser invocation get SIGTERM'd by the ccqa hard-timeout watchdog? */
|
|
17355
|
-
function looksLikeHardTimeout(result) {
|
|
17356
|
-
return result.stderr.includes("agent-browser killed after hard timeout");
|
|
17357
|
-
}
|
|
17358
17609
|
/**
|
|
17359
17610
|
* Passive (read-only) actions whose only effect is observation. When a
|
|
17360
17611
|
* preceding action fails, dropping these too is the right move because
|
|
@@ -19218,7 +19469,7 @@ function selectSpecsNeedingAudit(targets, report, stillDrifted = /* @__PURE__ */
|
|
|
19218
19469
|
//#endregion
|
|
19219
19470
|
//#region src/cli/audit.ts
|
|
19220
19471
|
const DEFAULT_CONCURRENCY = 3;
|
|
19221
|
-
const auditCommand = addProfileOption(addLanguageOption(new Command("audit").argument("[feature/spec]", "Optional spec id. If omitted, every spec under .ccqa/features/ is checked.").description("Read each spec against the code it describes and report where the two have drifted. Static: no browser is run, so this is the cheap check to put in front of `ccqa run`.").optionsGroup("Which specs to audit:").option("--only-affected-by <ref>", "Only specs `ccqa select-specs`
|
|
19472
|
+
const auditCommand = addProfileOption(addLanguageOption(new Command("audit").argument("[feature/spec]", "Optional spec id. If omitted, every spec under .ccqa/features/ is checked.").description("Read each spec against the code it describes and report where the two have drifted. Static: no browser is run, so this is the cheap check to put in front of `ccqa run`.").optionsGroup("Which specs to audit:").option("--only-affected-by <ref>", "Only specs `ccqa select-specs` decides the diff against <ref> reaches (e.g. origin/main), by intersecting it with each spec's measured coverage from the hub. In pull_request CI, pass $GITHUB_BASE_REF. Specs without a measurement are audited rather than skipped.").option("--only-hub-audit-needed", "Only specs the hub says a deploy has landed on since the audit last read them. A spec that was never audited is always included, one the hub cannot answer for is audited rather than skipped, and one whose drift entry is still open is always re-audited — a merged fix changes only the spec tree, which no deploy answer covers. No git diff involved. Requires a hub connection and --hub-profile.").optionsGroup("How to run it:").option("--concurrency <n>", `Parallel spec checks (default: ${DEFAULT_CONCURRENCY})`).option("-m, --model <name>", "Claude model alias ('sonnet'|'opus'|'haiku') or full ID. Overrides CCQA_MODEL.").optionsGroup("What to do with the results:").option("--report-format <fmt>", "Output format: text | json | github", "text").option("--report-to-hub", "Push the result to a ccqa hub as a run (kind: drift), which is what updates the drift ledger. A spec it finds drifted answers `needsRepair` to `ccqa run --only-hub-rerun-needed`, and is not run until a person repairs it.").option("--exit-on <level>", "Exit non-zero on this severity or higher: warn | error", "error").optionsGroup("Environment and connection:").option("--cwd <path>", "Working directory used as both the .ccqa root and the codebase Claude reads. Useful for monorepos. Defaults to process.cwd().").option("--project <name>", "Logical project name for the pushed run. Defaults to the current directory's name.").option(...hubUrlOption).option(...hubTokenOption).option(...hubHeaderOption))).action(withUsageErrors(async (specPath, opts) => {
|
|
19222
19473
|
await withCostReporting("audit", () => runAudit(specPath, opts));
|
|
19223
19474
|
}));
|
|
19224
19475
|
async function runAudit(specPath, opts) {
|
|
@@ -19287,7 +19538,7 @@ async function runAudit(specPath, opts) {
|
|
|
19287
19538
|
cwd,
|
|
19288
19539
|
base: opts.onlyAffectedBy,
|
|
19289
19540
|
quiet: format !== "text",
|
|
19290
|
-
|
|
19541
|
+
hub: resolveAuditHubContext(opts, cwd)
|
|
19291
19542
|
});
|
|
19292
19543
|
targets = selection.specs;
|
|
19293
19544
|
baseRef = selection.base.ref;
|
|
@@ -20060,26 +20311,40 @@ function parseSummaries(json) {
|
|
|
20060
20311
|
//#endregion
|
|
20061
20312
|
//#region src/cli/select-specs.ts
|
|
20062
20313
|
const DEFAULT_HEAD = "HEAD";
|
|
20063
|
-
const selectSpecsCommand = new Command("select-specs").description("Decide which specs a range of commits reaches.
|
|
20064
|
-
await withCostReporting("select-specs", () => runSelectSpecs(opts));
|
|
20065
|
-
});
|
|
20314
|
+
const selectSpecsCommand = new Command("select-specs").description("Decide which specs a range of commits reaches. Intersects the diff with each spec's last measured reach from the hub (`ccqa run --coverage`) and returns one verdict per spec: needed | notNeeded | unknown. Requires a hub connection; a spec with no measurement is unknown, which runs.").requiredOption("--base <ref>", "Commit the range starts at — typically what is currently deployed, or the previous commit on the branch.").option("--head <ref>", `Commit the range ends at (default: ${DEFAULT_HEAD})`).option("--cwd <path>", "Working directory used as the .ccqa root. Changes outside it are reported but never attributed to a spec. Defaults to process.cwd().").option("--project <name>", "Project whose coverage measurements are read from the hub. Defaults to the current directory's name.").option(...hubUrlOption).option(...hubTokenOption).option(...hubHeaderOption).option("--format <fmt>", "Output format: text | json", "text").action(runSelectSpecs);
|
|
20066
20315
|
async function runSelectSpecs(opts) {
|
|
20067
20316
|
const format = parseFormat(opts.format);
|
|
20068
20317
|
const cwd = resolveCwd(opts.cwd);
|
|
20069
20318
|
const head = opts.head ?? DEFAULT_HEAD;
|
|
20070
|
-
const
|
|
20071
|
-
|
|
20072
|
-
specs
|
|
20073
|
-
|
|
20074
|
-
|
|
20075
|
-
|
|
20076
|
-
|
|
20077
|
-
|
|
20078
|
-
|
|
20079
|
-
|
|
20080
|
-
|
|
20081
|
-
|
|
20082
|
-
|
|
20319
|
+
const hub = resolveHubClient(opts);
|
|
20320
|
+
if (!hub) {
|
|
20321
|
+
error(needsHubConnection("select-specs"));
|
|
20322
|
+
process.exit(2);
|
|
20323
|
+
}
|
|
20324
|
+
const project = resolveProject({
|
|
20325
|
+
project: opts.project,
|
|
20326
|
+
cwd: opts.cwd
|
|
20327
|
+
});
|
|
20328
|
+
const [specsResult, changedResult, edges] = await Promise.all([
|
|
20329
|
+
loadSpecInventory(cwd).then((specs) => ({
|
|
20330
|
+
ok: true,
|
|
20331
|
+
specs
|
|
20332
|
+
}), (e) => ({
|
|
20333
|
+
ok: false,
|
|
20334
|
+
error: e
|
|
20335
|
+
})),
|
|
20336
|
+
getChangedFilesBetween(opts.base, head, cwd, { detectRenames: false }).then((changed) => ({
|
|
20337
|
+
ok: true,
|
|
20338
|
+
changed
|
|
20339
|
+
}), (e) => ({
|
|
20340
|
+
ok: false,
|
|
20341
|
+
error: e
|
|
20342
|
+
})),
|
|
20343
|
+
loadCoverageEdges({
|
|
20344
|
+
hub,
|
|
20345
|
+
project
|
|
20346
|
+
})
|
|
20347
|
+
]);
|
|
20083
20348
|
if (!specsResult.ok) {
|
|
20084
20349
|
error(specsResult.error.message);
|
|
20085
20350
|
process.exit(1);
|
|
@@ -20097,8 +20362,10 @@ async function runSelectSpecs(opts) {
|
|
|
20097
20362
|
if (format === "text") {
|
|
20098
20363
|
header("select-specs", `${opts.base} → ${head}`);
|
|
20099
20364
|
if (opts.cwd) meta("cwd", cwd);
|
|
20365
|
+
meta("project", project);
|
|
20100
20366
|
meta("changed-files", changed.length);
|
|
20101
20367
|
meta("specs", specs.length);
|
|
20368
|
+
meta("measured-specs", edges.size);
|
|
20102
20369
|
}
|
|
20103
20370
|
const report = await selectSpecs({
|
|
20104
20371
|
changed,
|
|
@@ -20106,7 +20373,7 @@ async function runSelectSpecs(opts) {
|
|
|
20106
20373
|
cwd,
|
|
20107
20374
|
base: opts.base,
|
|
20108
20375
|
head,
|
|
20109
|
-
|
|
20376
|
+
edges
|
|
20110
20377
|
});
|
|
20111
20378
|
process.stdout.write(format === "json" ? `${JSON.stringify(report, null, 2)}\n` : renderText(report));
|
|
20112
20379
|
process.exit(0);
|
|
@@ -21266,68 +21533,6 @@ function createGetDriftLedgerHandler(storage) {
|
|
|
21266
21533
|
});
|
|
21267
21534
|
};
|
|
21268
21535
|
}
|
|
21269
|
-
function emptyDeployLog() {
|
|
21270
|
-
return {
|
|
21271
|
-
nextIndex: 0,
|
|
21272
|
-
entries: []
|
|
21273
|
-
};
|
|
21274
|
-
}
|
|
21275
|
-
/** Append `input` to `current`; the appended entry is always the last of `entries`. */
|
|
21276
|
-
function appendDeploy(current, input) {
|
|
21277
|
-
const log = current ?? emptyDeployLog();
|
|
21278
|
-
const head = log.entries[log.entries.length - 1];
|
|
21279
|
-
const gapBefore = head ? head.sha !== input.previousSha : log.nextIndex > 0;
|
|
21280
|
-
const entries = [...log.entries, {
|
|
21281
|
-
...input,
|
|
21282
|
-
index: log.nextIndex,
|
|
21283
|
-
changedPaths: input.changedPaths === null ? null : input.changedPaths.slice(0, 500),
|
|
21284
|
-
gapBefore
|
|
21285
|
-
}];
|
|
21286
|
-
if (entries.length > 200) {
|
|
21287
|
-
entries.splice(0, entries.length - 200);
|
|
21288
|
-
entries[0] = {
|
|
21289
|
-
...entries[0],
|
|
21290
|
-
gapBefore: true
|
|
21291
|
-
};
|
|
21292
|
-
}
|
|
21293
|
-
return {
|
|
21294
|
-
nextIndex: log.nextIndex + 1,
|
|
21295
|
-
entries
|
|
21296
|
-
};
|
|
21297
|
-
}
|
|
21298
|
-
/**
|
|
21299
|
-
* Fold one deploy's selection into the touch index.
|
|
21300
|
-
*
|
|
21301
|
-
* Only `needed` moves a position a verdict reads. An `unknown` records the
|
|
21302
|
-
* newest undecided position too, but that one is record-only (ADR-0023) —
|
|
21303
|
-
* freshness reads an undecided judgment as "did not reach". A `notNeeded`
|
|
21304
|
-
* writes neither — it is the absence of a marker at this position, which is
|
|
21305
|
-
* exactly what a later baseline comparison reads it as.
|
|
21306
|
-
*
|
|
21307
|
-
* Positions only ever advance. Deploys are folded in log order, so a spec
|
|
21308
|
-
* needed at #7 and cleared at #9 keeps `needed.index: 7`: a baseline at #5
|
|
21309
|
-
* must still see that #7 touched it.
|
|
21310
|
-
*/
|
|
21311
|
-
function foldTouchIndex(current, entry, selection) {
|
|
21312
|
-
const out = { ...current };
|
|
21313
|
-
for (const [key, decision] of Object.entries(selection)) {
|
|
21314
|
-
const previous = out[key] ?? {};
|
|
21315
|
-
if (decision.verdict === "needed") out[key] = {
|
|
21316
|
-
...previous,
|
|
21317
|
-
needed: {
|
|
21318
|
-
index: entry.index,
|
|
21319
|
-
sha: entry.sha,
|
|
21320
|
-
at: entry.at,
|
|
21321
|
-
matchedPaths: (decision.touchedBy ?? []).slice(0, 10)
|
|
21322
|
-
}
|
|
21323
|
-
};
|
|
21324
|
-
else if (decision.verdict === "unknown") out[key] = {
|
|
21325
|
-
...previous,
|
|
21326
|
-
undecidedIndex: entry.index
|
|
21327
|
-
};
|
|
21328
|
-
}
|
|
21329
|
-
return out;
|
|
21330
|
-
}
|
|
21331
21536
|
//#endregion
|
|
21332
21537
|
//#region src/hub/api/handlers/deploys.ts
|
|
21333
21538
|
/** `changedPaths` for a wide refactor can run to tens of thousands of entries. */
|