argus-reviewer-e2e 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +84 -71
- package/action/action.yml +134 -10
- package/action/approval-review.mjs +13 -3
- package/action/bootstrap.mjs +4 -0
- package/action/emit-review.mjs +16 -0
- package/action/runtime.mjs +32 -0
- package/action/sticky-comment.cjs +1260 -479
- package/dist/cli.d.ts +103 -7
- package/dist/cli.js +1208 -186
- package/dist/config.d.ts +96 -11
- package/dist/config.js +102 -4
- package/dist/detect.d.ts +29 -2
- package/dist/detect.js +98 -7
- package/dist/driver/browser.d.ts +32 -0
- package/dist/driver/browser.js +56 -1
- package/dist/driver/target.d.ts +4 -1
- package/dist/driver/target.js +27 -6
- package/dist/engine/actions.d.ts +5 -0
- package/dist/engine/actions.js +8 -0
- package/dist/engine/explore.d.ts +78 -0
- package/dist/engine/explore.js +373 -0
- package/dist/engine/loop.d.ts +2 -2
- package/dist/engine/loop.js +8 -8
- package/dist/engine/prompts.d.ts +28 -1
- package/dist/engine/prompts.js +88 -0
- package/dist/evidence/ci.d.ts +13 -1
- package/dist/evidence/ci.js +38 -3
- package/dist/evidence/gate.d.ts +8 -0
- package/dist/evidence/gate.js +1 -1
- package/dist/evidence/link.js +1 -1
- package/dist/executor/a0.d.ts +114 -1
- package/dist/executor/a0.js +216 -4
- package/dist/fsutil.d.ts +3 -2
- package/dist/fsutil.js +7 -4
- package/dist/journal/schema.d.ts +1 -1
- package/dist/log.d.ts +2 -1
- package/dist/log.js +10 -2
- package/dist/mention.d.ts +45 -0
- package/dist/mention.js +107 -0
- package/dist/pipeline/app.d.ts +126 -0
- package/dist/pipeline/app.js +250 -0
- package/dist/pipeline/budget.d.ts +1 -0
- package/dist/pipeline/budget.js +1 -1
- package/dist/pipeline/verify.d.ts +20 -3
- package/dist/pipeline/verify.js +189 -35
- package/dist/probe/persist.d.ts +68 -0
- package/dist/probe/persist.js +184 -0
- package/dist/probe/queue.d.ts +12 -0
- package/dist/probe/queue.js +10 -2
- package/dist/report/brand-assets.generated.d.ts +9 -0
- package/dist/report/brand-assets.generated.js +8 -0
- package/dist/report/comment.d.ts +99 -6
- package/dist/report/comment.js +292 -103
- package/dist/report/html.d.ts +50 -0
- package/dist/report/html.js +879 -0
- package/dist/report/manifest.d.ts +29 -0
- package/dist/report/manifest.js +37 -0
- package/dist/report/run.d.ts +54 -1
- package/dist/report/run.js +34 -9
- package/dist/report/viewmodel.d.ts +91 -0
- package/dist/report/viewmodel.js +241 -0
- package/dist/review/adjudicate.d.ts +6 -6
- package/dist/review/adjudicate.js +2 -2
- package/dist/review/inline.d.ts +44 -0
- package/dist/review/inline.js +95 -0
- package/dist/review/packs.d.ts +21 -0
- package/dist/review/packs.js +47 -0
- package/dist/review/scope.d.ts +16 -0
- package/dist/review/scope.js +74 -0
- package/dist/review/secrets.d.ts +10 -10
- package/dist/review/secrets.js +7 -7
- package/dist/review/testfiles.d.ts +18 -0
- package/dist/review/testfiles.js +26 -0
- package/dist/review/triage.d.ts +1 -1
- package/dist/review/triage.js +10 -10
- package/dist/review/validate.d.ts +41 -0
- package/dist/review/validate.js +76 -0
- package/dist/ui/errors.d.ts +54 -0
- package/dist/ui/errors.js +236 -0
- package/dist/ui/style.d.ts +34 -0
- package/dist/ui/style.js +48 -0
- package/dist/ui/summary.d.ts +38 -0
- package/dist/ui/summary.js +101 -0
- package/dist/vision/cost.d.ts +1 -1
- package/dist/vision/decisions.d.ts +9 -3
- package/dist/vision/decisions.js +31 -21
- package/dist/vision/openrouter.d.ts +4 -0
- package/dist/vision/openrouter.js +30 -4
- package/package.json +11 -2
|
@@ -32,6 +32,14 @@ export interface BudgetSummary {
|
|
|
32
32
|
maxTasks: number | undefined;
|
|
33
33
|
tasks: number;
|
|
34
34
|
}
|
|
35
|
+
export interface CacheSummary {
|
|
36
|
+
hits: number;
|
|
37
|
+
misses: number;
|
|
38
|
+
heals: number;
|
|
39
|
+
staleEntries: number;
|
|
40
|
+
assertionHits: number;
|
|
41
|
+
assertionMisses: number;
|
|
42
|
+
}
|
|
35
43
|
export interface LaneManifest {
|
|
36
44
|
lane: LaneId;
|
|
37
45
|
selected: boolean;
|
|
@@ -45,6 +53,8 @@ export interface LaneManifest {
|
|
|
45
53
|
usage: UsageSummary;
|
|
46
54
|
budget: BudgetSummary;
|
|
47
55
|
headBinding: HeadBinding | undefined;
|
|
56
|
+
/** Flow-lane replay economics — defined only when the lane produced a run.json. */
|
|
57
|
+
cache: CacheSummary | undefined;
|
|
48
58
|
}
|
|
49
59
|
export interface RunIdentity {
|
|
50
60
|
repo: string | undefined;
|
|
@@ -52,6 +62,15 @@ export interface RunIdentity {
|
|
|
52
62
|
intendedHeadSha: string | undefined;
|
|
53
63
|
checkoutSha: string | undefined;
|
|
54
64
|
baseSha: string | undefined;
|
|
65
|
+
/**
|
|
66
|
+
* Run-scoped nonce (GITHUB_RUN_ID) — not knowable when a commit or a
|
|
67
|
+
* planted file is authored, so residue and plants fail the post step's
|
|
68
|
+
* freshness gate even when their head sha happens to match. A freshness
|
|
69
|
+
* marker, not a secret: it is public once the run exists. RUN_ATTEMPT is
|
|
70
|
+
* deliberately excluded — a re-run of failed jobs would otherwise
|
|
71
|
+
* invalidate evidence the same run produced. Undefined for local runs.
|
|
72
|
+
*/
|
|
73
|
+
runNonce: string | undefined;
|
|
55
74
|
}
|
|
56
75
|
export interface RunManifest {
|
|
57
76
|
schemaVersion: typeof MANIFEST_SCHEMA_VERSION;
|
|
@@ -85,3 +104,13 @@ export declare function aggregateLanes(lanes: Record<LaneId, LaneManifest>): {
|
|
|
85
104
|
tokens: number;
|
|
86
105
|
};
|
|
87
106
|
export declare function addProviderUsage(usage: UsageSummary, calls: CallCost[] | undefined): UsageSummary;
|
|
107
|
+
/** Run manifests are archived under `<reportDir>/manifests/<runId>.json`. */
|
|
108
|
+
export declare const MANIFEST_HISTORY_DIR = "manifests";
|
|
109
|
+
/**
|
|
110
|
+
* Archive a completed verify manifest into local history and prune to the
|
|
111
|
+
* retention bound — the dashboard/TUI run list reads this directory.
|
|
112
|
+
* runIds are timestamp-prefixed, so name sort is chronological; pruning
|
|
113
|
+
* drops the oldest names beyond `keep`. `keep <= 0` writes nothing and
|
|
114
|
+
* clears nothing existing (retention governs new archives, not deletes).
|
|
115
|
+
*/
|
|
116
|
+
export declare function archiveManifest(reportDir: string, manifest: RunManifest, keep: number): Promise<void>;
|
package/dist/report/manifest.js
CHANGED
|
@@ -1,4 +1,7 @@
|
|
|
1
|
+
import { readdir, unlink } from 'node:fs/promises';
|
|
2
|
+
import { join } from 'node:path';
|
|
1
3
|
import { defaultExec } from '../detect.js';
|
|
4
|
+
import { writeAtomicJson } from '../fsutil.js';
|
|
2
5
|
export const LANE_IDS = ['review', 'flow', 'app', 'a0'];
|
|
3
6
|
export const LANE_STATUSES = [
|
|
4
7
|
'passed',
|
|
@@ -102,6 +105,7 @@ export function emptyLane(lane, selected) {
|
|
|
102
105
|
usage: emptyUsage(),
|
|
103
106
|
budget: emptyBudget(),
|
|
104
107
|
headBinding: undefined,
|
|
108
|
+
cache: undefined,
|
|
105
109
|
};
|
|
106
110
|
}
|
|
107
111
|
export function aggregateLanes(lanes) {
|
|
@@ -141,3 +145,36 @@ export function addProviderUsage(usage, calls) {
|
|
|
141
145
|
metered: true,
|
|
142
146
|
};
|
|
143
147
|
}
|
|
148
|
+
/** Run manifests are archived under `<reportDir>/manifests/<runId>.json`. */
|
|
149
|
+
export const MANIFEST_HISTORY_DIR = 'manifests';
|
|
150
|
+
/**
|
|
151
|
+
* Archive a completed verify manifest into local history and prune to the
|
|
152
|
+
* retention bound — the dashboard/TUI run list reads this directory.
|
|
153
|
+
* runIds are timestamp-prefixed, so name sort is chronological; pruning
|
|
154
|
+
* drops the oldest names beyond `keep`. `keep <= 0` writes nothing and
|
|
155
|
+
* clears nothing existing (retention governs new archives, not deletes).
|
|
156
|
+
*/
|
|
157
|
+
export async function archiveManifest(reportDir, manifest, keep) {
|
|
158
|
+
if (keep <= 0)
|
|
159
|
+
return;
|
|
160
|
+
const dir = join(reportDir, MANIFEST_HISTORY_DIR);
|
|
161
|
+
// runId becomes a filename — never trust it as a path component.
|
|
162
|
+
const safeName = manifest.runId.replace(/[^\w.-]/g, '-');
|
|
163
|
+
await writeAtomicJson(join(dir, `${safeName}.json`), manifest);
|
|
164
|
+
let names;
|
|
165
|
+
try {
|
|
166
|
+
names = (await readdir(dir)).filter((n) => n.endsWith('.json')).sort();
|
|
167
|
+
}
|
|
168
|
+
catch {
|
|
169
|
+
return;
|
|
170
|
+
}
|
|
171
|
+
const excess = names.length - keep;
|
|
172
|
+
for (const name of names.slice(0, Math.max(0, excess))) {
|
|
173
|
+
try {
|
|
174
|
+
await unlink(join(dir, name));
|
|
175
|
+
}
|
|
176
|
+
catch {
|
|
177
|
+
// already gone — pruning is best-effort
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
}
|
package/dist/report/run.d.ts
CHANGED
|
@@ -1,4 +1,6 @@
|
|
|
1
1
|
import type { HealEvent, TdAssertRecord, TdStepRecord } from '../api.js';
|
|
2
|
+
import type { PageCapture } from '../driver/browser.js';
|
|
3
|
+
import type { ExploreStopReason } from '../engine/explore.js';
|
|
2
4
|
import type { CacheStats } from '../engine/loop.js';
|
|
3
5
|
import type { CallCost } from '../vision/cost.js';
|
|
4
6
|
/**
|
|
@@ -25,6 +27,13 @@ export interface TestReport {
|
|
|
25
27
|
cache?: CacheStats;
|
|
26
28
|
/** Agent Zero's autonomous second opinion on a failure (heal: 'a0'). */
|
|
27
29
|
a0Diagnosis?: string;
|
|
30
|
+
/**
|
|
31
|
+
* Exploratory captures from this file's browser session (U4a):
|
|
32
|
+
* console errors, page errors, and failed same-origin requests observed
|
|
33
|
+
* while the tests ran. `observed` findings — never verdict-changing.
|
|
34
|
+
* Present only when explore.enabled and anomalies occurred.
|
|
35
|
+
*/
|
|
36
|
+
captures?: PageCapture[];
|
|
28
37
|
}
|
|
29
38
|
export interface RunTotals {
|
|
30
39
|
tests: number;
|
|
@@ -32,6 +41,8 @@ export interface RunTotals {
|
|
|
32
41
|
failed: number;
|
|
33
42
|
visionCalls: number;
|
|
34
43
|
visionCostUsd: number;
|
|
44
|
+
/** Token rollup across per-test calls AND lane-level extraCalls. */
|
|
45
|
+
visionTokens: number;
|
|
35
46
|
sandboxSeconds: number;
|
|
36
47
|
budgetExceeded: boolean;
|
|
37
48
|
/** OpenRouter spend grouped by model id. */
|
|
@@ -51,10 +62,52 @@ export interface RunReport {
|
|
|
51
62
|
durationMs: number;
|
|
52
63
|
ok: boolean;
|
|
53
64
|
totals: RunTotals;
|
|
65
|
+
/** Workflow-run nonce (GITHUB_RUN_ID) — freshness binding for post steps. */
|
|
66
|
+
runNonce?: string;
|
|
54
67
|
tests: TestReport[];
|
|
55
68
|
artifacts: {
|
|
56
69
|
videos: string[];
|
|
57
70
|
};
|
|
71
|
+
/**
|
|
72
|
+
* Exploratory-lane summary (U4). Present only when `explore.enabled` —
|
|
73
|
+
* the comment renders an Exploratory section for it. `skipped` carries an
|
|
74
|
+
* explicit reason when the lane was on but could not observe anything
|
|
75
|
+
* (e.g. the target URL was unreachable), so a silent no-op is impossible.
|
|
76
|
+
* `steps`/`visited`/`stopReason`/`visionCalls`/`visionCostUsd`/`captures`
|
|
77
|
+
* describe the bounded free-explore act pass (U4b) when it ran; they are
|
|
78
|
+
* `observed` evidence, never verdict-changing.
|
|
79
|
+
*/
|
|
80
|
+
explore?: {
|
|
81
|
+
enabled: boolean;
|
|
82
|
+
skipped?: string;
|
|
83
|
+
steps?: number;
|
|
84
|
+
visited?: number;
|
|
85
|
+
stopReason?: ExploreStopReason;
|
|
86
|
+
visionCalls?: number;
|
|
87
|
+
visionCostUsd?: number;
|
|
88
|
+
captures?: PageCapture[];
|
|
89
|
+
/** Video of the explore session itself (also in `artifacts.videos`). */
|
|
90
|
+
videoPath?: string;
|
|
91
|
+
};
|
|
58
92
|
}
|
|
59
|
-
export declare function buildRunReport(tests: TestReport[], startedAt: Date, durationMs: number
|
|
93
|
+
export declare function buildRunReport(tests: TestReport[], startedAt: Date, durationMs: number,
|
|
94
|
+
/**
|
|
95
|
+
* Model calls outside the per-test sessions (the explore act pass bills
|
|
96
|
+
* its own ledger). Folded into the run totals so `run.json` spend is
|
|
97
|
+
* complete even though no TestReport owns these calls.
|
|
98
|
+
*/
|
|
99
|
+
extraCalls?: CallCost[],
|
|
100
|
+
/**
|
|
101
|
+
* An explore lane that actually ran makes a zero-test run a legitimate
|
|
102
|
+
* shape — the act pass is the evidence source for repos with no recorded
|
|
103
|
+
* flows. A configured-but-skipped/errored pass is not evidence: the run
|
|
104
|
+
* fails closed instead of reading a silent no-op as green.
|
|
105
|
+
*/
|
|
106
|
+
exploreEvidence?: boolean,
|
|
107
|
+
/**
|
|
108
|
+
* Workflow-run nonce (GITHUB_RUN_ID). The run manifests bind evidence to
|
|
109
|
+
* a run via `identity.runNonce`; run.json carries the same stamp so the
|
|
110
|
+
* serialized-verdict fallback in the post step is bound too.
|
|
111
|
+
*/
|
|
112
|
+
runNonce?: string): RunReport;
|
|
60
113
|
export declare function writeRunReport(path: string, report: RunReport): Promise<void>;
|
package/dist/report/run.js
CHANGED
|
@@ -1,25 +1,49 @@
|
|
|
1
1
|
import { writeAtomicJson } from '../fsutil.js';
|
|
2
|
-
export function buildRunReport(tests, startedAt, durationMs
|
|
2
|
+
export function buildRunReport(tests, startedAt, durationMs,
|
|
3
|
+
/**
|
|
4
|
+
* Model calls outside the per-test sessions (the explore act pass bills
|
|
5
|
+
* its own ledger). Folded into the run totals so `run.json` spend is
|
|
6
|
+
* complete even though no TestReport owns these calls.
|
|
7
|
+
*/
|
|
8
|
+
extraCalls = [],
|
|
9
|
+
/**
|
|
10
|
+
* An explore lane that actually ran makes a zero-test run a legitimate
|
|
11
|
+
* shape — the act pass is the evidence source for repos with no recorded
|
|
12
|
+
* flows. A configured-but-skipped/errored pass is not evidence: the run
|
|
13
|
+
* fails closed instead of reading a silent no-op as green.
|
|
14
|
+
*/
|
|
15
|
+
exploreEvidence = false,
|
|
16
|
+
/**
|
|
17
|
+
* Workflow-run nonce (GITHUB_RUN_ID). The run manifests bind evidence to
|
|
18
|
+
* a run via `identity.runNonce`; run.json carries the same stamp so the
|
|
19
|
+
* serialized-verdict fallback in the post step is bound too.
|
|
20
|
+
*/
|
|
21
|
+
runNonce) {
|
|
3
22
|
const failed = tests.filter((t) => !t.ok).length;
|
|
4
23
|
const callsByModel = {};
|
|
5
24
|
const costByModel = {};
|
|
6
|
-
for (const
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
costByModel[c.model] = (costByModel[c.model] ?? 0) + c.costUsd;
|
|
10
|
-
}
|
|
25
|
+
for (const c of [...tests.flatMap((t) => t.calls), ...extraCalls]) {
|
|
26
|
+
callsByModel[c.model] = (callsByModel[c.model] ?? 0) + 1;
|
|
27
|
+
costByModel[c.model] = (costByModel[c.model] ?? 0) + c.costUsd;
|
|
11
28
|
}
|
|
29
|
+
const extraVisionCost = extraCalls.reduce((s, c) => s + c.costUsd, 0);
|
|
30
|
+
const visionTokens = [...tests.flatMap((t) => t.calls), ...extraCalls].reduce((s, c) => s + c.tokens, 0);
|
|
12
31
|
return {
|
|
13
32
|
tool: 'argus-reviewer',
|
|
14
33
|
startedAt: startedAt.toISOString(),
|
|
15
34
|
durationMs,
|
|
16
|
-
|
|
35
|
+
// Zero executed tests is not a pass — an empty suite produces no evidence,
|
|
36
|
+
// so the report fails closed rather than letting a misconfigured testsDir
|
|
37
|
+
// or a non-matching pattern read as green. A completed explore pass is
|
|
38
|
+
// the exception: its observation is the evidence.
|
|
39
|
+
ok: (tests.length > 0 || exploreEvidence) && failed === 0,
|
|
17
40
|
totals: {
|
|
18
41
|
tests: tests.length,
|
|
19
42
|
passed: tests.length - failed,
|
|
20
43
|
failed,
|
|
21
|
-
visionCalls: tests.reduce((sum, t) => sum + t.visionCalls, 0),
|
|
22
|
-
visionCostUsd: tests.reduce((sum, t) => sum + t.visionCostUsd, 0),
|
|
44
|
+
visionCalls: tests.reduce((sum, t) => sum + t.visionCalls, 0) + extraCalls.length,
|
|
45
|
+
visionCostUsd: tests.reduce((sum, t) => sum + t.visionCostUsd, 0) + extraVisionCost,
|
|
46
|
+
visionTokens,
|
|
23
47
|
sandboxSeconds: tests.reduce((sum, t) => sum + t.sandboxSeconds, 0),
|
|
24
48
|
budgetExceeded: tests.some((t) => t.budgetExceeded),
|
|
25
49
|
callsByModel,
|
|
@@ -32,6 +56,7 @@ export function buildRunReport(tests, startedAt, durationMs) {
|
|
|
32
56
|
assertionMisses: tests.reduce((sum, t) => sum + (t.cache?.assertionMisses ?? 0), 0),
|
|
33
57
|
},
|
|
34
58
|
tests,
|
|
59
|
+
...(runNonce !== undefined ? { runNonce } : {}),
|
|
35
60
|
artifacts: {
|
|
36
61
|
videos: tests.map((t) => t.videoPath).filter((p) => p !== undefined),
|
|
37
62
|
},
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
import { type BudgetSummary, type CacheSummary, type HeadBinding, type LaneId, type LaneManifest, type LaneStatus, type RunManifest, type UsageSummary } from './manifest.js';
|
|
2
|
+
/**
|
|
3
|
+
* Manifest → view model. One contract for the three evidence surfaces —
|
|
4
|
+
* sticky PR comment, TUI, and Electron dashboard — so lane names, status
|
|
5
|
+
* labels, model/cost fields, and head identity agree by construction, not
|
|
6
|
+
* by convention (R15, AE-C).
|
|
7
|
+
*
|
|
8
|
+
* Everything here is a pure function over an already-parsed RunManifest:
|
|
9
|
+
* no fs, no fetch — the collector owns reading, this owns shaping.
|
|
10
|
+
*/
|
|
11
|
+
/** One display label per status — surfaces may color it, never rename it. */
|
|
12
|
+
export declare const LANE_STATUS_LABEL: Record<LaneStatus, string>;
|
|
13
|
+
/** Status glyph per lane status; no two statuses share one. */
|
|
14
|
+
export declare const STATUS_GLYPH: Record<LaneStatus, string>;
|
|
15
|
+
/** Proof strength, weakest to strongest (A4 ladder). */
|
|
16
|
+
export declare const PROOF_LEVELS: readonly ["suspected", "corroborated", "exercised", "reproduced"];
|
|
17
|
+
export type ProofLevel = (typeof PROOF_LEVELS)[number];
|
|
18
|
+
/**
|
|
19
|
+
* Four-notch meter: one filled notch per ladder step reached. Any other
|
|
20
|
+
* value (an evidence status outside the ladder, a missing level) is the
|
|
21
|
+
* empty meter, so a renderer never throws on unexpected input.
|
|
22
|
+
*/
|
|
23
|
+
export declare function proofMeter(level: string | undefined): string;
|
|
24
|
+
/** Finding severities as the review pipeline emits them (`q` is a question). */
|
|
25
|
+
export declare const SEVERITIES: readonly ["bug", "risk", "nit", "q"];
|
|
26
|
+
export type Severity = (typeof SEVERITIES)[number];
|
|
27
|
+
/** Geometric, shape-only severity glyphs (A5). */
|
|
28
|
+
export declare const SEVERITY_GLYPH: Record<Severity, string>;
|
|
29
|
+
export declare const SEVERITY_LABEL: Record<Severity, string>;
|
|
30
|
+
/** Review verdicts as the synthesis step emits them. */
|
|
31
|
+
export declare const VERDICTS: readonly ["approve", "needs_changes", "pass"];
|
|
32
|
+
export type Verdict = (typeof VERDICTS)[number];
|
|
33
|
+
/** Verdicts reuse status glyphs, so there is no third vocabulary. */
|
|
34
|
+
export declare const VERDICT_STATUS: Record<Verdict, LaneStatus>;
|
|
35
|
+
export declare const VERDICT_LABEL: Record<Verdict, string>;
|
|
36
|
+
export declare function verdictGlyph(verdict: Verdict): string;
|
|
37
|
+
export interface LaneView {
|
|
38
|
+
lane: LaneId;
|
|
39
|
+
selected: boolean;
|
|
40
|
+
status: LaneStatus;
|
|
41
|
+
statusLabel: string;
|
|
42
|
+
statusIcon: string;
|
|
43
|
+
summary: string | undefined;
|
|
44
|
+
reason: string | undefined;
|
|
45
|
+
reportPath: string | undefined;
|
|
46
|
+
model: string | undefined;
|
|
47
|
+
usage: UsageSummary;
|
|
48
|
+
budget: BudgetSummary;
|
|
49
|
+
cache: CacheSummary | undefined;
|
|
50
|
+
headBinding: HeadBinding | undefined;
|
|
51
|
+
startedAt: string | undefined;
|
|
52
|
+
finishedAt: string | undefined;
|
|
53
|
+
durationMs: number | undefined;
|
|
54
|
+
}
|
|
55
|
+
export interface RunView {
|
|
56
|
+
runId: string;
|
|
57
|
+
schemaVersion: number;
|
|
58
|
+
startedAt: string;
|
|
59
|
+
finishedAt: string;
|
|
60
|
+
status: LaneStatus;
|
|
61
|
+
statusLabel: string;
|
|
62
|
+
statusIcon: string;
|
|
63
|
+
ok: boolean;
|
|
64
|
+
costUsd: number;
|
|
65
|
+
calls: number;
|
|
66
|
+
tokens: number;
|
|
67
|
+
repo: string | undefined;
|
|
68
|
+
pr: string | undefined;
|
|
69
|
+
intendedHeadSha: string | undefined;
|
|
70
|
+
checkoutSha: string | undefined;
|
|
71
|
+
/** Review lane's head binding — the run-level binding contract. */
|
|
72
|
+
headBinding: HeadBinding | undefined;
|
|
73
|
+
/** All lanes in LANE_IDS order — selection state included. */
|
|
74
|
+
lanes: LaneView[];
|
|
75
|
+
/** Only the lanes that ran or were asked to — the matrix surfaces show. */
|
|
76
|
+
selectedLanes: LaneView[];
|
|
77
|
+
}
|
|
78
|
+
export declare function laneView(lane: LaneManifest): LaneView;
|
|
79
|
+
export declare function manifestToRunView(manifest: RunManifest): RunView;
|
|
80
|
+
/** Structural validation — a manifest the view-model can trust enough to render. */
|
|
81
|
+
export declare function isRunManifest(value: unknown): value is RunManifest;
|
|
82
|
+
/**
|
|
83
|
+
* One lane record the view-model can render. Exported so a surface that
|
|
84
|
+
* degrades per lane (the HTML report) applies the same check as the whole-
|
|
85
|
+
* manifest guard.
|
|
86
|
+
*/
|
|
87
|
+
export declare function isLaneManifest(value: unknown, id: LaneId): value is LaneManifest;
|
|
88
|
+
export declare function maskSecrets(s: string): string;
|
|
89
|
+
export declare function formatUsd(n: number | undefined): string;
|
|
90
|
+
export declare function formatDuration(ms: number | undefined): string;
|
|
91
|
+
export declare function shortSha(sha: string | undefined): string | undefined;
|
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
import { LANE_IDS, LANE_STATUSES, } from './manifest.js';
|
|
2
|
+
/**
|
|
3
|
+
* Manifest → view model. One contract for the three evidence surfaces —
|
|
4
|
+
* sticky PR comment, TUI, and Electron dashboard — so lane names, status
|
|
5
|
+
* labels, model/cost fields, and head identity agree by construction, not
|
|
6
|
+
* by convention (R15, AE-C).
|
|
7
|
+
*
|
|
8
|
+
* Everything here is a pure function over an already-parsed RunManifest:
|
|
9
|
+
* no fs, no fetch — the collector owns reading, this owns shaping.
|
|
10
|
+
*/
|
|
11
|
+
/** One display label per status — surfaces may color it, never rename it. */
|
|
12
|
+
export const LANE_STATUS_LABEL = {
|
|
13
|
+
passed: 'passed',
|
|
14
|
+
failed: 'failed',
|
|
15
|
+
skipped: 'skipped',
|
|
16
|
+
blocked: 'blocked',
|
|
17
|
+
unavailable: 'unavailable',
|
|
18
|
+
inconclusive: 'inconclusive',
|
|
19
|
+
};
|
|
20
|
+
/*
|
|
21
|
+
* Ocellus vocabulary (DESIGN.md section 6.7, A4, A5). Text surfaces use these
|
|
22
|
+
* glyphs: each is a single-cell, text-presentation code point, never emoji.
|
|
23
|
+
* A status is always shown as glyph plus its LANE_STATUS_LABEL word.
|
|
24
|
+
* `action/sticky-comment.cjs` keeps its own copy, guarded by parity tests.
|
|
25
|
+
*/
|
|
26
|
+
/** Status glyph per lane status; no two statuses share one. */
|
|
27
|
+
export const STATUS_GLYPH = {
|
|
28
|
+
passed: '●',
|
|
29
|
+
failed: '⊘',
|
|
30
|
+
skipped: '–',
|
|
31
|
+
blocked: '⊖',
|
|
32
|
+
unavailable: '◌',
|
|
33
|
+
inconclusive: '◐',
|
|
34
|
+
};
|
|
35
|
+
/** Proof strength, weakest to strongest (A4 ladder). */
|
|
36
|
+
export const PROOF_LEVELS = ['suspected', 'corroborated', 'exercised', 'reproduced'];
|
|
37
|
+
const PROOF_NOTCH_FILLED = '▰';
|
|
38
|
+
const PROOF_NOTCH_EMPTY = '▱';
|
|
39
|
+
/**
|
|
40
|
+
* Four-notch meter: one filled notch per ladder step reached. Any other
|
|
41
|
+
* value (an evidence status outside the ladder, a missing level) is the
|
|
42
|
+
* empty meter, so a renderer never throws on unexpected input.
|
|
43
|
+
*/
|
|
44
|
+
export function proofMeter(level) {
|
|
45
|
+
const filled = PROOF_LEVELS.indexOf(level ?? '') + 1;
|
|
46
|
+
return PROOF_NOTCH_FILLED.repeat(filled) + PROOF_NOTCH_EMPTY.repeat(PROOF_LEVELS.length - filled);
|
|
47
|
+
}
|
|
48
|
+
/** Finding severities as the review pipeline emits them (`q` is a question). */
|
|
49
|
+
export const SEVERITIES = ['bug', 'risk', 'nit', 'q'];
|
|
50
|
+
/** Geometric, shape-only severity glyphs (A5). */
|
|
51
|
+
export const SEVERITY_GLYPH = {
|
|
52
|
+
bug: '◆',
|
|
53
|
+
risk: '◈',
|
|
54
|
+
nit: '○',
|
|
55
|
+
q: '□',
|
|
56
|
+
};
|
|
57
|
+
export const SEVERITY_LABEL = {
|
|
58
|
+
bug: 'bug',
|
|
59
|
+
risk: 'risk',
|
|
60
|
+
nit: 'nit',
|
|
61
|
+
q: 'question',
|
|
62
|
+
};
|
|
63
|
+
/** Review verdicts as the synthesis step emits them. */
|
|
64
|
+
export const VERDICTS = ['approve', 'needs_changes', 'pass'];
|
|
65
|
+
/** Verdicts reuse status glyphs, so there is no third vocabulary. */
|
|
66
|
+
export const VERDICT_STATUS = {
|
|
67
|
+
approve: 'passed',
|
|
68
|
+
needs_changes: 'failed',
|
|
69
|
+
pass: 'passed',
|
|
70
|
+
};
|
|
71
|
+
export const VERDICT_LABEL = {
|
|
72
|
+
approve: 'approve',
|
|
73
|
+
needs_changes: 'needs changes',
|
|
74
|
+
pass: 'clean',
|
|
75
|
+
};
|
|
76
|
+
export function verdictGlyph(verdict) {
|
|
77
|
+
return STATUS_GLYPH[VERDICT_STATUS[verdict]];
|
|
78
|
+
}
|
|
79
|
+
function laneDurationMs(lane) {
|
|
80
|
+
if (lane.startedAt === undefined || lane.finishedAt === undefined)
|
|
81
|
+
return undefined;
|
|
82
|
+
const ms = Date.parse(lane.finishedAt) - Date.parse(lane.startedAt);
|
|
83
|
+
return Number.isFinite(ms) && ms >= 0 ? ms : undefined;
|
|
84
|
+
}
|
|
85
|
+
export function laneView(lane) {
|
|
86
|
+
return {
|
|
87
|
+
lane: lane.lane,
|
|
88
|
+
selected: lane.selected,
|
|
89
|
+
status: lane.status,
|
|
90
|
+
statusLabel: LANE_STATUS_LABEL[lane.status],
|
|
91
|
+
statusIcon: STATUS_GLYPH[lane.status],
|
|
92
|
+
summary: lane.summary,
|
|
93
|
+
reason: lane.reason,
|
|
94
|
+
reportPath: lane.reportPath,
|
|
95
|
+
model: lane.model ?? lane.usage?.model,
|
|
96
|
+
usage: lane.usage,
|
|
97
|
+
budget: lane.budget,
|
|
98
|
+
cache: lane.cache,
|
|
99
|
+
headBinding: lane.headBinding,
|
|
100
|
+
startedAt: lane.startedAt,
|
|
101
|
+
finishedAt: lane.finishedAt,
|
|
102
|
+
durationMs: laneDurationMs(lane),
|
|
103
|
+
};
|
|
104
|
+
}
|
|
105
|
+
export function manifestToRunView(manifest) {
|
|
106
|
+
const lanes = LANE_IDS.map((id) => laneView(manifest.lanes[id]));
|
|
107
|
+
return {
|
|
108
|
+
runId: manifest.runId,
|
|
109
|
+
schemaVersion: manifest.schemaVersion,
|
|
110
|
+
startedAt: manifest.startedAt,
|
|
111
|
+
finishedAt: manifest.finishedAt,
|
|
112
|
+
status: manifest.aggregate.status,
|
|
113
|
+
statusLabel: LANE_STATUS_LABEL[manifest.aggregate.status],
|
|
114
|
+
statusIcon: STATUS_GLYPH[manifest.aggregate.status],
|
|
115
|
+
ok: manifest.aggregate.ok,
|
|
116
|
+
costUsd: manifest.aggregate.costUsd,
|
|
117
|
+
calls: manifest.aggregate.calls,
|
|
118
|
+
tokens: manifest.aggregate.tokens,
|
|
119
|
+
repo: manifest.identity.repo,
|
|
120
|
+
pr: manifest.identity.pr,
|
|
121
|
+
intendedHeadSha: manifest.identity.intendedHeadSha,
|
|
122
|
+
checkoutSha: manifest.identity.checkoutSha,
|
|
123
|
+
headBinding: manifest.lanes.review.headBinding,
|
|
124
|
+
lanes,
|
|
125
|
+
selectedLanes: lanes.filter((lane) => lane.selected),
|
|
126
|
+
};
|
|
127
|
+
}
|
|
128
|
+
/** Structural validation — a manifest the view-model can trust enough to render. */
|
|
129
|
+
export function isRunManifest(value) {
|
|
130
|
+
if (value === null || typeof value !== 'object' || Array.isArray(value))
|
|
131
|
+
return false;
|
|
132
|
+
const m = value;
|
|
133
|
+
if (m.schemaVersion !== 1)
|
|
134
|
+
return false;
|
|
135
|
+
if (typeof m.runId !== 'string' || typeof m.startedAt !== 'string')
|
|
136
|
+
return false;
|
|
137
|
+
if (m.identity === undefined ||
|
|
138
|
+
typeof m.identity !== 'object' ||
|
|
139
|
+
m.identity === null ||
|
|
140
|
+
Array.isArray(m.identity)) {
|
|
141
|
+
return false;
|
|
142
|
+
}
|
|
143
|
+
for (const v of [
|
|
144
|
+
m.identity.repo,
|
|
145
|
+
m.identity.pr,
|
|
146
|
+
m.identity.intendedHeadSha,
|
|
147
|
+
m.identity.checkoutSha,
|
|
148
|
+
m.identity.baseSha,
|
|
149
|
+
m.identity.runNonce,
|
|
150
|
+
]) {
|
|
151
|
+
if (v !== undefined && typeof v !== 'string')
|
|
152
|
+
return false;
|
|
153
|
+
}
|
|
154
|
+
const aggregate = m.aggregate;
|
|
155
|
+
if (aggregate === undefined || aggregate === null || typeof aggregate !== 'object') {
|
|
156
|
+
return false;
|
|
157
|
+
}
|
|
158
|
+
// aggregate.status feeds LANE_STATUS_LABEL lookups and ok the verdict —
|
|
159
|
+
// a type-confused aggregate must degrade to last-valid, not reach a
|
|
160
|
+
// renderer that throws mid-post.
|
|
161
|
+
if (typeof aggregate.status !== 'string')
|
|
162
|
+
return false;
|
|
163
|
+
if (!LANE_STATUSES.includes(aggregate.status))
|
|
164
|
+
return false;
|
|
165
|
+
if (typeof aggregate.ok !== 'boolean')
|
|
166
|
+
return false;
|
|
167
|
+
for (const n of [aggregate.calls, aggregate.tokens, aggregate.costUsd]) {
|
|
168
|
+
if (typeof n !== 'number' || !Number.isFinite(n))
|
|
169
|
+
return false;
|
|
170
|
+
}
|
|
171
|
+
if (m.lanes === undefined || m.lanes === null || typeof m.lanes !== 'object')
|
|
172
|
+
return false;
|
|
173
|
+
return LANE_IDS.every((id) => isLaneManifest(m.lanes[id], id));
|
|
174
|
+
}
|
|
175
|
+
/**
|
|
176
|
+
* One lane record the view-model can render. Exported so a surface that
|
|
177
|
+
* degrades per lane (the HTML report) applies the same check as the whole-
|
|
178
|
+
* manifest guard.
|
|
179
|
+
*/
|
|
180
|
+
export function isLaneManifest(value, id) {
|
|
181
|
+
if (value === undefined || value === null || typeof value !== 'object')
|
|
182
|
+
return false;
|
|
183
|
+
const lane = value;
|
|
184
|
+
if (lane.lane !== id || typeof lane.status !== 'string')
|
|
185
|
+
return false;
|
|
186
|
+
if (!LANE_STATUSES.includes(lane.status))
|
|
187
|
+
return false;
|
|
188
|
+
if (typeof lane.selected !== 'boolean')
|
|
189
|
+
return false;
|
|
190
|
+
// usage/budget are dereferenced by laneView — a guard that certifies a
|
|
191
|
+
// shape it doesn't check is a lying guard. `typeof null === 'object'`,
|
|
192
|
+
// so a null here must be rejected before the field reads, not crash
|
|
193
|
+
// inside the guard.
|
|
194
|
+
if (lane.usage === null || lane.usage === undefined || typeof lane.usage !== 'object') {
|
|
195
|
+
return false;
|
|
196
|
+
}
|
|
197
|
+
for (const n of [lane.usage.calls, lane.usage.tokens, lane.usage.costUsd]) {
|
|
198
|
+
if (typeof n !== 'number' || !Number.isFinite(n))
|
|
199
|
+
return false;
|
|
200
|
+
}
|
|
201
|
+
if (lane.budget === null || lane.budget === undefined || typeof lane.budget !== 'object') {
|
|
202
|
+
return false;
|
|
203
|
+
}
|
|
204
|
+
return true;
|
|
205
|
+
}
|
|
206
|
+
/**
|
|
207
|
+
* Evidence strings pass through a secret-shaped-token mask before render —
|
|
208
|
+
* a lane summary that captured a credential must not re-emit it onto a PR
|
|
209
|
+
* comment or dashboard (R16's sanitized-evidence surface).
|
|
210
|
+
*/
|
|
211
|
+
const SECRET_PATTERNS = [
|
|
212
|
+
/sk-or-[A-Za-z0-9_-]{4,}/g,
|
|
213
|
+
/sk-[A-Za-z0-9_-]{8,}/g,
|
|
214
|
+
/gh[pousr]_[A-Za-z0-9_]{8,}/g,
|
|
215
|
+
/github_pat_[A-Za-z0-9_]{8,}/g,
|
|
216
|
+
/xox[baprs]-[A-Za-z0-9-]{8,}/g,
|
|
217
|
+
/AKIA[A-Z0-9]{16}/g,
|
|
218
|
+
/npm_[A-Za-z0-9]{8,}/g,
|
|
219
|
+
/Bearer\s+[A-Za-z0-9._~+/=-]{10,}/gi,
|
|
220
|
+
/eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{5,}/g,
|
|
221
|
+
/:\/\/[^/\s:@]{1,64}:[^/\s:@]{6,}@/g,
|
|
222
|
+
];
|
|
223
|
+
export function maskSecrets(s) {
|
|
224
|
+
let out = s;
|
|
225
|
+
for (const re of SECRET_PATTERNS)
|
|
226
|
+
out = out.replace(re, '•••');
|
|
227
|
+
return out;
|
|
228
|
+
}
|
|
229
|
+
export function formatUsd(n) {
|
|
230
|
+
return `$${(n ?? 0).toFixed(6)}`;
|
|
231
|
+
}
|
|
232
|
+
export function formatDuration(ms) {
|
|
233
|
+
if (ms === undefined || !Number.isFinite(ms))
|
|
234
|
+
return '–';
|
|
235
|
+
if (ms < 1000)
|
|
236
|
+
return `${Math.round(ms)}ms`;
|
|
237
|
+
return `${(ms / 1000).toFixed(1)}s`;
|
|
238
|
+
}
|
|
239
|
+
export function shortSha(sha) {
|
|
240
|
+
return sha === undefined || sha === '' ? undefined : sha.slice(0, 7);
|
|
241
|
+
}
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import { DecisionClient } from '../vision/decisions.js';
|
|
2
2
|
/**
|
|
3
|
-
* U8 finding adjudication — one batched
|
|
3
|
+
* U8 finding adjudication — one batched confidence-model decide() scores each
|
|
4
4
|
* synthesized finding's true-positive probability before posting. Same
|
|
5
|
-
* posture as the secrets lane:
|
|
5
|
+
* posture as the secrets lane: the confidence model annotates/routes, never gates.
|
|
6
6
|
* `p` lands on the finding record and the sticky confidence column;
|
|
7
7
|
* suppression is scoped to `nit`/`q` severities whose false-positive
|
|
8
8
|
* confidence exceeds `findingThreshold` — `bug`/`risk` are never
|
|
9
|
-
* suppressed, and a
|
|
9
|
+
* suppressed, and a confidence-model outage leaves every finding unadjudicated and
|
|
10
10
|
* unsuppressed (degrade open).
|
|
11
11
|
*
|
|
12
12
|
* `findingThreshold` is the required P(false positive): a nit/q is
|
|
@@ -28,7 +28,7 @@ export interface FindingAdjudicationRecord {
|
|
|
28
28
|
/** Finding text — capped; suppressed findings keep their message in audit. */
|
|
29
29
|
message: string;
|
|
30
30
|
adjudicated: boolean;
|
|
31
|
-
/**
|
|
31
|
+
/** Confidence-model P(true positive) for this finding. */
|
|
32
32
|
p?: number;
|
|
33
33
|
/** nit/q below the FP bar — kept for audit, removed from findings. */
|
|
34
34
|
suppressed?: boolean;
|
|
@@ -49,13 +49,13 @@ export interface FindingAdjudicationResult<T extends AdjudicableFinding> extends
|
|
|
49
49
|
}
|
|
50
50
|
export declare function adjudicateFindings<T extends AdjudicableFinding>(opts: {
|
|
51
51
|
findings: T[];
|
|
52
|
-
/** PR file patches keyed by filename —
|
|
52
|
+
/** PR file patches keyed by filename — confidence-model state context. */
|
|
53
53
|
patchByFile?: Map<string, string>;
|
|
54
54
|
threshold: number;
|
|
55
55
|
/**
|
|
56
56
|
* User-configured blocking severities — a severity listed here drives
|
|
57
57
|
* the verdict, so it must never be suppressed even if it is nit/q
|
|
58
|
-
* (otherwise
|
|
58
|
+
* (otherwise confidence-model suppression could flip the commit-status gate).
|
|
59
59
|
*/
|
|
60
60
|
blockSeverities?: string[];
|
|
61
61
|
client: DecisionClient;
|
|
@@ -63,11 +63,11 @@ export async function adjudicateFindings(opts) {
|
|
|
63
63
|
}
|
|
64
64
|
catch (e) {
|
|
65
65
|
adjudicationFailed = true;
|
|
66
|
-
debug('adjudicate', `decision call failed
|
|
66
|
+
debug('adjudicate', `decision call failed – no suppression: ${describeDecisionError(e)}`);
|
|
67
67
|
}
|
|
68
68
|
}
|
|
69
69
|
const blocking = new Set(opts.blockSeverities ?? []);
|
|
70
|
-
// Only
|
|
70
|
+
// Only the confidence model may attach p — a model-emitted p on an unadjudicated
|
|
71
71
|
// finding is spoofed confidence, so strip it.
|
|
72
72
|
const stripP = (f) => {
|
|
73
73
|
const out = { ...f };
|