@open-cr-agent/core 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/pipeline/output.d.ts +1 -0
- package/dist/pipeline/output.js +2 -0
- package/dist/pipeline/report.d.ts +8 -1
- package/dist/pipeline/report.js +6 -3
- package/dist/pipeline/run.js +48 -15
- package/dist/pipeline/usage.d.ts +1 -0
- package/dist/pipeline/usage.js +6 -0
- package/dist/plugin/types.d.ts +11 -0
- package/package.json +1 -1
package/dist/pipeline/output.js
CHANGED
|
@@ -29,6 +29,8 @@ export function toReportOutput(report) {
|
|
|
29
29
|
output.judgement = report.judgement;
|
|
30
30
|
if (report.anchoring)
|
|
31
31
|
output.anchoring = report.anchoring;
|
|
32
|
+
if (report.spendLimit)
|
|
33
|
+
output.spendLimit = report.spendLimit;
|
|
32
34
|
if (report.rereview) {
|
|
33
35
|
const r = report.rereview;
|
|
34
36
|
output.rereview = {
|
|
@@ -13,7 +13,10 @@ export type CoverageEntry = {
|
|
|
13
13
|
status: "excluded";
|
|
14
14
|
reason: ExclusionReason;
|
|
15
15
|
};
|
|
16
|
-
export declare function coverageGaps(
|
|
16
|
+
export declare function coverageGaps(run: {
|
|
17
|
+
coverage: readonly CoverageEntry[];
|
|
18
|
+
tasks: readonly Pick<TaskOutcome, "status">[];
|
|
19
|
+
}): {
|
|
17
20
|
notReviewed: number;
|
|
18
21
|
nothingReviewed: boolean;
|
|
19
22
|
};
|
|
@@ -60,6 +63,10 @@ export interface ReviewReport {
|
|
|
60
63
|
dismissed: PriorFinding[];
|
|
61
64
|
};
|
|
62
65
|
anchoring?: AnchoringSummary;
|
|
66
|
+
spendLimit?: {
|
|
67
|
+
usd: number;
|
|
68
|
+
reached?: "review" | "total";
|
|
69
|
+
};
|
|
63
70
|
usage: Usage;
|
|
64
71
|
warnings: string[];
|
|
65
72
|
}
|
package/dist/pipeline/report.js
CHANGED
|
@@ -1,11 +1,14 @@
|
|
|
1
1
|
// How much of the selection this run actually reviewed. Every surface (exit
|
|
2
2
|
// code, terminal, pull request summary) reads it from here, so none of them
|
|
3
3
|
// can call a run that reviewed nothing "approved".
|
|
4
|
-
export function coverageGaps(
|
|
5
|
-
const notReviewed = coverage.filter((c) => c.status === "failed" || c.status === "unreviewed").length;
|
|
4
|
+
export function coverageGaps(run) {
|
|
5
|
+
const notReviewed = run.coverage.filter((c) => c.status === "failed" || c.status === "unreviewed").length;
|
|
6
6
|
// Unchanged files were reviewed by an earlier run, and their findings and
|
|
7
7
|
// verdict carry over: a re-review whose new files all failed still has them.
|
|
8
|
-
|
|
8
|
+
// A file stays failed while any of its reviewers failed, so a reviewer that
|
|
9
|
+
// finished its tasks has still reviewed something (#263).
|
|
10
|
+
const reviewed = run.coverage.some((c) => c.status === "reviewed" || c.status === "unchanged") ||
|
|
11
|
+
run.tasks.some((t) => t.status === "completed");
|
|
9
12
|
return { notReviewed, nothingReviewed: notReviewed > 0 && !reviewed };
|
|
10
13
|
}
|
|
11
14
|
export function summarizeAnchoring(findings, relocationCalls) {
|
package/dist/pipeline/run.js
CHANGED
|
@@ -13,7 +13,7 @@ import { DEFAULT_MAX_TASKS, planTasks } from "./matrix.js";
|
|
|
13
13
|
import { planReview } from "./plan.js";
|
|
14
14
|
import { mapWithConcurrency } from "./pool.js";
|
|
15
15
|
import { coverageGaps, summarizeAnchoring, } from "./report.js";
|
|
16
|
-
import { addUsage, emptyUsage } from "./usage.js";
|
|
16
|
+
import { addUsage, emptyUsage, unpricedCalls } from "./usage.js";
|
|
17
17
|
export { GUIDELINES_PATH } from "./plan.js";
|
|
18
18
|
const DEFAULTS = { concurrency: 4, taskTimeoutMs: 10 * 60_000, runTimeoutMs: 25 * 60_000 };
|
|
19
19
|
// Plan (deterministic stages) → execute (one agent task per cell) → report.
|
|
@@ -74,21 +74,31 @@ export async function runReview(options) {
|
|
|
74
74
|
onUsage: spend,
|
|
75
75
|
signal: AbortSignal.any([signal, spendLimit.signal]),
|
|
76
76
|
};
|
|
77
|
+
// Tasks that never started leave their files unreviewed, not failed.
|
|
78
|
+
const notStarted = new Set();
|
|
79
|
+
let unaffordable = 0;
|
|
77
80
|
const results = await mapWithConcurrency(matrix.cells, options.concurrency ?? DEFAULTS.concurrency, async (cell) => {
|
|
78
81
|
// Once the run is cancelled or timed out, remaining cells are not started.
|
|
79
|
-
if (signal.aborted)
|
|
82
|
+
if (signal.aborted) {
|
|
83
|
+
notStarted.add(cell.taskId);
|
|
80
84
|
return skipCell(cell, "run cancelled before this task started", emit);
|
|
85
|
+
}
|
|
81
86
|
if (budget.reviewExhausted()) {
|
|
87
|
+
notStarted.add(cell.taskId);
|
|
88
|
+
unaffordable += 1;
|
|
82
89
|
return skipCell(cell, `spend limit of $${options.maxCostUsd} reached`, emit);
|
|
83
90
|
}
|
|
84
91
|
return runJob(cell, plan, execute);
|
|
85
92
|
});
|
|
93
|
+
if (unaffordable > 0) {
|
|
94
|
+
plan.warnings.push(`spend limit of $${options.maxCostUsd} reached: ${unaffordable} review task(s) did not start; their files are reported as not reviewed`);
|
|
95
|
+
}
|
|
86
96
|
// Memory and the previous review filter first, so Verify and Judge are not
|
|
87
97
|
// paid for findings that will not be reported, and a person's dismissal
|
|
88
98
|
// keeps a finding out of the verdict whatever the models say.
|
|
89
99
|
const concurrency = options.concurrency ?? DEFAULTS.concurrency;
|
|
90
100
|
const found = dedupeFindings(results.flatMap((r) => r.findings));
|
|
91
|
-
const fileCoverage = coverage(plan.decisions, results, plan.unchanged, matrix.limited ?? []);
|
|
101
|
+
const fileCoverage = coverage(plan.decisions, results, plan.unchanged, matrix.limited ?? [], notStarted);
|
|
92
102
|
const remembered = applyMemory(found, plan.memory);
|
|
93
103
|
const reported = new Set(found.map((f) => f.fingerprint));
|
|
94
104
|
const priorReview = withoutRemembered(prior.review, plan.memory);
|
|
@@ -137,12 +147,22 @@ export async function runReview(options) {
|
|
|
137
147
|
if (judgeWanted && !judgeAffordable) {
|
|
138
148
|
judged.warnings.push(`spend limit of $${options.maxCostUsd} reached: findings were not judged`);
|
|
139
149
|
}
|
|
140
|
-
const { nothingReviewed } = coverageGaps(
|
|
150
|
+
const { nothingReviewed } = coverageGaps({
|
|
151
|
+
coverage: fileCoverage,
|
|
152
|
+
tasks: results.map((r) => r.outcome),
|
|
153
|
+
});
|
|
141
154
|
if (!nothingReviewed) {
|
|
142
155
|
emit(judged.decisions
|
|
143
156
|
? { type: "judge_finished", verdict: judged.verdict, judgement: judged.decisions }
|
|
144
157
|
: { type: "judge_finished", verdict: judged.verdict });
|
|
145
158
|
}
|
|
159
|
+
const calls = [
|
|
160
|
+
...plan.usage,
|
|
161
|
+
...results.map((r) => r.usage),
|
|
162
|
+
...relocationUsage,
|
|
163
|
+
...verification.usage,
|
|
164
|
+
...judged.usage,
|
|
165
|
+
];
|
|
146
166
|
const report = {
|
|
147
167
|
changeRequest: plan.changeRequest,
|
|
148
168
|
tier: plan.tier,
|
|
@@ -160,14 +180,9 @@ export async function runReview(options) {
|
|
|
160
180
|
unverifiedCriticals: countMissedCriticals(verification.kept, verification.missed),
|
|
161
181
|
refuted: verification.refuted,
|
|
162
182
|
remembered: remembered.remembered,
|
|
163
|
-
usage: sumUsage(
|
|
164
|
-
...plan.usage,
|
|
165
|
-
...results.map((r) => r.usage),
|
|
166
|
-
...relocationUsage,
|
|
167
|
-
...verification.usage,
|
|
168
|
-
...judged.usage,
|
|
169
|
-
]),
|
|
183
|
+
usage: sumUsage(calls),
|
|
170
184
|
warnings: [
|
|
185
|
+
...unpricedWarning(calls),
|
|
171
186
|
...plan.warnings,
|
|
172
187
|
...results.flatMap((r) => r.warnings),
|
|
173
188
|
...verification.warnings,
|
|
@@ -176,6 +191,14 @@ export async function runReview(options) {
|
|
|
176
191
|
};
|
|
177
192
|
if (judged.decisions)
|
|
178
193
|
report.judgement = judged.decisions;
|
|
194
|
+
if (options.maxCostUsd !== undefined) {
|
|
195
|
+
const reached = budget.exhausted()
|
|
196
|
+
? "total"
|
|
197
|
+
: spendLimit.signal.aborted || unaffordable > 0
|
|
198
|
+
? "review"
|
|
199
|
+
: undefined;
|
|
200
|
+
report.spendLimit = { usd: options.maxCostUsd, ...(reached ? { reached } : {}) };
|
|
201
|
+
}
|
|
179
202
|
report.anchoring = summarizeAnchoring(report.findings, relocationUsage.length);
|
|
180
203
|
if (prior.review) {
|
|
181
204
|
report.rereview = {
|
|
@@ -247,18 +270,20 @@ async function loadPriorReview(vcs) {
|
|
|
247
270
|
return { warning: `could not load the previous review: ${errorMessage(error)}` };
|
|
248
271
|
}
|
|
249
272
|
}
|
|
250
|
-
function coverage(decisions, results, unchanged, limited) {
|
|
273
|
+
function coverage(decisions, results, unchanged, limited, notStarted) {
|
|
251
274
|
// A file is reviewed when every reviewer assigned to it finished at least
|
|
252
275
|
// one of its tasks: under --ultra one completed sample is enough. A
|
|
253
|
-
// reviewer whose task failed makes the file "failed"; one
|
|
254
|
-
//
|
|
276
|
+
// reviewer whose task failed makes the file "failed"; one whose task never
|
|
277
|
+
// started (the task or spend limit, a cancelled run) makes it "unreviewed".
|
|
255
278
|
const done = new Map();
|
|
256
279
|
const ran = new Set();
|
|
257
280
|
const key = (reviewer, file) => `${reviewer}\0${file}`;
|
|
258
281
|
for (const { outcome } of results) {
|
|
282
|
+
const started = !notStarted.has(outcome.taskId);
|
|
259
283
|
for (const file of outcome.files) {
|
|
260
284
|
const k = key(outcome.reviewer, file);
|
|
261
|
-
|
|
285
|
+
if (started)
|
|
286
|
+
ran.add(k);
|
|
262
287
|
done.set(k, done.get(k) === true || outcome.status === "completed");
|
|
263
288
|
}
|
|
264
289
|
}
|
|
@@ -294,6 +319,14 @@ function sortFindings(findings) {
|
|
|
294
319
|
a.file.localeCompare(b.file) ||
|
|
295
320
|
(a.lineRange?.start ?? 0) - (b.lineRange?.start ?? 0));
|
|
296
321
|
}
|
|
322
|
+
function unpricedWarning(calls) {
|
|
323
|
+
const unpriced = unpricedCalls(calls);
|
|
324
|
+
return unpriced > 0
|
|
325
|
+
? [
|
|
326
|
+
`${unpriced} model call(s) used tokens but reported no cost: their model has no price, so the reported cost and the spend limit do not count them`,
|
|
327
|
+
]
|
|
328
|
+
: [];
|
|
329
|
+
}
|
|
297
330
|
function sumUsage(usages) {
|
|
298
331
|
return usages.reduce(addUsage, emptyUsage());
|
|
299
332
|
}
|
package/dist/pipeline/usage.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { Usage } from "../contracts.js";
|
|
2
2
|
export declare function emptyUsage(): Usage;
|
|
3
3
|
export declare function addUsage(total: Usage, next: Usage): Usage;
|
|
4
|
+
export declare function unpricedCalls(usages: readonly Usage[]): number;
|
|
4
5
|
//# sourceMappingURL=usage.d.ts.map
|
package/dist/pipeline/usage.js
CHANGED
|
@@ -10,4 +10,10 @@ export function addUsage(total, next) {
|
|
|
10
10
|
costUsd: total.costUsd + next.costUsd,
|
|
11
11
|
};
|
|
12
12
|
}
|
|
13
|
+
// Calls that used tokens but cost nothing: their model has no price (a
|
|
14
|
+
// declared model priced at 0, or one missing from the pricing catalog), so
|
|
15
|
+
// reported cost and the spend limit cannot count them.
|
|
16
|
+
export function unpricedCalls(usages) {
|
|
17
|
+
return usages.filter((u) => u.inputTokens + u.outputTokens > 0 && u.costUsd === 0).length;
|
|
18
|
+
}
|
|
13
19
|
//# sourceMappingURL=usage.js.map
|
package/dist/plugin/types.d.ts
CHANGED
|
@@ -11,6 +11,17 @@ export interface RuntimeOptions {
|
|
|
11
11
|
models: ModelChains;
|
|
12
12
|
tools: readonly ToolDefinition[];
|
|
13
13
|
env: Env;
|
|
14
|
+
providers?: Readonly<Record<string, CustomProvider>>;
|
|
15
|
+
}
|
|
16
|
+
export interface CustomProvider {
|
|
17
|
+
baseUrl: string;
|
|
18
|
+
apiKeyEnv?: string;
|
|
19
|
+
models: Readonly<Record<string, ModelPrice>>;
|
|
20
|
+
}
|
|
21
|
+
export interface ModelPrice {
|
|
22
|
+
input: number;
|
|
23
|
+
output: number;
|
|
24
|
+
cachedInput?: number;
|
|
14
25
|
}
|
|
15
26
|
export type VcsFactory = (options: unknown) => VcsAdapter;
|
|
16
27
|
export type RuntimeFactory = (options: RuntimeOptions) => AgentRuntime;
|