jev-agent-tools 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/LICENSE +21 -0
- package/README.md +119 -0
- package/docs/design.md +187 -0
- package/docs/tools/jev_ask.md +92 -0
- package/docs/tools/jev_ask_files.md +80 -0
- package/docs/tools/jev_check_diff.md +77 -0
- package/docs/tools/jev_find_files.md +59 -0
- package/docs/tools/jev_locate_in_file.md +51 -0
- package/docs/tools/jev_select_tests.md +59 -0
- package/package.json +85 -0
- package/rules/jev-ask.md +4 -0
- package/src/adapters/analysis-context.ts +100 -0
- package/src/adapters/ask-files.ts +236 -0
- package/src/adapters/ask-proof.ts +202 -0
- package/src/adapters/ask-syntax.ts +463 -0
- package/src/adapters/command.ts +240 -0
- package/src/adapters/docs.ts +222 -0
- package/src/adapters/files.ts +411 -0
- package/src/adapters/find.ts +151 -0
- package/src/adapters/git-base.ts +32 -0
- package/src/adapters/git-inventory.ts +94 -0
- package/src/adapters/git.ts +525 -0
- package/src/adapters/locate-file.ts +197 -0
- package/src/adapters/output-lines.ts +50 -0
- package/src/adapters/risk-callers.ts +525 -0
- package/src/adapters/runner-version.ts +102 -0
- package/src/adapters/syntax.ts +229 -0
- package/src/adapters/test-inventory.ts +168 -0
- package/src/adapters/usage.ts +21 -0
- package/src/adapters/utf8.ts +57 -0
- package/src/constants.ts +109 -0
- package/src/core/ask-closure.ts +419 -0
- package/src/core/ask-proof.ts +32 -0
- package/src/core/ask-references.ts +249 -0
- package/src/core/asks.ts +616 -0
- package/src/core/batches.ts +83 -0
- package/src/core/command-output.ts +249 -0
- package/src/core/diff.ts +226 -0
- package/src/core/docs.ts +399 -0
- package/src/core/find.ts +157 -0
- package/src/core/git.ts +5 -0
- package/src/core/import-boundaries.ts +102 -0
- package/src/core/imports.ts +691 -0
- package/src/core/integrity.ts +64 -0
- package/src/core/lexical.ts +130 -0
- package/src/core/locate.ts +213 -0
- package/src/core/output.ts +264 -0
- package/src/core/pointer.ts +51 -0
- package/src/core/risk-callers.ts +1270 -0
- package/src/core/runner-version.ts +66 -0
- package/src/core/sections.ts +269 -0
- package/src/core/state.ts +53 -0
- package/src/core/syntax.ts +8 -0
- package/src/core/test-commands.ts +430 -0
- package/src/core/test-coverage.ts +103 -0
- package/src/core/test-discovery.ts +1695 -0
- package/src/core/test-evidence.ts +649 -0
- package/src/core/test-state.ts +99 -0
- package/src/core/truncate.ts +14 -0
- package/src/core/units.ts +531 -0
- package/src/describe.ts +26 -0
- package/src/guide.ts +42 -0
- package/src/host.ts +22 -0
- package/src/index.ts +40 -0
- package/src/jev/client.ts +505 -0
- package/src/jev/pool.ts +60 -0
- package/src/jev/types.ts +60 -0
- package/src/presets/docs.ts +85 -0
- package/src/presets/risk.ts +263 -0
- package/src/presets/spec.ts +111 -0
- package/src/presets/witnesses.ts +313 -0
- package/src/render.ts +45 -0
- package/src/result.ts +4 -0
- package/src/run-end.ts +157 -0
- package/src/runtime.ts +16 -0
- package/src/session.ts +106 -0
- package/src/texts/ask-files.ts +2 -0
- package/src/texts/ask.ts +4 -0
- package/src/texts/check-diff.ts +24 -0
- package/src/texts/configuration.ts +2 -0
- package/src/texts/find.ts +14 -0
- package/src/texts/guide.ts +16 -0
- package/src/texts/locate.ts +10 -0
- package/src/texts/run-end.ts +13 -0
- package/src/texts/select-tests.ts +3 -0
- package/src/tools/ask-files.ts +263 -0
- package/src/tools/ask-schema.ts +116 -0
- package/src/tools/ask.ts +925 -0
- package/src/tools/check-diff.ts +510 -0
- package/src/tools/docs-check.ts +399 -0
- package/src/tools/find.ts +529 -0
- package/src/tools/locate.ts +369 -0
- package/src/tools/select-tests.ts +746 -0
- package/src/tools/spec-check.ts +210 -0
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
import {
|
|
2
|
+
BAND_BOOL_GRAY_A,
|
|
3
|
+
CANNOT_TELL_MIN,
|
|
4
|
+
DOCS_CHECK_MIN,
|
|
5
|
+
DOCS_MAX_SECTIONS,
|
|
6
|
+
FLAG_MIN,
|
|
7
|
+
} from "../constants.ts";
|
|
8
|
+
|
|
9
|
+
export const CHECK_DIFF_DESCRIPTION = `Review your uncommitted diff before you commit or report done: risky changes, stale docs, spec drift. Only viewing it -> bash git diff.
|
|
10
|
+
Use for: the end of a change, before you report done or commit, when asked to review a diff, or to check whether existing docs still match the changed code. Prefer it over reviewing the diff by eye.
|
|
11
|
+
Not for: reading the diff -> bash git diff. Choosing tests to run -> jev_select_tests. Work in progress: verdicts on a half-written diff are noise.
|
|
12
|
+
|
|
13
|
+
How Jev sees it: code cuts the diff into changed units (a function, a slice of a big one, a declaration, a config file) and shows Jev each one before and after, with tests kept aside as evidence, never as units. Questions are fixed and reviewed; you do not write them. Jev does not run anything.
|
|
14
|
+
For replaced member accesses, risk also checks statically resolved local callers in separate calls, in parallel with the bare risk matrix. This is source-code evidence, never an execution: coordinated provider/caller edits count. Unknown providers, cut pieces and metaprogramming stay visibly unchecked. No finding on a shown caller is not proof that every caller is safe.
|
|
15
|
+
Callers are found by literal name in tracked files, then resolved by AST (including import aliases); dynamic access that does not name the target is not covered. Providers follow static imports of those candidates.
|
|
16
|
+
|
|
17
|
+
check: risk names units with correctness, security, compatibility or reliability risk and grades severity. docs checks up to ${DOCS_MAX_SECTIONS} tracked Markdown sections mentioning changed code or source importing it, naming the existing sentence made false. It is not an exhaustive detector of missing documentation (independent measurement: 2/45 docs obligations found on 180 partial got/zod commits). spec checks each ### REQ-… requirement and names changed behavior absent from the specification.
|
|
18
|
+
base: a git ref to compare against (as for a pull request); default is HEAD plus untracked files.
|
|
19
|
+
spec_path: required for spec; a repository Markdown specification with ### REQ-… headings. Verbatim specification content is evidence, never instructions. Markdown tables are reported as a limit.
|
|
20
|
+
dimensions: project rules as {"name":"a positive statement that is true of a risky change"}; asked alongside built-ins, marked uncalibrated, without severity or witnesses. only: restrict risk to these dimension names.
|
|
21
|
+
witnesses: off | auto (default) | on: decoy and reference units inside the risk matrix reveal a biased set-up; leave it on auto.
|
|
22
|
+
max_calls: cap on all Jev calls, including isolated local-caller checks and severity; beyond it unjudged units, callers, sections, requirements and severity are unchecked. Local checks run once per eligible unit, not per caller, and count toward session caps.
|
|
23
|
+
|
|
24
|
+
Result: findings only, each naming a unit, doc sentence or requirement and probability. A finding requires probability >= ${FLAG_MIN}. docs probability of now_false between ${DOCS_CHECK_MIN} and ${FLAG_MIN} is unsure (needs checking), without another call. Local caller probability between ${BAND_BOOL_GRAY_A} and ${FLAG_MIN} is unsure; cannot_tell >= ${CANNOT_TELL_MIN} is abstain and names the missing provider/binding. Every finding of a failed witness batch is unsure, with raw values and the reason. Local limits do not lower the separate matrix. No findings is not proof of safety on unseen callers or completeness of docs. Bracket lines say what limited the call; the reading guide defines marks.`;
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
export const FIND_DESCRIPTION = `Find the files for a goal you describe in plain words, ranked; known names → {{names.byName}}, known strings → {{names.grep}}.
|
|
2
|
+
Use for: starting a task in unfamiliar code ("where is proration computed", "what handles webhook retries") when you do not know file names or exact identifiers.
|
|
3
|
+
Not for: known strings, regexes or symbols → {{names.grep}}. Files by name → {{names.byName}}. Questions about files you already have → jev_ask_files.
|
|
4
|
+
|
|
5
|
+
How Jev sees it: it ranks paths by name first, then reads a short passage of the best ones, chosen by your keywords, and picks the entry point among the few that survive. It sees passages, not whole files, and a bare path says little: a full sentence of behavior finds the right file far more often than a few keywords. It cannot count or grep; an identifier you know goes in keywords.
|
|
6
|
+
|
|
7
|
+
goal: what you are trying to do or find, as one sentence of at least {{FIND_GOAL_MIN_WORDS}} words. Describe behavior, not a guessed identifier: in our tests a full sentence found the right file in 49 of 51 cases, two or three keywords alone in 16 of 21.
|
|
8
|
+
keywords: identifiers or terms you already know that should appear in the code; [] if none. They steer the pre-filter and the passage Jev reads.
|
|
9
|
+
scope: directories to search instead of the whole repo. A wrong scope returns "none", not a wrong file.
|
|
10
|
+
exclude: globs to skip. Excluding tests keeps the entry point but drops the tests and docs from the list.
|
|
11
|
+
effort: quick for a first look, default, thorough for a cleaner list of related files (the entry point is rarely better).
|
|
12
|
+
max_calls: cap on Jev calls; beyond it the search stops at the best-ranked files so far and says so.
|
|
13
|
+
|
|
14
|
+
Result: entry: the file to open first with its probability, then the ranked list (probability that the file helps with the goal), and the read to make next. entry: unsure means two files are plausible, the answer depends on the order of the candidates or on a short goal: read the first two. entry: none means nothing fit: rephrase the goal or widen scope. A goal under {{FIND_GOAL_MIN_WORDS}} content words is capped at unsure and the line says so. You receive paths and scores, never file contents. Bracket lines say what limited the call. The reading guide defines every mark once.`;
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
export const GUIDE_TEMPLATE = `jev_* tools: how to read their output.
|
|
2
|
+
|
|
3
|
+
Jev reads what you pass at a glance and answers with a probability. A line with no mark is a verdict: a lead to check before an irreversible action, not a proof. About 1 in 100 clear verdicts is wrong, and far more when the evidence is cut, from the wrong file, or depends on a file that was not passed; no mark can see that, so check the evidence yourself before you edit, delete or report done.
|
|
4
|
+
|
|
5
|
+
Marks:
|
|
6
|
+
- unsure: Jev does not see it clearly (yes/no between {{BAND_BOOL_GRAY_A}} and {{BAND_BOOL_YES_MIN}}; a category or level under {{BAND_CHOICE_VERDICT_MIN}} confidence; findings between their thresholds), or a control of the call failed and the line says which. Read the passage or the file the line points to. Adding more context afterwards or rewording the question does not help; adding the file that settles it does.
|
|
7
|
+
- abstain: the piece is missing from what you passed. The line names it. Add it (paths or command) and ask once.
|
|
8
|
+
- "no (not shown)" or "not addressed": the file or state does not show it. That is not "false".
|
|
9
|
+
- uncalibrated: no error rate has been measured for this kind of ask; read it as a hint.
|
|
10
|
+
- Lines in brackets: what limited the call and what to do next; take the parameter they name.
|
|
11
|
+
|
|
12
|
+
Last line: calls · questions · cost · cache · time. Use max_calls to bound a wide ask.
|
|
13
|
+
|
|
14
|
+
In your final answer, identify each claim or conclusion marked unsure or abstain by a jev_* tool. Say explicitly that Jev did not confirm it, and explain what evidence is still needed. Do not present an unsure or abstained result as an established fact. If you subsequently settled it by reading the decisive evidence, distinguish your own verification from Jev's result and cite that evidence. Otherwise keep the conclusion explicitly unconfirmed, including in your summary and recommended action.
|
|
15
|
+
|
|
16
|
+
Ask about facts the files show, in positive sentences, with the evidence attached; do the counting and searching with {{names.grep}} or code.`;
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
export const LOCATE_DESCRIPTION = `Find which line range of ONE big file (over {{LOCATE_MIN_KB}} KB) serves a goal, instead of reading it whole; known string or symbol -> {{names.grep}}.
|
|
2
|
+
Use for: a file too big to read whole (over {{LOCATE_MIN_KB}} KB) when you know what you need from it ("where are retries scheduled", "the part that parses flags"). Prefer it over reading the file in chunks.
|
|
3
|
+
Not for: an exact string or symbol → {{names.grep}}. Files under {{LOCATE_MIN_KB}} KB → read them directly. Several files → jev_ask_files or {{names.semantic}}.{{ompOutline}}
|
|
4
|
+
|
|
5
|
+
How Jev sees it: code cuts the file into declarations or sections, and Jev reads them all at a glance and picks the one that serves the goal. Several sections often share the answer (a getter and its setter), so the mass splits and the pick looks unsure even when both are right. It reads what is written, not what a function does at run time.
|
|
6
|
+
|
|
7
|
+
path: one repo-relative file.
|
|
8
|
+
goal: what you need from the file, as a sentence.
|
|
9
|
+
|
|
10
|
+
Result: the range as path:start-end with its label and probability, and the read call to make next. An unmarked line above {{LOCATE_VERDICT_MIN}} is a verdict: read that range. unsure ({{LOCATE_GRAY_MIN}} to {{LOCATE_VERDICT_MIN}}) lists the two best by probability, often adjacent code that shares the answer, not the next section of the file: read both. Below {{LOCATE_GRAY_MIN}} Jev is asked once more among the three best plus none; the original unsure band is retained. A "none" means no section fits and the goal is probably not in this file: search elsewhere. Bracket lines say what limited the call. The reading guide defines every mark once.`;
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import type { DocsFinding } from "../presets/docs.ts";
|
|
2
|
+
|
|
3
|
+
/** Only flagged, designated sentences are supplied by the hook. */
|
|
4
|
+
export function runEndMessage(findings: readonly DocsFinding[]): string {
|
|
5
|
+
return [
|
|
6
|
+
"jev_check_diff (docs) flagged documentation that may no longer match your changes:",
|
|
7
|
+
...findings.map(
|
|
8
|
+
(finding) =>
|
|
9
|
+
`- ${finding.section.path} § ${finding.section.heading} — ${JSON.stringify(finding.sentence?.text)} no longer true after ${finding.units.map((unit) => `${unit.name} in ${unit.file}`).join(", ")}`,
|
|
10
|
+
),
|
|
11
|
+
"Update them, or say why they are still correct.",
|
|
12
|
+
].join("\n");
|
|
13
|
+
}
|
|
@@ -0,0 +1,3 @@
|
|
|
1
|
+
import { SELECT_FILE_SHARE, SELECT_MIN } from "../constants.ts";
|
|
2
|
+
|
|
3
|
+
export const SELECT_TESTS_DESCRIPTION = `Select existing tests affected by a diff and return runner commands; this tool never runs or collects tests. Use after editing code, before choosing a test subset; use native read/grep/bash for exact names and execution. base defaults to HEAD; paths are candidate test files or globs, not diff paths. Static discovery reads tracked files, literal project configuration and imports, without evaluating third-party config. Touched tests are selected for free; import closure includes unchanged intermediates and applicable pytest conftest fixtures. Remaining scenarios receive a changed-unit pointer; selected when 1-p(none) >= ${SELECT_MIN}. Missing answers are selected; unavailable Jev falls back to all. Residual exported-unit checks report changed, run by no discovered test only within the discovered inventory, never global coverage or an obligation to add a scenario. Unsupported, unknown, calculated or unresolved runners keep that conclusion unsure and name the next manual action. Runtime and type tests are separate. Commands remain separate by cwd, config, project and framework; project options are preserved. At least ${SELECT_FILE_SHARE * 100}% selected, uncertain names or unknown counts run the entire file. max_calls bounds Jev calls; unjudged tests remain selected. witnesses controls residual coverage checks (auto by default); pointers have no witnesses. This does not detect that a new test scenario is required.`;
|
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import { resolve } from "node:path";
|
|
3
|
+
import type { ToolDefinition } from "@earendil-works/pi-coding-agent";
|
|
4
|
+
import { type Static, Type } from "@sinclair/typebox";
|
|
5
|
+
import { collectAskFiles } from "../adapters/ask-files.ts";
|
|
6
|
+
import { shareGitInventory } from "../adapters/git-inventory.ts";
|
|
7
|
+
import { hostUsage } from "../adapters/usage.ts";
|
|
8
|
+
import { compileAsks, readAsks, reverseQuestions } from "../core/asks.ts";
|
|
9
|
+
import { checkIntegrity } from "../core/integrity.ts";
|
|
10
|
+
import type {
|
|
11
|
+
AnswerInput,
|
|
12
|
+
BudgetRefusal,
|
|
13
|
+
Envelope,
|
|
14
|
+
EnvelopeInput,
|
|
15
|
+
} from "../core/output.ts";
|
|
16
|
+
import { buildEnvelope } from "../core/output.ts";
|
|
17
|
+
import { interpolate } from "../describe.ts";
|
|
18
|
+
import type { GuideContext } from "../guide.ts";
|
|
19
|
+
import type { Answer, Judgment, JudgmentMetadata } from "../jev/types.ts";
|
|
20
|
+
import { renderEnvelope } from "../render.ts";
|
|
21
|
+
import type { ToolDependencies } from "../runtime.ts";
|
|
22
|
+
import { ASK_FILES_DESCRIPTION } from "../texts/ask-files.ts";
|
|
23
|
+
import { NOT_CONFIGURED } from "../texts/configuration.ts";
|
|
24
|
+
import { asksParameter } from "./ask-schema.ts";
|
|
25
|
+
|
|
26
|
+
export const askFilesParameters = Type.Object(
|
|
27
|
+
{
|
|
28
|
+
paths: Type.Array(Type.String({ minLength: 1 }), { minItems: 1 }),
|
|
29
|
+
asks: asksParameter("files"),
|
|
30
|
+
max_calls: Type.Optional(Type.Integer({ minimum: 0 })),
|
|
31
|
+
},
|
|
32
|
+
{ additionalProperties: false },
|
|
33
|
+
);
|
|
34
|
+
export type AskFilesArgs = Static<typeof askFilesParameters>;
|
|
35
|
+
interface FileJudgment {
|
|
36
|
+
path: string;
|
|
37
|
+
initial: Judgment;
|
|
38
|
+
reversed?: Judgment;
|
|
39
|
+
}
|
|
40
|
+
export interface AskFilesDetails {
|
|
41
|
+
files: FileJudgment[];
|
|
42
|
+
skipped: string[];
|
|
43
|
+
envelope: Envelope;
|
|
44
|
+
}
|
|
45
|
+
export function createAskFilesTool({
|
|
46
|
+
client,
|
|
47
|
+
host,
|
|
48
|
+
runtime,
|
|
49
|
+
exec: execute,
|
|
50
|
+
}: ToolDependencies) {
|
|
51
|
+
return {
|
|
52
|
+
name: "jev_ask_files",
|
|
53
|
+
label: "Jev ask files",
|
|
54
|
+
description: interpolate(ASK_FILES_DESCRIPTION, host.names),
|
|
55
|
+
parameters: askFilesParameters,
|
|
56
|
+
...(host.isOmp
|
|
57
|
+
? { approval: "read" as const, loadMode: "essential" as const }
|
|
58
|
+
: {
|
|
59
|
+
promptSnippet:
|
|
60
|
+
"Typed answers about many files from intents you declare, without reading them",
|
|
61
|
+
promptGuidelines: [
|
|
62
|
+
"jev_ask_files: put every ask in one call; declare intents (verify, classify, rate, decide), never raw questions",
|
|
63
|
+
"jev_ask_files: use it to decide what to read, then read only the files that matter",
|
|
64
|
+
],
|
|
65
|
+
}),
|
|
66
|
+
async execute(
|
|
67
|
+
_id: string,
|
|
68
|
+
args: AskFilesArgs,
|
|
69
|
+
signal: AbortSignal | undefined,
|
|
70
|
+
_update: unknown,
|
|
71
|
+
ctx: { cwd: string } & GuideContext,
|
|
72
|
+
) {
|
|
73
|
+
const started = performance.now();
|
|
74
|
+
const exec = shareGitInventory(execute);
|
|
75
|
+
const totals = {
|
|
76
|
+
calls: 0,
|
|
77
|
+
questions: 0,
|
|
78
|
+
cacheHits: 0,
|
|
79
|
+
cacheRequests: 0,
|
|
80
|
+
inputTokens: 0,
|
|
81
|
+
costUsd: 0,
|
|
82
|
+
};
|
|
83
|
+
const records: FileJudgment[] = [];
|
|
84
|
+
let skipped: string[] = [];
|
|
85
|
+
const finish = (input: Omit<EnvelopeInput, "yield">) => {
|
|
86
|
+
const envelope = buildEnvelope({
|
|
87
|
+
...input,
|
|
88
|
+
yield: { ...totals, elapsedMs: performance.now() - started },
|
|
89
|
+
});
|
|
90
|
+
runtime.session.record(envelope);
|
|
91
|
+
runtime.guide.deliver(ctx);
|
|
92
|
+
const usage = hostUsage(host.isOmp, {
|
|
93
|
+
inputTokens: totals.inputTokens,
|
|
94
|
+
costUsd: totals.costUsd,
|
|
95
|
+
});
|
|
96
|
+
return {
|
|
97
|
+
content: [{ type: "text" as const, text: renderEnvelope(envelope) }],
|
|
98
|
+
details: { files: records, skipped, envelope },
|
|
99
|
+
...usage,
|
|
100
|
+
};
|
|
101
|
+
};
|
|
102
|
+
if (!client) return finish({ refusal: NOT_CONFIGURED });
|
|
103
|
+
const plan = compileAsks(args.asks, { surface: "files" });
|
|
104
|
+
if (!plan.ok) return finish({ refusal: plan.error });
|
|
105
|
+
const collected = await collectAskFiles(
|
|
106
|
+
ctx.cwd,
|
|
107
|
+
args.paths,
|
|
108
|
+
signal,
|
|
109
|
+
exec,
|
|
110
|
+
);
|
|
111
|
+
if (!collected.ok) return finish({ refusal: collected.error });
|
|
112
|
+
skipped = collected.skipped;
|
|
113
|
+
if (!collected.files.length && skipped.length)
|
|
114
|
+
return finish({
|
|
115
|
+
refusal: `No readable evidence: ${skipped.join("; ")}`,
|
|
116
|
+
});
|
|
117
|
+
let sent = 0;
|
|
118
|
+
let budget: BudgetRefusal | undefined;
|
|
119
|
+
const options = {
|
|
120
|
+
signal,
|
|
121
|
+
beforeRequest: (count: number) => {
|
|
122
|
+
if (args.max_calls !== undefined && sent >= args.max_calls) {
|
|
123
|
+
budget = {
|
|
124
|
+
kind: "max_calls",
|
|
125
|
+
message: `max_calls=${args.max_calls} reached`,
|
|
126
|
+
};
|
|
127
|
+
return { ok: false as const, error: budget.message };
|
|
128
|
+
}
|
|
129
|
+
const admission = runtime.session.admit(count);
|
|
130
|
+
if (!admission.ok) {
|
|
131
|
+
budget = { kind: "session", message: admission.error };
|
|
132
|
+
return admission;
|
|
133
|
+
}
|
|
134
|
+
sent++;
|
|
135
|
+
return admission;
|
|
136
|
+
},
|
|
137
|
+
onUsage: (usage: { inputTokens: number; costUsd: number }) =>
|
|
138
|
+
runtime.session.recordUsage(usage),
|
|
139
|
+
};
|
|
140
|
+
const accumulate = (result: JudgmentMetadata) => {
|
|
141
|
+
totals.calls += result.calls ?? 0;
|
|
142
|
+
totals.questions += result.questions ?? 0;
|
|
143
|
+
totals.cacheHits += result.cacheHits ?? 0;
|
|
144
|
+
totals.cacheRequests += result.cacheRequests ?? 0;
|
|
145
|
+
totals.inputTokens += result.usage?.inputTokens ?? 0;
|
|
146
|
+
totals.costUsd += result.usage?.costUsd ?? 0;
|
|
147
|
+
};
|
|
148
|
+
const rows = await Promise.all(
|
|
149
|
+
collected.files.map(async (file) => {
|
|
150
|
+
const state = { path: file.path, content: file.content };
|
|
151
|
+
const integrity = checkIntegrity(state, plan.asks, [
|
|
152
|
+
{
|
|
153
|
+
...file.identity,
|
|
154
|
+
insertedPath: resolve(ctx.cwd, state.path),
|
|
155
|
+
content: state.content,
|
|
156
|
+
insertedSha256: createHash("sha256")
|
|
157
|
+
.update(state.content)
|
|
158
|
+
.digest("hex"),
|
|
159
|
+
},
|
|
160
|
+
]);
|
|
161
|
+
const initial = await client.judge(state, plan.questions, {
|
|
162
|
+
...options,
|
|
163
|
+
groups: plan.groups,
|
|
164
|
+
});
|
|
165
|
+
accumulate(initial);
|
|
166
|
+
let reversed: Judgment | undefined;
|
|
167
|
+
let reverseAnswers: Record<string, Answer> | undefined;
|
|
168
|
+
if (initial.ok) {
|
|
169
|
+
const questions = reverseQuestions(plan, initial.answers);
|
|
170
|
+
if (Object.keys(questions).length) {
|
|
171
|
+
reversed = await client.judge(state, questions, options);
|
|
172
|
+
accumulate(reversed);
|
|
173
|
+
reverseAnswers = reversed.ok
|
|
174
|
+
? reversed.answers
|
|
175
|
+
: Object.fromEntries(
|
|
176
|
+
Object.keys(questions).map((id) => [
|
|
177
|
+
id,
|
|
178
|
+
{
|
|
179
|
+
type: "unjudged" as const,
|
|
180
|
+
reason:
|
|
181
|
+
reversed && !reversed.ok
|
|
182
|
+
? reversed.error
|
|
183
|
+
: "reverse control failed",
|
|
184
|
+
},
|
|
185
|
+
]),
|
|
186
|
+
);
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
const answers: AnswerInput[] = initial.ok
|
|
190
|
+
? readAsks(plan, initial.answers, reverseAnswers)
|
|
191
|
+
: [];
|
|
192
|
+
return { file, initial, reversed, integrity, answers };
|
|
193
|
+
}),
|
|
194
|
+
);
|
|
195
|
+
const answers: AnswerInput[] = [];
|
|
196
|
+
const unchecked: string[] = [];
|
|
197
|
+
const unjudged: NonNullable<EnvelopeInput["unjudged"]>[number][] = [];
|
|
198
|
+
for (const row of rows) {
|
|
199
|
+
records.push({
|
|
200
|
+
path: row.file.path,
|
|
201
|
+
initial: row.initial,
|
|
202
|
+
...(row.reversed ? { reversed: row.reversed } : {}),
|
|
203
|
+
});
|
|
204
|
+
if (!row.initial.ok) {
|
|
205
|
+
unchecked.push(row.file.path);
|
|
206
|
+
unjudged.push({
|
|
207
|
+
label: row.file.path,
|
|
208
|
+
reason: row.initial.error,
|
|
209
|
+
next: "read the file or retry",
|
|
210
|
+
});
|
|
211
|
+
continue;
|
|
212
|
+
}
|
|
213
|
+
if (
|
|
214
|
+
Object.values(row.initial.answers).some(
|
|
215
|
+
(answer) => answer.type === "unjudged",
|
|
216
|
+
) ||
|
|
217
|
+
(row.reversed &&
|
|
218
|
+
(!row.reversed.ok ||
|
|
219
|
+
Object.values(row.reversed.answers).some(
|
|
220
|
+
(answer) => answer.type === "unjudged",
|
|
221
|
+
)))
|
|
222
|
+
)
|
|
223
|
+
unchecked.push(row.file.path);
|
|
224
|
+
for (const answer of row.answers)
|
|
225
|
+
answers.push({
|
|
226
|
+
...answer,
|
|
227
|
+
label: `${row.file.path} ${answer.label}`,
|
|
228
|
+
...(row.integrity.controls.length
|
|
229
|
+
? {
|
|
230
|
+
band: "unsure" as const,
|
|
231
|
+
reason: row.integrity.controls
|
|
232
|
+
.map((control) => control.fact)
|
|
233
|
+
.join("; "),
|
|
234
|
+
}
|
|
235
|
+
: {}),
|
|
236
|
+
});
|
|
237
|
+
}
|
|
238
|
+
return finish({
|
|
239
|
+
state: {
|
|
240
|
+
files: collected.files.length,
|
|
241
|
+
integrity: rows.some((row) => row.integrity.controls.length)
|
|
242
|
+
? "failed"
|
|
243
|
+
: "ok",
|
|
244
|
+
warnings: plan.warnings.map((warning) => warning.fact),
|
|
245
|
+
},
|
|
246
|
+
answers,
|
|
247
|
+
unjudged,
|
|
248
|
+
unchecked,
|
|
249
|
+
budget,
|
|
250
|
+
lines: skipped.length
|
|
251
|
+
? [{ type: "list", title: "skipped", items: skipped }]
|
|
252
|
+
: [],
|
|
253
|
+
limitations: [
|
|
254
|
+
...plan.warnings,
|
|
255
|
+
...rows.flatMap((row) => [
|
|
256
|
+
...row.integrity.controls,
|
|
257
|
+
...row.integrity.notes,
|
|
258
|
+
]),
|
|
259
|
+
],
|
|
260
|
+
});
|
|
261
|
+
},
|
|
262
|
+
} satisfies ToolDefinition<typeof askFilesParameters, AskFilesDetails>;
|
|
263
|
+
}
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
import { Type } from "@sinclair/typebox";
|
|
2
|
+
import {
|
|
3
|
+
CHOICE_MAX_OPTIONS,
|
|
4
|
+
SCORE_MAX_LEVELS,
|
|
5
|
+
SCORE_MIN_LEVELS,
|
|
6
|
+
} from "../constants.ts";
|
|
7
|
+
|
|
8
|
+
const criteria = Type.Record(Type.String(), Type.String(), {
|
|
9
|
+
minProperties: 1,
|
|
10
|
+
maxProperties: CHOICE_MAX_OPTIONS,
|
|
11
|
+
});
|
|
12
|
+
const question = Type.Union([
|
|
13
|
+
Type.Object(
|
|
14
|
+
{
|
|
15
|
+
type: Type.Literal("bool"),
|
|
16
|
+
instructions: Type.String({ minLength: 1 }),
|
|
17
|
+
criteria: Type.Optional(
|
|
18
|
+
Type.Object(
|
|
19
|
+
{
|
|
20
|
+
true: Type.Optional(Type.String()),
|
|
21
|
+
false: Type.Optional(Type.String()),
|
|
22
|
+
},
|
|
23
|
+
{ additionalProperties: false },
|
|
24
|
+
),
|
|
25
|
+
),
|
|
26
|
+
},
|
|
27
|
+
{ additionalProperties: false },
|
|
28
|
+
),
|
|
29
|
+
Type.Object(
|
|
30
|
+
{
|
|
31
|
+
type: Type.Literal("choice"),
|
|
32
|
+
instructions: Type.String({ minLength: 1 }),
|
|
33
|
+
criteria,
|
|
34
|
+
},
|
|
35
|
+
{ additionalProperties: false },
|
|
36
|
+
),
|
|
37
|
+
Type.Object(
|
|
38
|
+
{
|
|
39
|
+
type: Type.Literal("score"),
|
|
40
|
+
instructions: Type.String({ minLength: 1 }),
|
|
41
|
+
criteria: Type.Array(Type.String(), {
|
|
42
|
+
minItems: SCORE_MIN_LEVELS,
|
|
43
|
+
maxItems: SCORE_MAX_LEVELS,
|
|
44
|
+
}),
|
|
45
|
+
},
|
|
46
|
+
{ additionalProperties: false },
|
|
47
|
+
),
|
|
48
|
+
]);
|
|
49
|
+
const fullAskSchema = Type.Union(
|
|
50
|
+
createAskVariants({ about: Type.Optional(Type.String({ minLength: 1 })) }),
|
|
51
|
+
);
|
|
52
|
+
const filesAskSchema = Type.Union(createAskVariants({}).slice(0, -1));
|
|
53
|
+
export function askSchema<S extends "ask" | "files">(
|
|
54
|
+
surface: S,
|
|
55
|
+
): S extends "ask" ? typeof fullAskSchema : typeof filesAskSchema;
|
|
56
|
+
export function askSchema(surface: "ask" | "files") {
|
|
57
|
+
return surface === "ask" ? fullAskSchema : filesAskSchema;
|
|
58
|
+
}
|
|
59
|
+
function createAskVariants<A extends object>(about: A) {
|
|
60
|
+
const variants = [
|
|
61
|
+
Type.Object(
|
|
62
|
+
{ intent: Type.Literal("free"), ...about, question },
|
|
63
|
+
{ additionalProperties: false },
|
|
64
|
+
),
|
|
65
|
+
Type.Object(
|
|
66
|
+
{ intent: Type.Literal("verify"), ...about, claims: criteria },
|
|
67
|
+
{ additionalProperties: false },
|
|
68
|
+
),
|
|
69
|
+
Type.Object(
|
|
70
|
+
{
|
|
71
|
+
intent: Type.Literal("classify"),
|
|
72
|
+
...about,
|
|
73
|
+
categories: criteria,
|
|
74
|
+
pick: Type.Optional(
|
|
75
|
+
Type.Union([Type.Literal("one"), Type.Literal("many")]),
|
|
76
|
+
),
|
|
77
|
+
by: Type.Optional(Type.String()),
|
|
78
|
+
},
|
|
79
|
+
{ additionalProperties: false },
|
|
80
|
+
),
|
|
81
|
+
Type.Object(
|
|
82
|
+
{ intent: Type.Literal("decide"), ...about, hypotheses: criteria },
|
|
83
|
+
{ additionalProperties: false },
|
|
84
|
+
),
|
|
85
|
+
Type.Object(
|
|
86
|
+
{
|
|
87
|
+
intent: Type.Literal("rate"),
|
|
88
|
+
...about,
|
|
89
|
+
dimension: Type.String(),
|
|
90
|
+
levels: Type.Array(Type.String(), {
|
|
91
|
+
minItems: SCORE_MIN_LEVELS,
|
|
92
|
+
maxItems: SCORE_MAX_LEVELS,
|
|
93
|
+
}),
|
|
94
|
+
},
|
|
95
|
+
{ additionalProperties: false },
|
|
96
|
+
),
|
|
97
|
+
Type.Object(
|
|
98
|
+
{
|
|
99
|
+
intent: Type.Literal("locate"),
|
|
100
|
+
...about,
|
|
101
|
+
target: Type.String(),
|
|
102
|
+
among: Type.Array(Type.String(), { minItems: 1 }),
|
|
103
|
+
count: Type.Optional(
|
|
104
|
+
Type.Union([Type.Literal("one"), Type.Literal("many")]),
|
|
105
|
+
),
|
|
106
|
+
attribution: Type.Optional(Type.Boolean()),
|
|
107
|
+
},
|
|
108
|
+
{ additionalProperties: false },
|
|
109
|
+
),
|
|
110
|
+
];
|
|
111
|
+
return variants;
|
|
112
|
+
}
|
|
113
|
+
export function asksParameter<S extends "ask" | "files">(surface: S) {
|
|
114
|
+
const ask = askSchema(surface);
|
|
115
|
+
return Type.Union([ask, Type.Array(ask, { minItems: 1 }), Type.String()]);
|
|
116
|
+
}
|