@itookit/dsht 0.3.7 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.i18n.yaml +2 -2
- package/README.md +31 -11
- package/README.zh.md +31 -11
- package/dist/cli/dsht.js +207 -19
- package/dist/cli/startup.d.ts +40 -0
- package/dist/cli/startup.js +295 -0
- package/dist/cli/trace-summary.d.ts +78 -0
- package/dist/cli/trace-summary.js +241 -0
- package/dist/cli/verifier.d.ts +60 -0
- package/dist/cli/verifier.js +242 -0
- package/dist/contracts.d.ts +344 -0
- package/dist/contracts.js +1 -0
- package/dist/controller/commands.d.ts +47 -0
- package/dist/controller/commands.js +322 -0
- package/dist/controller/connection.d.ts +11 -29
- package/dist/controller/connection.js +26 -60
- package/dist/controller/controller.d.ts +619 -164
- package/dist/controller/controller.js +1420 -141
- package/dist/controller/index.d.ts +8 -1
- package/dist/controller/index.js +5 -0
- package/dist/controller/loop-contract.d.ts +136 -0
- package/dist/controller/loop-contract.js +308 -0
- package/dist/controller/loop-prompts-schema.d.ts +56 -0
- package/dist/controller/loop-prompts-schema.js +144 -0
- package/dist/controller/loop-prompts.d.ts +55 -0
- package/dist/controller/loop-prompts.generated.d.ts +104 -0
- package/dist/controller/loop-prompts.generated.js +185 -0
- package/dist/controller/loop-prompts.js +104 -0
- package/dist/controller/loop-protocols.d.ts +39 -0
- package/dist/controller/loop-protocols.js +115 -0
- package/dist/controller/loop.d.ts +275 -0
- package/dist/controller/loop.js +378 -0
- package/dist/controller/prompts.d.ts +54 -0
- package/dist/controller/prompts.js +162 -0
- package/dist/controller/trace-log.d.ts +45 -0
- package/dist/controller/trace-log.js +144 -0
- package/dist/controller/verifier.d.ts +126 -0
- package/dist/controller/verifier.js +75 -0
- package/dist/cost/index.d.ts +1 -1
- package/dist/cost/index.js +1 -1
- package/dist/cost/ledger.d.ts +0 -1
- package/dist/cost/ledger.js +0 -1
- package/dist/json.d.ts +18 -0
- package/dist/json.js +19 -0
- package/dist/references.d.ts +25 -0
- package/dist/references.js +26 -0
- package/dist/session/connection-view.d.ts +2 -11
- package/dist/session/controller.d.ts +82 -72
- package/dist/session/controller.js +211 -209
- package/dist/session/history.d.ts +9 -1
- package/dist/session/history.js +1 -9
- package/dist/session/index.d.ts +9 -4
- package/dist/session/index.js +7 -3
- package/dist/session/info.d.ts +25 -52
- package/dist/session/info.js +39 -25
- package/dist/session/markdown.js +1 -1
- package/dist/session/math.js +1 -1
- package/dist/session/mutation-gate.d.ts +51 -0
- package/dist/session/mutation-gate.js +73 -0
- package/dist/session/navigation.d.ts +2 -89
- package/dist/session/navigation.js +2 -129
- package/dist/session/peek.d.ts +38 -0
- package/dist/session/peek.js +103 -0
- package/dist/session/references.d.ts +2 -20
- package/dist/session/references.js +1 -26
- package/dist/session/runtime.d.ts +26 -0
- package/dist/session/runtime.js +28 -0
- package/dist/session/telemetry.d.ts +12 -13
- package/dist/session/telemetry.js +27 -58
- package/dist/session/transcript.d.ts +0 -6
- package/dist/session/transcript.js +2 -15
- package/dist/session/types.d.ts +25 -0
- package/dist/session/types.js +0 -1
- package/dist/session-title.d.ts +9 -0
- package/dist/session-title.js +21 -0
- package/dist/shell/controller.d.ts +97 -0
- package/dist/shell/controller.js +158 -0
- package/dist/shell/index.d.ts +5 -0
- package/dist/shell/index.js +3 -0
- package/dist/shell/runner.d.ts +38 -0
- package/dist/shell/runner.js +147 -0
- package/dist/slash/index.d.ts +10 -0
- package/dist/slash/index.js +7 -0
- package/dist/slash/parse.d.ts +166 -0
- package/dist/slash/parse.js +259 -0
- package/dist/slash/pipeline.d.ts +140 -0
- package/dist/slash/pipeline.js +115 -0
- package/dist/slash/registry.d.ts +88 -0
- package/dist/slash/registry.js +177 -0
- package/dist/state.d.ts +14 -4
- package/dist/state.js +3 -2
- package/dist/text.d.ts +28 -0
- package/dist/text.js +55 -0
- package/dist/transport/events.d.ts +104 -0
- package/dist/transport/events.js +149 -0
- package/dist/transport/wire.d.ts +9 -17
- package/dist/transport/wire.js +2 -27
- package/dist/ui/app.js +865 -431
- package/dist/ui/chat/header.js +1 -1
- package/dist/ui/chat/history-view.d.ts +1 -1
- package/dist/ui/chat/history-view.js +1 -1
- package/dist/ui/chat/loop-status.d.ts +11 -0
- package/dist/ui/chat/loop-status.js +28 -0
- package/dist/ui/chat/navigation-model.d.ts +86 -0
- package/dist/ui/chat/navigation-model.js +107 -0
- package/dist/ui/chat/shell-view.d.ts +47 -0
- package/dist/ui/chat/shell-view.js +145 -0
- package/dist/ui/chat/status.d.ts +47 -3
- package/dist/ui/chat/status.js +65 -50
- package/dist/ui/chat/viewport.d.ts +1 -1
- package/dist/ui/dialogs/cost.d.ts +21 -4
- package/dist/ui/dialogs/cost.js +7 -12
- package/dist/ui/dialogs/index.d.ts +22 -5
- package/dist/ui/dialogs/index.js +19 -3
- package/dist/ui/dialogs/loop.d.ts +43 -0
- package/dist/ui/dialogs/loop.js +224 -0
- package/dist/ui/dialogs/peek.d.ts +25 -0
- package/dist/ui/dialogs/peek.js +35 -0
- package/dist/ui/dialogs/picker.d.ts +2 -0
- package/dist/ui/dialogs/picker.js +4 -2
- package/dist/ui/input/mouse.d.ts +12 -2
- package/dist/ui/input/mouse.js +20 -7
- package/dist/ui/input/references.d.ts +1 -1
- package/dist/ui/status/model.d.ts +7 -0
- package/dist/ui/status/model.js +5 -0
- package/dist/ui/theme/index.d.ts +6 -1
- package/dist/ui/theme/index.js +2 -1
- package/package.json +6 -4
- package/dist/ui/commands/parse.d.ts +0 -99
- package/dist/ui/commands/parse.js +0 -126
- package/dist/ui/commands/registry.d.ts +0 -33
- package/dist/ui/commands/registry.js +0 -73
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import { type LoopProtocol } from './loop.ts';
|
|
2
|
+
import type { LoopRecord } from '../contracts.ts';
|
|
3
|
+
/** Names `/loop` may run, in file order. */
|
|
4
|
+
export declare function loopProtocolNames(): string[];
|
|
5
|
+
/** Every record `/loop` may run, in file order, with the defaults a run would start from.
|
|
6
|
+
*
|
|
7
|
+
* The list and the run read the same records, so a chooser can show exactly the name, round count,
|
|
8
|
+
* artifact and defaults the runner would use — never a second table that could drift.
|
|
9
|
+
* @returns One summary per record.
|
|
10
|
+
*/
|
|
11
|
+
export declare function loopRecords(): LoopRecord[];
|
|
12
|
+
/** One record's declared variables, so a caller can offer or validate them before a run starts.
|
|
13
|
+
* @param name - Record key in loop.yaml.
|
|
14
|
+
* @returns The record's `vars`, or undefined when no record has that name.
|
|
15
|
+
*/
|
|
16
|
+
export declare function loopRecordVars(name: string): Readonly<Record<string, string>> | undefined;
|
|
17
|
+
/** Rubric the verifier scores against: the round's own checklist plus the record's standard.
|
|
18
|
+
*
|
|
19
|
+
* The standard is part of the record, so it travels with the run instead of living in session
|
|
20
|
+
* state; a record without one scores on its checklist alone.
|
|
21
|
+
* @param checks - This round's checklist.
|
|
22
|
+
* @param standard - The record's extra requirements, when it declares any.
|
|
23
|
+
* @returns The rubric text.
|
|
24
|
+
*/
|
|
25
|
+
export declare function roundStandard(checks: string, standard?: string): string;
|
|
26
|
+
/** Build one record, or undefined when no record has that name.
|
|
27
|
+
*
|
|
28
|
+
* The record's data is read once here, so a run keeps the rubric, standard and vars it started
|
|
29
|
+
* with even if `loop.yaml` is edited (or reloaded) while it is in flight.
|
|
30
|
+
* @param name - Record key in loop.yaml.
|
|
31
|
+
* @param forked - Delegate each round's verdict to an independent verifier process.
|
|
32
|
+
* @param vars - Values that replace the record's own `vars` for this run, e.g. another document.
|
|
33
|
+
* @param selfScoring - Whether this client's own reply may decide an attempt. Without a forked
|
|
34
|
+
* verifier it always does; with one, only when the operator allowed the verifier's fallback to it.
|
|
35
|
+
* When it does not, the brief stops asking for a verdict block: nothing reads it, and a visible
|
|
36
|
+
* score that moves nothing is exactly what a reader mistakes for the real one.
|
|
37
|
+
* @returns The protocol the scored loop runs, or undefined for an unknown name.
|
|
38
|
+
*/
|
|
39
|
+
export declare function loopProtocolFor(name: string, forked?: boolean, vars?: Readonly<Record<string, string>>, selfScoring?: boolean): LoopProtocol | undefined;
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
/** Turn one loop.yaml record into the protocol the scored loop runs.
|
|
2
|
+
*
|
|
3
|
+
* The record is the whole protocol: title, steps, rubric, brief, follow-up, artifact and the extra
|
|
4
|
+
* standard. This module only wires that data to the mechanical contract in `loop-contract.ts`, so a
|
|
5
|
+
* new review protocol is a YAML edit — not a new TypeScript file — and `/loop <name>` can run any
|
|
6
|
+
* record without code knowing which one it is.
|
|
7
|
+
*/
|
|
8
|
+
import { LOOP_MARKER, followUpContract, resultContract, verdictBrief } from "./loop-contract.js";
|
|
9
|
+
import { loopPrompts } from "./loop-prompts.js";
|
|
10
|
+
import { coversWholeProtocol } from "./loop.js";
|
|
11
|
+
/** Names `/loop` may run, in file order. */
|
|
12
|
+
export function loopProtocolNames() {
|
|
13
|
+
return [...loopPrompts().names];
|
|
14
|
+
}
|
|
15
|
+
/** Every record `/loop` may run, in file order, with the defaults a run would start from.
|
|
16
|
+
*
|
|
17
|
+
* The list and the run read the same records, so a chooser can show exactly the name, round count,
|
|
18
|
+
* artifact and defaults the runner would use — never a second table that could drift.
|
|
19
|
+
* @returns One summary per record.
|
|
20
|
+
*/
|
|
21
|
+
export function loopRecords() {
|
|
22
|
+
return loopPrompts().names.flatMap(name => {
|
|
23
|
+
const text = loopPrompts().find(name);
|
|
24
|
+
if (text === undefined)
|
|
25
|
+
return [];
|
|
26
|
+
return [{ name, title: text.title, steps: text.steps,
|
|
27
|
+
...(text.artifact === undefined ? {} : { artifact: text.artifact }),
|
|
28
|
+
defaultScore: text.defaultScore, defaultTries: text.defaultTries,
|
|
29
|
+
vars: loopPrompts().vars(name) }];
|
|
30
|
+
});
|
|
31
|
+
}
|
|
32
|
+
/** One record's declared variables, so a caller can offer or validate them before a run starts.
|
|
33
|
+
* @param name - Record key in loop.yaml.
|
|
34
|
+
* @returns The record's `vars`, or undefined when no record has that name.
|
|
35
|
+
*/
|
|
36
|
+
export function loopRecordVars(name) {
|
|
37
|
+
return loopPrompts().find(name) === undefined ? undefined : loopPrompts().vars(name);
|
|
38
|
+
}
|
|
39
|
+
/** Rubric the verifier scores against: the round's own checklist plus the record's standard.
|
|
40
|
+
*
|
|
41
|
+
* The standard is part of the record, so it travels with the run instead of living in session
|
|
42
|
+
* state; a record without one scores on its checklist alone.
|
|
43
|
+
* @param checks - This round's checklist.
|
|
44
|
+
* @param standard - The record's extra requirements, when it declares any.
|
|
45
|
+
* @returns The rubric text.
|
|
46
|
+
*/
|
|
47
|
+
export function roundStandard(checks, standard) {
|
|
48
|
+
return standard === undefined ? checks : `${checks}\n\n附加要求(记录自带):\n${standard}`;
|
|
49
|
+
}
|
|
50
|
+
/** Build one record, or undefined when no record has that name.
|
|
51
|
+
*
|
|
52
|
+
* The record's data is read once here, so a run keeps the rubric, standard and vars it started
|
|
53
|
+
* with even if `loop.yaml` is edited (or reloaded) while it is in flight.
|
|
54
|
+
* @param name - Record key in loop.yaml.
|
|
55
|
+
* @param forked - Delegate each round's verdict to an independent verifier process.
|
|
56
|
+
* @param vars - Values that replace the record's own `vars` for this run, e.g. another document.
|
|
57
|
+
* @param selfScoring - Whether this client's own reply may decide an attempt. Without a forked
|
|
58
|
+
* verifier it always does; with one, only when the operator allowed the verifier's fallback to it.
|
|
59
|
+
* When it does not, the brief stops asking for a verdict block: nothing reads it, and a visible
|
|
60
|
+
* score that moves nothing is exactly what a reader mistakes for the real one.
|
|
61
|
+
* @returns The protocol the scored loop runs, or undefined for an unknown name.
|
|
62
|
+
*/
|
|
63
|
+
export function loopProtocolFor(name, forked = false, vars, selfScoring = !forked) {
|
|
64
|
+
const text = loopPrompts().find(name, vars);
|
|
65
|
+
if (text === undefined)
|
|
66
|
+
return undefined;
|
|
67
|
+
const standard = (step) => roundStandard(text.checks(step), text.standard);
|
|
68
|
+
const artifact = text.artifact;
|
|
69
|
+
// A run over the whole record ends on its consolidation round, and that round is the only place
|
|
70
|
+
// where "passed" may mean the whole artifact: the verifier is handed every earlier round to
|
|
71
|
+
// re-check, so a later round that broke an earlier requirement cannot pass unnoticed.
|
|
72
|
+
const consolidates = (limits, step) => coversWholeProtocol(limits.from, limits.to, text.steps) && step === text.steps;
|
|
73
|
+
const coverage = (step) => Array.from({ length: Math.max(0, step - 1) }, (_, index) => ({ title: `第 ${index + 1} 轮 · ${text.roundTitle(index + 1)}`, checks: text.checks(index + 1) }));
|
|
74
|
+
return {
|
|
75
|
+
marker: LOOP_MARKER,
|
|
76
|
+
kind: name,
|
|
77
|
+
title: text.title,
|
|
78
|
+
steps: text.steps,
|
|
79
|
+
...(artifact === undefined ? {} : { artifact }),
|
|
80
|
+
defaultScore: text.defaultScore,
|
|
81
|
+
defaultTries: text.defaultTries,
|
|
82
|
+
...(text.starts === undefined ? {} : { starts: text.starts }),
|
|
83
|
+
stepLabel: step => text.roundTitle(step),
|
|
84
|
+
artifactMarker: step => text.artifactMarker(step),
|
|
85
|
+
brief: (limits, step, attempt) => [
|
|
86
|
+
...text.brief({ ...limits, step, attempt }),
|
|
87
|
+
'',
|
|
88
|
+
...resultContract(name, limits, step, attempt, {
|
|
89
|
+
standard: standard(step),
|
|
90
|
+
...(artifact === undefined ? {} : { artifact }),
|
|
91
|
+
...(text.focus(step) === undefined ? {} : { focus: text.focus(step) }),
|
|
92
|
+
...(consolidates(limits, step) ? { final: true } : {}),
|
|
93
|
+
}, forked ? 'forked' : 'subagent', selfScoring),
|
|
94
|
+
].join('\n'),
|
|
95
|
+
// The verifier's only input is this prompt and the artifact, so the run's own variables travel
|
|
96
|
+
// with it: without `path` it cannot tell which document this run reviews and can only follow the
|
|
97
|
+
// artifact left by an earlier run against a different one.
|
|
98
|
+
...(forked ? { verify: (limits, step, attempt, target, previous) => {
|
|
99
|
+
// The verifier is told the same heading the client will check, so "which section is this round's"
|
|
100
|
+
// is one fact on both sides instead of two readings of the same file.
|
|
101
|
+
const marker = text.artifactMarker(step);
|
|
102
|
+
return verdictBrief({
|
|
103
|
+
...target, kind: name, step, attempt, previous, standard: standard(step), vars: text.vars,
|
|
104
|
+
...(artifact === undefined ? {} : { artifact }),
|
|
105
|
+
...(marker === undefined ? {} : { marker }),
|
|
106
|
+
...(text.focus(step) === undefined ? {} : { focus: text.focus(step) }),
|
|
107
|
+
...(consolidates(limits, step) ? { coverage: coverage(step) } : {}),
|
|
108
|
+
});
|
|
109
|
+
} } : {}),
|
|
110
|
+
followUp: (limits, step, attempt) => [
|
|
111
|
+
...text.followUp({ ...limits, step, attempt }),
|
|
112
|
+
followUpContract(name, step, attempt, selfScoring),
|
|
113
|
+
].join('\n'),
|
|
114
|
+
};
|
|
115
|
+
}
|
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
/** The generic scored agent loop: what any "run steps until they pass" command reuses.
|
|
2
|
+
*
|
|
3
|
+
* A protocol supplies the prompt text and the step count; this module supplies everything else —
|
|
4
|
+
* the step/attempt state machine, the passing-score decision, and the reply parser. The controller
|
|
5
|
+
* adds the I/O around it (sending, watching for the turn to end, cancellation), so a second review
|
|
6
|
+
* command only writes a `LoopProtocol`.
|
|
7
|
+
*/
|
|
8
|
+
import type { LoopActivity, LoopLimits, LoopProgress, LoopTerminalReason } from '../contracts.ts';
|
|
9
|
+
import type { LoopOptions } from '../slash/index.ts';
|
|
10
|
+
import type { Message } from '../session/transcript.ts';
|
|
11
|
+
/** One protocol's fully resolved limits, shared with the form that confirms them. */
|
|
12
|
+
export type { LoopLimits };
|
|
13
|
+
/** What one loop run is: its label, its step count, and the prompts it sends. */
|
|
14
|
+
export interface LoopProtocol {
|
|
15
|
+
/** Fence marker of the reply's result block, without the backticks, e.g. `dsht-loop`. */
|
|
16
|
+
marker: string;
|
|
17
|
+
/** `kind` the result block must declare, so one marker can serve several protocols. */
|
|
18
|
+
kind: string;
|
|
19
|
+
/** Label shown in the progress line, e.g. `Design review`. */
|
|
20
|
+
title: string;
|
|
21
|
+
/** Steps the protocol defines; `--to` defaults to this. */
|
|
22
|
+
steps: number;
|
|
23
|
+
/** Workspace file the rounds maintain, when there is one; verified for tampering when readable. */
|
|
24
|
+
artifact?: string;
|
|
25
|
+
/** Line that file must contain once one step is done, when the protocol requires one.
|
|
26
|
+
*
|
|
27
|
+
* The score is the verifier's judgement; whether the round's conclusion actually reached the
|
|
28
|
+
* artifact is a fact this client checks, and a hard condition no score may override.
|
|
29
|
+
*/
|
|
30
|
+
artifactMarker?(step: number): string | undefined;
|
|
31
|
+
/** Optional name of one step, shown in the progress line. */
|
|
32
|
+
stepLabel?(step: number): string;
|
|
33
|
+
/** Passing score used when the command omits `--score`; default 8. */
|
|
34
|
+
defaultScore?: number;
|
|
35
|
+
/** Attempts per step used when the command omits `--tries`; default 10. */
|
|
36
|
+
defaultTries?: number;
|
|
37
|
+
/** Which phase a step starts in: verifying the existing artifact, or working on it.
|
|
38
|
+
*
|
|
39
|
+
* A review of something that already exists starts by verifying it, so a step that passes costs
|
|
40
|
+
* no work turn; anything else starts by asking for the work. Defaults to `work`.
|
|
41
|
+
*/
|
|
42
|
+
starts?: 'verify' | 'work';
|
|
43
|
+
/** Opening prompt for one step and attempt. */
|
|
44
|
+
brief(limits: LoopLimits, step: number, attempt: number): string;
|
|
45
|
+
/** Prompt for a later attempt, once the protocol is already in context. */
|
|
46
|
+
followUp(limits: LoopLimits, step: number, attempt: number): string;
|
|
47
|
+
/** Prompt a forked verifier session receives for one finished round.
|
|
48
|
+
*
|
|
49
|
+
* A protocol that defines this delegates the verdict to an independent process and names the file
|
|
50
|
+
* that process must write. Independent verification is the only scorer: a reply block is never
|
|
51
|
+
* used to override its absence, so an outage stops the run instead of passing it.
|
|
52
|
+
* @param limits - Resolved run limits.
|
|
53
|
+
* @param step - Step in flight.
|
|
54
|
+
* @param attempt - Attempt in flight.
|
|
55
|
+
* @param target - Where the verdict goes and the identity it must declare.
|
|
56
|
+
* @param previous - What the previous attempt on this step concluded, when there was one.
|
|
57
|
+
* @returns The verifier's prompt.
|
|
58
|
+
*/
|
|
59
|
+
verify?(limits: LoopLimits, step: number, attempt: number, target: VerifyTarget, previous?: PriorVerdict): string;
|
|
60
|
+
}
|
|
61
|
+
/** What one reply's result block reported. */
|
|
62
|
+
export interface LoopResult {
|
|
63
|
+
/** Completion score in 0–10; absent when the block omitted or malformed it. */
|
|
64
|
+
score?: number;
|
|
65
|
+
/** What the verifier called it; advisory only — the loop decides from `score`, `blocked` and `abstained`. */
|
|
66
|
+
status?: 'done' | 'retry' | 'blocked' | 'abstained';
|
|
67
|
+
/** The verifier proved the task impossible at any score; the one field that overrides `score`. */
|
|
68
|
+
blocked?: boolean;
|
|
69
|
+
/** The verifier cannot judge and needs a person; ends the run without a score. */
|
|
70
|
+
abstained?: boolean;
|
|
71
|
+
/** Why it stopped, when it stopped before the budget: `cannot-fix` or `needs-human`. */
|
|
72
|
+
exitReason?: 'cannot-fix' | 'needs-human';
|
|
73
|
+
/** That reason in one line a reader can act on; required by an `exitReason`. */
|
|
74
|
+
reason?: string;
|
|
75
|
+
/** One line a person must supply, when the verifier asked for one. */
|
|
76
|
+
needs?: string;
|
|
77
|
+
/** Advisory note that never changes control: "nothing to change here", "already covered", … */
|
|
78
|
+
explanation?: string;
|
|
79
|
+
/** What the verifier still objects to, so the next attempt can answer it. */
|
|
80
|
+
findings?: readonly string[];
|
|
81
|
+
/** The basis it gave for the score, so a retry does not have to guess. */
|
|
82
|
+
evidence?: string;
|
|
83
|
+
}
|
|
84
|
+
/** Where one round's verdict goes, and the identity that file must declare. */
|
|
85
|
+
export interface VerifyTarget {
|
|
86
|
+
/** Absolute path the verdict must be written to. */
|
|
87
|
+
file: string;
|
|
88
|
+
/** Identity embedded in the verdict, checked before the round. */
|
|
89
|
+
verificationId: string;
|
|
90
|
+
}
|
|
91
|
+
/** What the previous attempt on this step concluded, when there was one. */
|
|
92
|
+
export interface PriorVerdict {
|
|
93
|
+
/** Step that attempt belonged to. */
|
|
94
|
+
step: number;
|
|
95
|
+
/** Attempt number it was. */
|
|
96
|
+
attempt: number;
|
|
97
|
+
/** Its verdict. */
|
|
98
|
+
result: LoopResult;
|
|
99
|
+
}
|
|
100
|
+
/** What one consumed attempt decided. */
|
|
101
|
+
export type LoopStepResult = {
|
|
102
|
+
kind: 'continue';
|
|
103
|
+
prompt: string;
|
|
104
|
+
} | {
|
|
105
|
+
kind: 'passed';
|
|
106
|
+
} | {
|
|
107
|
+
kind: 'exhausted';
|
|
108
|
+
} | {
|
|
109
|
+
kind: 'stalled';
|
|
110
|
+
} | {
|
|
111
|
+
kind: 'blocked';
|
|
112
|
+
} | {
|
|
113
|
+
kind: 'needs-human';
|
|
114
|
+
};
|
|
115
|
+
/** Apply a protocol's defaults and reject a range that cannot run.
|
|
116
|
+
* @param protocol - Protocol being started.
|
|
117
|
+
* @param options - Flags exactly as parsed, absent when the operator omitted them.
|
|
118
|
+
* @returns The resolved limits, or undefined when `to < from`.
|
|
119
|
+
*/
|
|
120
|
+
export declare function resolveLoop(protocol: LoopProtocol, options: LoopOptions): LoopLimits | undefined;
|
|
121
|
+
/** Whether a run's rounds cover the whole record rather than a selected range.
|
|
122
|
+
*
|
|
123
|
+
* Only this case may report more than "the rounds that ran passed": the last round is then the
|
|
124
|
+
* record's consolidation round, and the verifier is handed every earlier round to re-check.
|
|
125
|
+
* @param from - First step of the run.
|
|
126
|
+
* @param to - Last step of the run.
|
|
127
|
+
* @param steps - Steps the protocol defines.
|
|
128
|
+
* @returns True when the run starts at the first round and ends at the last one.
|
|
129
|
+
*/
|
|
130
|
+
export declare function coversWholeProtocol(from: number, to: number, steps: number): boolean;
|
|
131
|
+
/** Read the result out of the last block of one reply.
|
|
132
|
+
*
|
|
133
|
+
* The block must be in the assistant's text, not reasoning, and this takes the last one so a reply
|
|
134
|
+
* that quotes its verifier still reports its own verdict. A missing, malformed or foreign block is
|
|
135
|
+
* undefined; a valid block with no usable field is an empty result, which counts as a failed attempt.
|
|
136
|
+
* @param text - One turn's assistant text.
|
|
137
|
+
* @param protocol - Protocol whose marker and kind identify the block.
|
|
138
|
+
* @returns The reported score and verdict, or undefined when the reply carries no valid block.
|
|
139
|
+
*/
|
|
140
|
+
export declare function parseLoopResult(text: string, protocol: Pick<LoopProtocol, 'marker' | 'kind'>): LoopResult | undefined;
|
|
141
|
+
/** Validate the score and verdict of one decoded result object.
|
|
142
|
+
*
|
|
143
|
+
* Shared with the forked verifier's verdict file, so a score means the same thing however it
|
|
144
|
+
* travelled: an out-of-range or unknown field is absent rather than a value the loop would trust.
|
|
145
|
+
* @param parsed - Decoded object expected to carry `score` and `status`.
|
|
146
|
+
* @returns The usable fields; an empty result when the object carried none, and `undefined` when it
|
|
147
|
+
* claimed an early stop that breaks the rules — such a claim is not a verdict at all, so the
|
|
148
|
+
* caller reports verification as unusable instead of guessing what was meant.
|
|
149
|
+
*/
|
|
150
|
+
export declare function readResultFields(parsed: object): LoopResult | undefined;
|
|
151
|
+
/** Assistant text of the turn that just finished: from the last user/context row to the end.
|
|
152
|
+
* @param messages - Projected conversation messages.
|
|
153
|
+
* @returns The assistant text, empty when the turn produced none.
|
|
154
|
+
*/
|
|
155
|
+
export declare function latestAssistantText(messages: readonly Message[]): string;
|
|
156
|
+
/** One run of the loop: which step and attempt is in flight, and what happens next.
|
|
157
|
+
*
|
|
158
|
+
* No I/O: the controller sends the prompt this returns and feeds back the parsed score, which keeps
|
|
159
|
+
* the state machine trivially testable and identical for every protocol.
|
|
160
|
+
*/
|
|
161
|
+
export declare class ScoredLoop {
|
|
162
|
+
readonly runId: string;
|
|
163
|
+
readonly sessionId: string;
|
|
164
|
+
readonly protocol: LoopProtocol;
|
|
165
|
+
private readonly limits;
|
|
166
|
+
private step;
|
|
167
|
+
private attempt;
|
|
168
|
+
private best;
|
|
169
|
+
private noProgress;
|
|
170
|
+
private phase;
|
|
171
|
+
private awaiting;
|
|
172
|
+
/** What the live run is waiting on; only the controller's own sends and verdicts set it. */
|
|
173
|
+
private activity?;
|
|
174
|
+
/** Why the run stopped; set once, together with a terminal phase. */
|
|
175
|
+
private terminalReason?;
|
|
176
|
+
private noteText?;
|
|
177
|
+
private interaction?;
|
|
178
|
+
private exit?;
|
|
179
|
+
/** When this run began, so a reader can clock it even while no host turn is running. */
|
|
180
|
+
private readonly startedAt;
|
|
181
|
+
constructor(runId: string, sessionId: string, protocol: LoopProtocol, limits: LoopLimits);
|
|
182
|
+
/** Snapshot the UI renders; the loop keeps the authoritative numbers. */
|
|
183
|
+
get progress(): LoopProgress;
|
|
184
|
+
/** Attach one line about the attempt just decided, shown until the next verdict replaces it.
|
|
185
|
+
* @param text - Note to show, or an empty string to clear it.
|
|
186
|
+
*/
|
|
187
|
+
note(text: string): void;
|
|
188
|
+
/** Whether the run still owns its session: it may send, settle — or be answered.
|
|
189
|
+
*
|
|
190
|
+
* A paused run counts: it is not finished, so a new run must not silently replace it (§8.3.1). What
|
|
191
|
+
* it is *not* doing is working; `activity` is absent while it waits.
|
|
192
|
+
*/
|
|
193
|
+
get active(): boolean;
|
|
194
|
+
/** Whether this run has sent any prompt yet, so `intro` knows the opening one is still owed. */
|
|
195
|
+
private briefed;
|
|
196
|
+
/** Whether a prompt was sent and its reply is still outstanding. */
|
|
197
|
+
get settled(): boolean;
|
|
198
|
+
/** The opening prompt; the caller sends it and then calls `sent()`. */
|
|
199
|
+
start(): string;
|
|
200
|
+
/** The opening prompt, when this run has not sent one yet; undefined once it has.
|
|
201
|
+
*
|
|
202
|
+
* A record that verifies first never sends its brief at the start — it checks the artifact instead —
|
|
203
|
+
* so the first work message after a failed check has to be the brief. A follow-up would open with
|
|
204
|
+
* "read the results and opinions above" and refer to a previous version, and neither exists on the
|
|
205
|
+
* first work turn; the round's own checklist only reaches the agent through this prompt.
|
|
206
|
+
*/
|
|
207
|
+
intro(): string | undefined;
|
|
208
|
+
/** The prompt that asks for work on the step in flight: the owed brief, else the follow-up. */
|
|
209
|
+
workPrompt(): string;
|
|
210
|
+
/** Record that the outstanding prompt reached the host. */
|
|
211
|
+
sent(): void;
|
|
212
|
+
/** Record that this attempt is judged by a forked verifier instead of the session's own turn. */
|
|
213
|
+
verifying(): void;
|
|
214
|
+
/** Record that the turn ended and its result is being read, which is neither work nor a verdict. */
|
|
215
|
+
settling(): void;
|
|
216
|
+
/** Consume one finished attempt.
|
|
217
|
+
* @param result - Parsed result, or undefined when the reply carried no usable block.
|
|
218
|
+
* @returns What the loop does next.
|
|
219
|
+
*/
|
|
220
|
+
settle(result: LoopResult | undefined): LoopStepResult;
|
|
221
|
+
/** Stop the run; a cancelled run never sends again.
|
|
222
|
+
* @param reason - Who ended it, so a cancellation is not confused with a verdict.
|
|
223
|
+
*/
|
|
224
|
+
cancel(reason?: LoopTerminalReason): void;
|
|
225
|
+
/** End the run because independent verification was impossible.
|
|
226
|
+
*
|
|
227
|
+
* Deliberately not a verdict: no attempt is consumed, so the run stops on an infrastructure
|
|
228
|
+
* failure instead of pretending the reviewer produced nothing.
|
|
229
|
+
* @param reason - Which verification failure ended it.
|
|
230
|
+
*/
|
|
231
|
+
unavailable(reason?: LoopTerminalReason): void;
|
|
232
|
+
/** End the run because the whole-run deadline expired.
|
|
233
|
+
*
|
|
234
|
+
* A budget stop, not a verdict: it bounds the sum of all steps, attempts and verifier retries, so
|
|
235
|
+
* no attempt is consumed and `best` is untouched.
|
|
236
|
+
*/
|
|
237
|
+
deadline(): void;
|
|
238
|
+
/** Pause the run because a judgment asked for a person.
|
|
239
|
+
*
|
|
240
|
+
* Deliberately not terminal: no `terminalReason` is written, so the run still owns its session and
|
|
241
|
+
* the progress line can offer `/loop answer` or `/loop abort` instead of a summary.
|
|
242
|
+
* @param request - What is being asked, and what the answer has to supply.
|
|
243
|
+
*/
|
|
244
|
+
pause(request: {
|
|
245
|
+
kind: string;
|
|
246
|
+
text: string;
|
|
247
|
+
needs?: string;
|
|
248
|
+
}): void;
|
|
249
|
+
/** Resume a paused run after the operator supplied what the judgment was missing.
|
|
250
|
+
*
|
|
251
|
+
* No attempt is consumed: the answer only adds a condition, and the current artifact is judged again
|
|
252
|
+
* under a new verification identity. The attempt counts as outstanding, so that verdict may decide it.
|
|
253
|
+
*/
|
|
254
|
+
resume(activity?: LoopActivity): void;
|
|
255
|
+
/** The prompt that asks again with the operator's addition.
|
|
256
|
+
*
|
|
257
|
+
* Used when no forked verifier judges this run: the answer becomes the next attempt's instruction
|
|
258
|
+
* instead of a condition for a judgment, and asking again costs no attempt by itself.
|
|
259
|
+
* @param answer - What the operator supplied.
|
|
260
|
+
* @returns The prompt to send.
|
|
261
|
+
*/
|
|
262
|
+
answerPrompt(answer: string): string;
|
|
263
|
+
/** End the run because the request cannot be answered from here.
|
|
264
|
+
*
|
|
265
|
+
* Used for the cases the operator cannot unblock through this client — a verifier child whose own
|
|
266
|
+
* session asked a question, or a prompt the host refused to accept — as opposed to `pause`, where
|
|
267
|
+
* `/loop answer` supplies what the judgment was missing. No attempt is consumed in either case.
|
|
268
|
+
* @param request - The approval or question the host is waiting for, or a verdict's request.
|
|
269
|
+
* @param reason - Which kind of human request this is.
|
|
270
|
+
*/
|
|
271
|
+
human(request: {
|
|
272
|
+
kind: string;
|
|
273
|
+
text: string;
|
|
274
|
+
}, reason?: LoopTerminalReason): void;
|
|
275
|
+
}
|