@a-t-h-i/bot-lobby 0.6.1 → 0.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +177 -834
- package/package.json +1 -1
- package/prompts/master.md +17 -0
- package/prompts/panel.md +3 -1
- package/prompts/planner.md +14 -0
- package/prompts/quickfix.md +3 -1
- package/prompts/scout.md +3 -0
- package/prompts/worker.md +3 -0
- package/src/classifier/answers.ts +110 -0
- package/src/classifier/classifier.ts +171 -0
- package/src/classifier/client.ts +239 -0
- package/src/classifier/effort.ts +143 -0
- package/src/classifier/files.ts +427 -0
- package/src/classifier/hosts.ts +157 -0
- package/src/classifier/instance.ts +98 -0
- package/src/classifier/limits.ts +37 -0
- package/src/classifier/seats.ts +102 -0
- package/src/classifier/tools.ts +58 -0
- package/src/classifier/triage.ts +203 -0
- package/src/execution/agent-runner.ts +5 -0
- package/src/index.ts +6 -0
- package/src/lobby/planner.ts +305 -27
- package/src/lobby/quickfix.ts +120 -8
- package/src/lobby/runtime.ts +12 -1
- package/src/lobby/tabs/metrics.ts +28 -1
- package/src/lobby/tabs/plan.ts +27 -4
- package/src/lobby/tabs/quickfix.ts +5 -1
- package/src/lobby/view.ts +28 -4
- package/src/master/master.ts +95 -29
- package/src/pi/commands.ts +4 -2
- package/src/pi/events.ts +11 -2
- package/src/pi/run-summary.ts +3 -1
- package/src/pi/settings-ui.ts +138 -2
- package/src/pi/start-task.ts +4 -0
- package/src/pi/tools.ts +7 -2
- package/src/schemas/configuration.ts +125 -1
- package/src/schemas/findings.ts +4 -0
- package/src/schemas/task.ts +22 -0
- package/src/state/metrics.ts +74 -1
- package/src/workflow/workflow.ts +49 -1
package/package.json
CHANGED
package/prompts/master.md
CHANGED
|
@@ -33,6 +33,23 @@ directly. The engine allows `clarifying -> awaiting_approval -> planning`, so no
|
|
|
33
33
|
state override is needed. Skip only when the change is small, obvious and
|
|
34
34
|
confined to one domain.
|
|
35
35
|
|
|
36
|
+
## Classifier hints
|
|
37
|
+
|
|
38
|
+
When the classifier is on, your task context carries a **Classifier
|
|
39
|
+
triage**: a fast model's read of the request (size, the domains it touches,
|
|
40
|
+
whether it needs outside research, whether it is ambiguous, its kind, likely
|
|
41
|
+
files) and a suggested path. Use it to skip reasoning you do not need —
|
|
42
|
+
scout only the domains it marks (0.5 or more), skip scouting when the task
|
|
43
|
+
is trivial or small in one domain and likely files are named, skip the
|
|
44
|
+
researcher when research is not needed, clarify only when it reads the
|
|
45
|
+
request as ambiguous — and overrule it whenever the repository says
|
|
46
|
+
otherwise. It is a hint, never a rule.
|
|
47
|
+
|
|
48
|
+
When you `clarify` with options, put your recommended option first and mark
|
|
49
|
+
it `(Recommended)`. When the request already makes it clearly right, the
|
|
50
|
+
classifier answers for you: the reply says so, the decision is recorded, and
|
|
51
|
+
you mention it in the proposal so the user can amend it.
|
|
52
|
+
|
|
36
53
|
## Architecture and systems thinking
|
|
37
54
|
|
|
38
55
|
You are the system's architect. Before you propose, build a model of the system
|
package/prompts/panel.md
CHANGED
|
@@ -6,7 +6,9 @@ the user answers everyone's questions in one conversation, so every agent that
|
|
|
6
6
|
later works on the task starts from the same decisions.
|
|
7
7
|
|
|
8
8
|
You never write code or change files. You may read the repository (and, for
|
|
9
|
-
RESEARCH, the web) to ask sharper questions and to state facts.
|
|
9
|
+
RESEARCH, the web) to ask sharper questions and to state facts. When the
|
|
10
|
+
conversation ends with **Likely files**, read those first; `find_relevant_files`
|
|
11
|
+
(when you have it) finds more by description.
|
|
10
12
|
|
|
11
13
|
## Each round
|
|
12
14
|
|
package/prompts/planner.md
CHANGED
|
@@ -40,6 +40,11 @@ below.
|
|
|
40
40
|
- Challenge answers that are vague, contradictory or risky, and ask again.
|
|
41
41
|
Do not accept "whatever you think" for a decision with real trade-offs:
|
|
42
42
|
propose one and ask the user to confirm it.
|
|
43
|
+
- The conversation may show questions **decided by the classifier**: a
|
|
44
|
+
fast model answered them with their recommended option because the
|
|
45
|
+
conversation already made it clearly right. Treat them as answered, list
|
|
46
|
+
each under `### Assumptions` (the user can overrule it), and do not ask
|
|
47
|
+
them again.
|
|
43
48
|
- Ground every claim about the codebase in files you read; name them.
|
|
44
49
|
- Keep a draft plan updated every turn so the user sees it converge.
|
|
45
50
|
|
|
@@ -47,6 +52,15 @@ Declare the plan READY only when every panel member is READY and nothing
|
|
|
47
52
|
that would change the implementation is still open. Until then the status is
|
|
48
53
|
GRILLING.
|
|
49
54
|
|
|
55
|
+
## Round limit
|
|
56
|
+
|
|
57
|
+
Planning may be limited to a number of rounds; your task says which round
|
|
58
|
+
this is. Ask the questions that change the most early. In the final round,
|
|
59
|
+
and in any round after it, no member runs and nothing more is asked: fold the
|
|
60
|
+
answers into the plan, decide every open point with its recommended option,
|
|
61
|
+
list each under `### Assumptions`, omit the Questions section and set the
|
|
62
|
+
status READY.
|
|
63
|
+
|
|
50
64
|
## Output format
|
|
51
65
|
|
|
52
66
|
## Status
|
package/prompts/quickfix.md
CHANGED
|
@@ -7,7 +7,9 @@ the change now, the way they would ask pi directly.
|
|
|
7
7
|
## How to work
|
|
8
8
|
|
|
9
9
|
- Read only what you need to make the change safely; follow the file's
|
|
10
|
-
existing conventions.
|
|
10
|
+
existing conventions. When the request ends with **Likely files**, start
|
|
11
|
+
there (and use `find_relevant_files`, when you have it, before a broad
|
|
12
|
+
search).
|
|
11
13
|
- Make the smallest correct change that does exactly what was asked. Do not
|
|
12
14
|
refactor, rename or tidy anything else.
|
|
13
15
|
- If the request is ambiguous, pick the most reasonable reading and say which
|
package/prompts/scout.md
CHANGED
|
@@ -32,6 +32,9 @@ You have read-only tools.
|
|
|
32
32
|
You are reconnaissance, not an audit. Answer the Master's instruction and stop.
|
|
33
33
|
|
|
34
34
|
- Budget: about 15 tool calls. Stop as soon as you can answer.
|
|
35
|
+
- When your context has a **Likely files** section, start with those files.
|
|
36
|
+
When the `find_relevant_files` tool is available, ask it in plain words
|
|
37
|
+
("where the session cookie is validated") before a broad `find`/`grep`.
|
|
35
38
|
- Prefer `grep` and `find` to locate code, then `read` only the relevant
|
|
36
39
|
ranges; do not read whole large files or walk the whole tree.
|
|
37
40
|
- Report what you found with file paths; mark anything you did not verify as
|
package/prompts/worker.md
CHANGED
|
@@ -9,6 +9,9 @@ domain-specific responsibility.
|
|
|
9
9
|
- Inspect existing patterns.
|
|
10
10
|
- Verify Scout findings against the repository.
|
|
11
11
|
- Grep/find callers before changing shared behavior.
|
|
12
|
+
- When your context has a **Likely files** section, open those first; with
|
|
13
|
+
the `find_relevant_files` tool, describe what you need in plain words
|
|
14
|
+
before walking the tree. Both are hints: verify what you rely on.
|
|
12
15
|
- Identify relevant tests.
|
|
13
16
|
|
|
14
17
|
You may disagree with Scout findings when repository evidence contradicts
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Obvious answers. The panel recommends an option for every question it
|
|
3
|
+
* asks; when the conversation and the draft already make that option
|
|
4
|
+
* clearly right, the classifier answers it instead of the user. The rule is
|
|
5
|
+
* strict: the classifier's pick must BE the recommended option, at
|
|
6
|
+
* `autoAnswerAt` or above and ahead of the runner-up by `autoAnswerMargin`.
|
|
7
|
+
* Each question also offers "ask the user" (a preference or a fact only the
|
|
8
|
+
* user has), so a question that is really the user's call is never forced
|
|
9
|
+
* onto one of the options.
|
|
10
|
+
*/
|
|
11
|
+
import type { ClassifierThresholds } from "../schemas/configuration.ts";
|
|
12
|
+
import type { Classifier } from "./classifier.ts";
|
|
13
|
+
import { choice, choiceOf, type Answer, type SystemOneRequest, type Text } from "./client.ts";
|
|
14
|
+
import { clip, clipTail, LIMITS } from "./limits.ts";
|
|
15
|
+
|
|
16
|
+
export const ASK_THE_USER = "ask the user";
|
|
17
|
+
|
|
18
|
+
export interface AnswerCandidate {
|
|
19
|
+
/** Position in the round's questions. */
|
|
20
|
+
index: number;
|
|
21
|
+
/** The seat the question serves. */
|
|
22
|
+
from: string;
|
|
23
|
+
text: string;
|
|
24
|
+
/** Option labels without any "(Recommended)" marker, and what each means. */
|
|
25
|
+
options: ReadonlyArray<{ label: string; description: string }>;
|
|
26
|
+
/** The asker's recommended option, by label. */
|
|
27
|
+
recommended: string;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export interface AutoAnswer {
|
|
31
|
+
index: number;
|
|
32
|
+
from: string;
|
|
33
|
+
question: string;
|
|
34
|
+
answer: string;
|
|
35
|
+
probability: number;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export interface AnswerContext {
|
|
39
|
+
/** The idea as first described. */
|
|
40
|
+
request: string;
|
|
41
|
+
/** The conversation so far, oldest first. */
|
|
42
|
+
conversation: string;
|
|
43
|
+
draft?: string;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
function questionKey(index: number): string {
|
|
47
|
+
return `q${index + 1}`;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** Unique option keys: the label, numbered when two labels collide. */
|
|
51
|
+
function optionKeys(candidate: AnswerCandidate): string[] {
|
|
52
|
+
const seen = new Set<string>();
|
|
53
|
+
return candidate.options.map((option) => {
|
|
54
|
+
let key = option.label.trim() || "option";
|
|
55
|
+
for (let n = 2; seen.has(key.toLowerCase()) || key.toLowerCase() === ASK_THE_USER; n++) key = `${option.label.trim() || "option"} (${n})`;
|
|
56
|
+
seen.add(key.toLowerCase());
|
|
57
|
+
return key;
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
export function answerRequest(candidates: readonly AnswerCandidate[], context: AnswerContext): SystemOneRequest {
|
|
62
|
+
const questions: Record<string, ReturnType<typeof choice>> = {};
|
|
63
|
+
for (const candidate of candidates) {
|
|
64
|
+
const keys = optionKeys(candidate);
|
|
65
|
+
const criteria: Record<string, Text> = {};
|
|
66
|
+
candidate.options.slice(0, LIMITS.choiceOptions - 1).forEach((option, position) => {
|
|
67
|
+
criteria[keys[position]!] = option.description || null;
|
|
68
|
+
});
|
|
69
|
+
criteria[ASK_THE_USER] = "Not settled by `conversation` or `draft`: it depends on the user's preference or priorities, or on facts only the user has.";
|
|
70
|
+
questions[questionKey(candidate.index)] = choice(
|
|
71
|
+
`The planning panel's ${candidate.from} seat asks the user: "${clip(candidate.text, 600)}". Which answer do \`request\`, \`conversation\` and \`draft\` already make clearly right? Pick ${JSON.stringify(ASK_THE_USER)} unless the evidence settles it.`,
|
|
72
|
+
criteria,
|
|
73
|
+
);
|
|
74
|
+
}
|
|
75
|
+
return {
|
|
76
|
+
state: {
|
|
77
|
+
request: clip(context.request, 3000),
|
|
78
|
+
conversation: clipTail(context.conversation, 8000),
|
|
79
|
+
draft: clip(context.draft ?? "", 6000),
|
|
80
|
+
},
|
|
81
|
+
questions,
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** The questions the classifier settles: its pick is the recommended option, confident and clearly ahead. */
|
|
86
|
+
export function decideAnswers(answers: Record<string, Answer> | undefined, candidates: readonly AnswerCandidate[], thresholds: Pick<ClassifierThresholds, "autoAnswerAt" | "autoAnswerMargin">): AutoAnswer[] {
|
|
87
|
+
const decided: AutoAnswer[] = [];
|
|
88
|
+
for (const candidate of candidates) {
|
|
89
|
+
const picked = choiceOf(answers, questionKey(candidate.index));
|
|
90
|
+
if (!picked || picked.choice === ASK_THE_USER) continue;
|
|
91
|
+
const keys = optionKeys(candidate);
|
|
92
|
+
const position = keys.indexOf(picked.choice);
|
|
93
|
+
if (position < 0) continue;
|
|
94
|
+
const label = candidate.options[position]!.label;
|
|
95
|
+
if (label.toLowerCase() !== candidate.recommended.toLowerCase()) continue;
|
|
96
|
+
if (picked.probability < thresholds.autoAnswerAt || picked.margin < thresholds.autoAnswerMargin) continue;
|
|
97
|
+
decided.push({ index: candidate.index, from: candidate.from, question: candidate.text, answer: label, probability: picked.probability });
|
|
98
|
+
}
|
|
99
|
+
return decided;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/** One call for the round's questions; undefined when the classifier is off or fails. */
|
|
103
|
+
export async function autoAnswer(classifier: Classifier, candidates: readonly AnswerCandidate[], context: AnswerContext, signal?: AbortSignal): Promise<AutoAnswer[] | undefined> {
|
|
104
|
+
const eligible = candidates.filter((candidate) => candidate.options.length >= 2 && candidate.recommended);
|
|
105
|
+
if (eligible.length === 0) return undefined;
|
|
106
|
+
const thresholds = classifier.config.thresholds;
|
|
107
|
+
const result = await classifier.ask("answers", answerRequest(eligible, context), { ...(signal ? { signal } : {}), saved: (answers) => decideAnswers(answers, eligible, thresholds).length });
|
|
108
|
+
if (!result) return undefined;
|
|
109
|
+
return decideAnswers(result.answers, eligible, classifier.config.thresholds);
|
|
110
|
+
}
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The classifier facade every feature calls. It never throws into the
|
|
3
|
+
* workflow: any failure (no key, a timeout, an API error, a request over
|
|
4
|
+
* budget) returns undefined, and the caller falls back to what bot-lobby did
|
|
5
|
+
* before the classifier existed. Three failures in a row pause it for ten
|
|
6
|
+
* minutes with a single warning, so a dead endpoint cannot tax every step.
|
|
7
|
+
* Every call lands in the metrics log (kind `classifier`).
|
|
8
|
+
*/
|
|
9
|
+
import type { ClassifierConfig, ClassifierFeature } from "../schemas/configuration.ts";
|
|
10
|
+
import type { MetricRecord } from "../state/metrics.ts";
|
|
11
|
+
import { JevError, noul, systemOne, type Answer, type FetchLike, type SystemOneRequest } from "./client.ts";
|
|
12
|
+
import { fitsBudget, requestSize, LIMITS } from "./limits.ts";
|
|
13
|
+
import { keyHint, resolveTarget, type KeySource } from "./hosts.ts";
|
|
14
|
+
|
|
15
|
+
export type ClassifierPurpose = ClassifierFeature | "test";
|
|
16
|
+
|
|
17
|
+
/** What a call answered, with the served model and how long it took. */
|
|
18
|
+
export interface Classified {
|
|
19
|
+
answers: Record<string, Answer>;
|
|
20
|
+
model: string;
|
|
21
|
+
ms: number;
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export type TestResult = { ok: true; model: string; ms: number } | { ok: false; error: string };
|
|
25
|
+
|
|
26
|
+
export interface ClassifierDeps {
|
|
27
|
+
/** Read live, so a settings change applies to the next call. */
|
|
28
|
+
config: () => ClassifierConfig;
|
|
29
|
+
keys?: KeySource;
|
|
30
|
+
fetch?: FetchLike;
|
|
31
|
+
sleep?: (ms: number) => Promise<void>;
|
|
32
|
+
env?: NodeJS.ProcessEnv;
|
|
33
|
+
/** Receives one record per call. */
|
|
34
|
+
metrics?: (record: MetricRecord) => void;
|
|
35
|
+
/** One-off notices: a missing key, the breaker pausing the classifier. */
|
|
36
|
+
warn?: (message: string) => void;
|
|
37
|
+
now?: () => number;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export const BREAKER_FAILURES = 3;
|
|
41
|
+
export const BREAKER_PAUSE_MS = 10 * 60 * 1000;
|
|
42
|
+
|
|
43
|
+
export class Classifier {
|
|
44
|
+
private readonly deps: ClassifierDeps;
|
|
45
|
+
private failures = 0;
|
|
46
|
+
private pausedUntil = 0;
|
|
47
|
+
private warnedNoKey = false;
|
|
48
|
+
private calls = 0;
|
|
49
|
+
|
|
50
|
+
constructor(deps: ClassifierDeps) {
|
|
51
|
+
this.deps = deps;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
private now(): number {
|
|
55
|
+
return this.deps.now?.() ?? Date.now();
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** The current settings (thresholds, limits), read live. */
|
|
59
|
+
get config(): ClassifierConfig {
|
|
60
|
+
return this.deps.config();
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/** Switched on (and, given a feature, that feature too) and not paused after failures. */
|
|
64
|
+
enabled(feature?: ClassifierFeature): boolean {
|
|
65
|
+
const config = this.deps.config();
|
|
66
|
+
if (!config.enabled) return false;
|
|
67
|
+
if (feature && !config.features[feature]) return false;
|
|
68
|
+
return this.now() >= this.pausedUntil;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/** Paused by the breaker until this time (ms), or 0. */
|
|
72
|
+
get pausedTill(): number {
|
|
73
|
+
return this.pausedUntil > this.now() ? this.pausedUntil : 0;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Ask every question over the state in one call. Undefined when the feature
|
|
78
|
+
* is off, the key is missing, the request is over budget, the call fails or
|
|
79
|
+
* `signal` aborts it.
|
|
80
|
+
*/
|
|
81
|
+
async ask(purpose: ClassifierFeature, request: SystemOneRequest, options: { signal?: AbortSignal; timeoutMs?: number; saved?: (answers: Record<string, Answer>) => number } = {}): Promise<Classified | undefined> {
|
|
82
|
+
if (!this.enabled(purpose)) return undefined;
|
|
83
|
+
if (!fitsBudget(request)) {
|
|
84
|
+
this.record(purpose, "failed", this.now(), 0);
|
|
85
|
+
this.deps.warn?.(`bot-lobby classifier: a ${purpose} request was ${requestSize(request)} characters (limit ${LIMITS.requestChars}); skipped.`);
|
|
86
|
+
return undefined;
|
|
87
|
+
}
|
|
88
|
+
const result = await this.call(purpose, request, options);
|
|
89
|
+
return result.ok ? result.value : undefined;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/** One tiny call that reports its error instead of hiding it; ignores the on/off switch and the breaker. */
|
|
93
|
+
async test(signal?: AbortSignal): Promise<TestResult> {
|
|
94
|
+
const result = await this.call("test", { state: { note: "bot-lobby connection test" }, questions: { test: noul("Is `note` a connection test?") } }, { signal, force: true });
|
|
95
|
+
return result.ok ? { ok: true, model: result.value.model, ms: result.value.ms } : { ok: false, error: result.error };
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
private async call(purpose: ClassifierPurpose, request: SystemOneRequest, options: { signal?: AbortSignal; timeoutMs?: number; force?: boolean; saved?: (answers: Record<string, Answer>) => number }): Promise<{ ok: true; value: Classified } | { ok: false; error: string }> {
|
|
99
|
+
const config = this.deps.config();
|
|
100
|
+
const { host, baseUrl, model, key } = await resolveTarget(config, this.deps.keys, this.deps.env);
|
|
101
|
+
if (!key) {
|
|
102
|
+
const error = `no Jev key: ${keyHint(config)}`;
|
|
103
|
+
if (!options.force && !this.warnedNoKey) {
|
|
104
|
+
this.warnedNoKey = true;
|
|
105
|
+
this.deps.warn?.(`bot-lobby classifier is on but has ${error}. Until then bot-lobby decides without it.`);
|
|
106
|
+
}
|
|
107
|
+
return { ok: false, error };
|
|
108
|
+
}
|
|
109
|
+
const started = this.now();
|
|
110
|
+
try {
|
|
111
|
+
const response = await systemOne(request, {
|
|
112
|
+
apiKey: key,
|
|
113
|
+
baseUrl,
|
|
114
|
+
model,
|
|
115
|
+
timeoutMs: options.timeoutMs ?? config.timeoutMs,
|
|
116
|
+
...(host.headers ? { headers: { ...host.headers } } : {}),
|
|
117
|
+
...(options.signal ? { signal: options.signal } : {}),
|
|
118
|
+
...(this.deps.fetch ? { fetch: this.deps.fetch } : {}),
|
|
119
|
+
...(this.deps.sleep ? { sleep: this.deps.sleep } : {}),
|
|
120
|
+
});
|
|
121
|
+
const ms = Math.max(0, this.now() - started);
|
|
122
|
+
this.failures = 0;
|
|
123
|
+
let saved = 0;
|
|
124
|
+
try {
|
|
125
|
+
saved = options.saved?.(response.answers) ?? 0;
|
|
126
|
+
} catch {
|
|
127
|
+
saved = 0;
|
|
128
|
+
}
|
|
129
|
+
this.record(purpose, "success", started, ms, response.model, response.usage, saved);
|
|
130
|
+
return { ok: true, value: { answers: response.answers, model: response.model, ms } };
|
|
131
|
+
} catch (error) {
|
|
132
|
+
const ms = Math.max(0, this.now() - started);
|
|
133
|
+
if (options.signal?.aborted) {
|
|
134
|
+
this.record(purpose, "cancelled", started, ms, model);
|
|
135
|
+
return { ok: false, error: "cancelled" };
|
|
136
|
+
}
|
|
137
|
+
let message = error instanceof JevError || error instanceof Error ? error.message : String(error);
|
|
138
|
+
// A limited-time free model that ends answers 404/410; never switch to the paid one silently.
|
|
139
|
+
if (error instanceof JevError && (error.status === 404 || error.status === 410) && model.endsWith("-free")) {
|
|
140
|
+
message += ` — ${host.label}'s free ${model} may have ended; set the classifier's model to ${model.replace(/-free$/, "")} (paid) in /bot-lobby settings to keep using Jev there`;
|
|
141
|
+
}
|
|
142
|
+
this.record(purpose, error instanceof JevError && /timed out/.test(message) ? "timeout" : "failed", started, ms, model);
|
|
143
|
+
if (!options.force) this.fail(message);
|
|
144
|
+
return { ok: false, error: message };
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
private fail(message: string): void {
|
|
149
|
+
this.failures += 1;
|
|
150
|
+
if (this.failures < BREAKER_FAILURES) return;
|
|
151
|
+
this.failures = 0;
|
|
152
|
+
this.pausedUntil = this.now() + BREAKER_PAUSE_MS;
|
|
153
|
+
this.deps.warn?.(`bot-lobby classifier paused for ${BREAKER_PAUSE_MS / 60_000} minutes after ${BREAKER_FAILURES} failures in a row (${message.split("\n")[0]}). bot-lobby decides without it meanwhile.`);
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
private record(purpose: ClassifierPurpose, status: MetricRecord["status"], started: number, ms: number, model?: string, usage?: { input_tokens: number; output_tokens: number }, saved = 0): void {
|
|
157
|
+
this.calls += 1;
|
|
158
|
+
this.deps.metrics?.({
|
|
159
|
+
id: `classifier-${purpose}-${started}-${this.calls}`,
|
|
160
|
+
kind: "classifier",
|
|
161
|
+
agent: "CLASSIFIER",
|
|
162
|
+
purpose,
|
|
163
|
+
...(model ? { model } : {}),
|
|
164
|
+
status,
|
|
165
|
+
startedAt: new Date(started).toISOString(),
|
|
166
|
+
durationMs: ms,
|
|
167
|
+
...(usage ? { input: usage.input_tokens, output: usage.output_tokens } : {}),
|
|
168
|
+
...(saved > 0 ? { saved } : {}),
|
|
169
|
+
});
|
|
170
|
+
}
|
|
171
|
+
}
|
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The Jev classifier's HTTP API (TypeSafe "System One"): typed questions over
|
|
3
|
+
* one state, all answered in a single call with calibrated probabilities.
|
|
4
|
+
* `POST {base}/v1/systemone` with a bearer key. Jev never writes text, so a
|
|
5
|
+
* call is fast (tens to hundreds of milliseconds) and cheap; the engine uses
|
|
6
|
+
* it for decisions that do not need a large model.
|
|
7
|
+
*
|
|
8
|
+
* Kept dependency-free: one request type, an injectable `fetch` for tests,
|
|
9
|
+
* a time limit per attempt and one retry on rate limits, server errors and
|
|
10
|
+
* unreachable hosts (a timeout is not retried).
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
export type JsonValue = string | number | boolean | null | JsonValue[] | { [key: string]: JsonValue };
|
|
14
|
+
/** What Jev judges: text, or JSON whose parts questions name with backticked paths (`` `draft` ``). */
|
|
15
|
+
export type State = string | JsonValue[] | { [key: string]: JsonValue };
|
|
16
|
+
/** Instructions and criteria: text or JSON; `null` leaves an option undescribed. */
|
|
17
|
+
export type Text = string | JsonValue[] | { [key: string]: JsonValue } | null;
|
|
18
|
+
|
|
19
|
+
/** Yes or no; the answer is the probability of yes. */
|
|
20
|
+
export interface NoulQuestion {
|
|
21
|
+
type: "noul";
|
|
22
|
+
instructions: Text;
|
|
23
|
+
criteria?: { true?: Text; false?: Text } | null;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** One of a set of labelled options (2 to 255). */
|
|
27
|
+
export interface ChoiceQuestion {
|
|
28
|
+
type: "choice";
|
|
29
|
+
instructions: Text;
|
|
30
|
+
criteria: Record<string, Text>;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** A position on ordered, described levels, lowest first. */
|
|
34
|
+
export interface ScoreQuestion {
|
|
35
|
+
type: "score";
|
|
36
|
+
instructions: Text;
|
|
37
|
+
criteria: Text[];
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export type Question = NoulQuestion | ChoiceQuestion | ScoreQuestion;
|
|
41
|
+
export type Questions = Record<string, Question>;
|
|
42
|
+
|
|
43
|
+
export interface NoulAnswer {
|
|
44
|
+
type: "noul";
|
|
45
|
+
noul: number;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export interface ChoiceAnswer {
|
|
49
|
+
type: "choice";
|
|
50
|
+
choice: string;
|
|
51
|
+
probabilities: Record<string, number>;
|
|
52
|
+
/** How peaked the distribution is, 0 to 1. */
|
|
53
|
+
confidence: number;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export interface ScoreAnswer {
|
|
57
|
+
type: "score";
|
|
58
|
+
/** Probability-weighted level index; may fall between two levels. */
|
|
59
|
+
score: number;
|
|
60
|
+
confidence: number;
|
|
61
|
+
probabilities?: Record<string, number>;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
export type Answer = NoulAnswer | ChoiceAnswer | ScoreAnswer;
|
|
65
|
+
|
|
66
|
+
export interface SystemOneRequest {
|
|
67
|
+
state: State;
|
|
68
|
+
questions: Questions;
|
|
69
|
+
model?: string;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
export interface SystemOneResponse {
|
|
73
|
+
model: string;
|
|
74
|
+
answers: Record<string, Answer>;
|
|
75
|
+
usage?: { input_tokens: number; output_tokens: number };
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
export const SYSTEM_ONE_PATH = "/v1/systemone";
|
|
79
|
+
|
|
80
|
+
export function noul(instructions: Text, yes?: Text, no?: Text): NoulQuestion {
|
|
81
|
+
return { type: "noul", instructions, ...(yes !== undefined || no !== undefined ? { criteria: { ...(yes !== undefined ? { true: yes } : {}), ...(no !== undefined ? { false: no } : {}) } } : {}) };
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
export function choice(instructions: Text, criteria: Record<string, Text>): ChoiceQuestion {
|
|
85
|
+
return { type: "choice", instructions, criteria };
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
export function score(instructions: Text, levels: Text[]): ScoreQuestion {
|
|
89
|
+
return { type: "score", instructions, criteria: levels };
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
function probabilityOf(value: unknown): number | undefined {
|
|
93
|
+
return typeof value === "number" && Number.isFinite(value) ? Math.min(1, Math.max(0, value)) : undefined;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/** The probability of yes for a `noul` answer, or undefined when it is missing or malformed. */
|
|
97
|
+
export function yesOf(answers: Record<string, Answer> | undefined, key: string): number | undefined {
|
|
98
|
+
const answer = answers?.[key];
|
|
99
|
+
return answer?.type === "noul" ? probabilityOf(answer.noul) : undefined;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/** A `choice` answer's pick, its probability and its lead over the runner-up; undefined when malformed. */
|
|
103
|
+
export function choiceOf(answers: Record<string, Answer> | undefined, key: string): { choice: string; probability: number; margin: number; confidence: number } | undefined {
|
|
104
|
+
const answer = answers?.[key];
|
|
105
|
+
if (answer?.type !== "choice" || typeof answer.choice !== "string" || !answer.probabilities || typeof answer.probabilities !== "object") return undefined;
|
|
106
|
+
const sorted = Object.values(answer.probabilities).map(probabilityOf).filter((value): value is number => value !== undefined).sort((a, b) => b - a);
|
|
107
|
+
const probability = probabilityOf(answer.probabilities[answer.choice]) ?? sorted[0] ?? 0;
|
|
108
|
+
return { choice: answer.choice, probability, margin: sorted.length > 1 ? sorted[0]! - sorted[1]! : sorted[0] ?? 0, confidence: probabilityOf(answer.confidence) ?? 0 };
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/** A `score` answer's position (level index) and confidence; undefined when malformed. */
|
|
112
|
+
export function scoreOf(answers: Record<string, Answer> | undefined, key: string): { score: number; level: number; confidence: number } | undefined {
|
|
113
|
+
const answer = answers?.[key];
|
|
114
|
+
if (answer?.type !== "score" || typeof answer.score !== "number" || !Number.isFinite(answer.score)) return undefined;
|
|
115
|
+
return { score: answer.score, level: Math.round(answer.score), confidence: probabilityOf(answer.confidence) ?? 0 };
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
export type FetchLike = (url: string, init: RequestInit) => Promise<Response>;
|
|
119
|
+
|
|
120
|
+
export interface JevCallOptions {
|
|
121
|
+
apiKey: string;
|
|
122
|
+
baseUrl: string;
|
|
123
|
+
model?: string;
|
|
124
|
+
/** Time limit per attempt. */
|
|
125
|
+
timeoutMs: number;
|
|
126
|
+
/** Retries after a rate limit, server error or timeout; 1 by default. */
|
|
127
|
+
retries?: number;
|
|
128
|
+
headers?: Record<string, string>;
|
|
129
|
+
signal?: AbortSignal;
|
|
130
|
+
fetch?: FetchLike;
|
|
131
|
+
sleep?: (ms: number) => Promise<void>;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** A failed call: `status` is the HTTP status when the API answered. */
|
|
135
|
+
export class JevError extends Error {
|
|
136
|
+
readonly status?: number;
|
|
137
|
+
readonly retryable: boolean;
|
|
138
|
+
/** How long the API asked us to wait before retrying (`retry-after`), capped. */
|
|
139
|
+
readonly waitMs?: number;
|
|
140
|
+
constructor(message: string, retryable: boolean, status?: number, waitMs?: number) {
|
|
141
|
+
super(message);
|
|
142
|
+
this.name = "JevError";
|
|
143
|
+
this.retryable = retryable;
|
|
144
|
+
if (status !== undefined) this.status = status;
|
|
145
|
+
if (waitMs !== undefined) this.waitMs = waitMs;
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
const RETRYABLE = new Set([408, 425, 429, 500, 502, 503, 504, 529]);
|
|
150
|
+
/** Waits are short: the classifier sits on the interactive path, and every caller has a fallback. */
|
|
151
|
+
const MAX_WAIT_MS = 2000;
|
|
152
|
+
|
|
153
|
+
const realSleep = (ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms));
|
|
154
|
+
|
|
155
|
+
function retryAfter(headers: Headers): number | undefined {
|
|
156
|
+
const ms = Number(headers.get("retry-after-ms"));
|
|
157
|
+
if (headers.has("retry-after-ms") && Number.isFinite(ms) && ms >= 0) return Math.min(ms, MAX_WAIT_MS);
|
|
158
|
+
const seconds = Number(headers.get("retry-after"));
|
|
159
|
+
if (headers.has("retry-after") && Number.isFinite(seconds) && seconds >= 0) return Math.min(seconds * 1000, MAX_WAIT_MS);
|
|
160
|
+
return undefined;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/** The readable part of an error body: TypeSafe nests it under `detail`, gateways under `error`. */
|
|
164
|
+
function summarize(text: string): string {
|
|
165
|
+
try {
|
|
166
|
+
let current: unknown = JSON.parse(text);
|
|
167
|
+
for (let depth = 0; depth < 2 && current && typeof current === "object"; depth += 1) {
|
|
168
|
+
const record = current as Record<string, unknown>;
|
|
169
|
+
const found = record.message ?? record.error ?? record.detail;
|
|
170
|
+
if (typeof found === "string") return found;
|
|
171
|
+
current = found;
|
|
172
|
+
}
|
|
173
|
+
} catch {
|
|
174
|
+
// Not JSON: the raw text below.
|
|
175
|
+
}
|
|
176
|
+
return text.slice(0, 200) || "empty response";
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
async function attempt(url: string, body: string, options: JevCallOptions): Promise<SystemOneResponse> {
|
|
180
|
+
const controller = new AbortController();
|
|
181
|
+
const onAbort = () => controller.abort(options.signal?.reason);
|
|
182
|
+
options.signal?.addEventListener("abort", onAbort, { once: true });
|
|
183
|
+
let timedOut = false;
|
|
184
|
+
const timer = setTimeout(() => {
|
|
185
|
+
timedOut = true;
|
|
186
|
+
controller.abort();
|
|
187
|
+
}, options.timeoutMs);
|
|
188
|
+
try {
|
|
189
|
+
let response: Response;
|
|
190
|
+
try {
|
|
191
|
+
response = await (options.fetch ?? globalThis.fetch)(url, {
|
|
192
|
+
method: "POST",
|
|
193
|
+
headers: { ...options.headers, Authorization: `Bearer ${options.apiKey}`, "Content-Type": "application/json", Accept: "application/json", "User-Agent": "bot-lobby" },
|
|
194
|
+
body,
|
|
195
|
+
signal: controller.signal,
|
|
196
|
+
});
|
|
197
|
+
} catch (error) {
|
|
198
|
+
if (options.signal?.aborted) throw error;
|
|
199
|
+
// Not retried: the time limit already bounds an interactive decision, and every caller has a fallback.
|
|
200
|
+
if (timedOut) throw new JevError(`timed out after ${options.timeoutMs} ms`, false);
|
|
201
|
+
throw new JevError(`could not reach ${url}: ${(error as Error).message}`, true);
|
|
202
|
+
}
|
|
203
|
+
const text = await response.text();
|
|
204
|
+
if (!response.ok) {
|
|
205
|
+
throw new JevError(`API error ${response.status}: ${summarize(text)}`, RETRYABLE.has(response.status), response.status, retryAfter(response.headers));
|
|
206
|
+
}
|
|
207
|
+
let parsed: unknown;
|
|
208
|
+
try {
|
|
209
|
+
parsed = JSON.parse(text);
|
|
210
|
+
} catch {
|
|
211
|
+
throw new JevError("the API answered with something that is not JSON", false, response.status);
|
|
212
|
+
}
|
|
213
|
+
const record = parsed as Partial<SystemOneResponse> | undefined;
|
|
214
|
+
if (!record || typeof record !== "object" || !record.answers || typeof record.answers !== "object") {
|
|
215
|
+
throw new JevError("the API answered without answers", false, response.status);
|
|
216
|
+
}
|
|
217
|
+
return { model: typeof record.model === "string" ? record.model : options.model ?? "jev", answers: record.answers, ...(record.usage ? { usage: record.usage } : {}) };
|
|
218
|
+
} finally {
|
|
219
|
+
clearTimeout(timer);
|
|
220
|
+
options.signal?.removeEventListener("abort", onAbort);
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/** Ask every question over the state in one call. Throws `JevError` (or the abort reason) on failure. */
|
|
225
|
+
export async function systemOne(request: SystemOneRequest, options: JevCallOptions): Promise<SystemOneResponse> {
|
|
226
|
+
const url = `${options.baseUrl.replace(/\/+$/, "")}${SYSTEM_ONE_PATH}`;
|
|
227
|
+
const body = JSON.stringify({ ...(options.model ? { model: options.model } : {}), ...request });
|
|
228
|
+
const retries = Math.max(0, options.retries ?? 1);
|
|
229
|
+
for (let tried = 0; ; tried += 1) {
|
|
230
|
+
options.signal?.throwIfAborted();
|
|
231
|
+
try {
|
|
232
|
+
return await attempt(url, body, options);
|
|
233
|
+
} catch (error) {
|
|
234
|
+
if (!(error instanceof JevError) || !error.retryable || tried >= retries) throw error;
|
|
235
|
+
const wait = error.waitMs ?? Math.min(MAX_WAIT_MS, 300 * 2 ** tried);
|
|
236
|
+
await (options.sleep ?? realSleep)(wait);
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
}
|