@oh-my-pi/pi-ai 18.2.2 → 18.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/dist/types/auth-storage.d.ts +10 -6
- package/dist/types/index.d.ts +1 -0
- package/dist/types/judgment/chat.d.ts +23 -0
- package/dist/types/judgment/index.d.ts +4 -0
- package/dist/types/judgment/text.d.ts +59 -0
- package/dist/types/judgment/types.d.ts +103 -0
- package/dist/types/judgment/typesafe.d.ts +53 -0
- package/dist/types/registry/oauth/github-copilot.d.ts +2 -6
- package/dist/types/registry/oauth/kimi.d.ts +2 -1
- package/dist/types/registry/oauth/types.d.ts +2 -0
- package/package.json +10 -6
- package/src/auth-storage.ts +17 -7
- package/src/index.ts +1 -0
- package/src/judgment/chat.ts +93 -0
- package/src/judgment/index.ts +4 -0
- package/src/judgment/text-judge-retry.md +3 -0
- package/src/judgment/text-judge-state.md +5 -0
- package/src/judgment/text-judge.md +41 -0
- package/src/judgment/text.ts +309 -0
- package/src/judgment/types.ts +128 -0
- package/src/judgment/typesafe.ts +173 -0
- package/src/registry/oauth/github-copilot.ts +2 -2
- package/src/registry/oauth/kimi.ts +4 -4
- package/src/registry/oauth/types.ts +2 -0
- package/src/stream.ts +36 -1
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Text bridge: answers {@link Questions} with any model that completes text.
|
|
3
|
+
*
|
|
4
|
+
* Questions render into one system prompt asking for keyword answers (an
|
|
5
|
+
* option label, `yes`/`no`, or a level number); the state is the user message,
|
|
6
|
+
* so the question part stays byte-identical across calls and prompt caches
|
|
7
|
+
* hit. Several questions batch into a single completion answered one line per
|
|
8
|
+
* question id. Answers are one-hot: a parsed keyword yields probability 1 and
|
|
9
|
+
* confidence 1, since a text completion carries no distribution.
|
|
10
|
+
*
|
|
11
|
+
* The {@link TextBackend} decides *how* text is completed — a chat model via
|
|
12
|
+
* {@link chatTextBackend}, or an on-device worker — so this file never
|
|
13
|
+
* depends on chat transports.
|
|
14
|
+
*/
|
|
15
|
+
import { escapeXmlAttribute, escapeXmlText, prompt } from "@oh-my-pi/pi-utils";
|
|
16
|
+
import { YAML } from "bun";
|
|
17
|
+
import type { Usage } from "../types";
|
|
18
|
+
import textJudgeRetryTemplate from "./text-judge-retry.md" with { type: "text" };
|
|
19
|
+
import textJudgeStateTemplate from "./text-judge-state.md" with { type: "text" };
|
|
20
|
+
import textJudgeTemplate from "./text-judge.md" with { type: "text" };
|
|
21
|
+
import {
|
|
22
|
+
type Answer,
|
|
23
|
+
type Judge,
|
|
24
|
+
type JudgeOptions,
|
|
25
|
+
type JudgmentRequest,
|
|
26
|
+
type JudgmentResult,
|
|
27
|
+
JudgmentParseError,
|
|
28
|
+
type JudgmentState,
|
|
29
|
+
type JsonValue,
|
|
30
|
+
type Question,
|
|
31
|
+
type Questions,
|
|
32
|
+
tokenUsage,
|
|
33
|
+
} from "./types";
|
|
34
|
+
|
|
35
|
+
export interface TextPrompt {
|
|
36
|
+
system: string;
|
|
37
|
+
user: string;
|
|
38
|
+
/** Format-correction attempt; chat backends may enforce tool suppression on the wire. */
|
|
39
|
+
retry?: boolean;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export interface TextCompletion {
|
|
43
|
+
text: string;
|
|
44
|
+
/** Omitted by backends that do not meter tokens (on-device workers). */
|
|
45
|
+
usage?: Usage;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/** Completes a rendered judgment prompt; identifies itself for usage attribution. */
|
|
49
|
+
export interface TextBackend {
|
|
50
|
+
readonly api: string;
|
|
51
|
+
readonly provider: string;
|
|
52
|
+
readonly model: string;
|
|
53
|
+
/** State guard for agent-tuned chat models; on-device classifier workers may disable it. */
|
|
54
|
+
readonly guardState?: boolean;
|
|
55
|
+
/** Format-correction retries after a completion cannot be parsed. */
|
|
56
|
+
readonly parseRetries?: number;
|
|
57
|
+
complete(prompt: TextPrompt, options: JudgeOptions): Promise<TextCompletion>;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
interface RenderedQuestion {
|
|
61
|
+
id: string;
|
|
62
|
+
instructions: string;
|
|
63
|
+
options?: { label: string; description: string | null }[];
|
|
64
|
+
levels?: { index: number; description: string }[];
|
|
65
|
+
yesno?: boolean;
|
|
66
|
+
yes?: string;
|
|
67
|
+
no?: string;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
const XML_TAG_NAME = /^[A-Za-z_][A-Za-z0-9_.-]*$/;
|
|
71
|
+
|
|
72
|
+
function isJsonArray(value: JudgmentState): value is readonly JsonValue[] {
|
|
73
|
+
return Array.isArray(value);
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** Render judgment state as top-level XML fields; nested values use block YAML. */
|
|
77
|
+
export function renderJudgmentState(state: JudgmentState): string {
|
|
78
|
+
if (typeof state !== "object" || state === null || isJsonArray(state)) {
|
|
79
|
+
return renderStateField("state", state);
|
|
80
|
+
}
|
|
81
|
+
const fields: string[] = [];
|
|
82
|
+
for (const key in state) {
|
|
83
|
+
if (!Object.hasOwn(state, key)) continue;
|
|
84
|
+
fields.push(renderStateField(key, state[key]));
|
|
85
|
+
}
|
|
86
|
+
return fields.length > 0 ? fields.join("\n") : "<state>{}</state>";
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
function renderStateField(key: string, value: JsonValue): string {
|
|
90
|
+
const validTag = XML_TAG_NAME.test(key);
|
|
91
|
+
const open = validTag ? `<${key}>` : `<field name="${escapeXmlAttribute(key)}">`;
|
|
92
|
+
const close = validTag ? `</${key}>` : "</field>";
|
|
93
|
+
if (typeof value !== "object" || value === null) {
|
|
94
|
+
return `${open}${escapeXmlText(value === null ? "null" : String(value))}${close}`;
|
|
95
|
+
}
|
|
96
|
+
const yaml = YAML.stringify(value, null, 2).trimEnd();
|
|
97
|
+
return `${open}\n${escapeXmlText(yaml)}\n${close}`;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* Render a request: the system prompt carries the preamble and question
|
|
102
|
+
* definitions (constant across states, so prompt caches hit); the user message
|
|
103
|
+
* carries XML-field state followed by the answer-format cue. Small models act
|
|
104
|
+
* on a bare request (answer it, emit tool calls) instead of classifying it —
|
|
105
|
+
* explicit tags and the trailing cue keep them on task.
|
|
106
|
+
*/
|
|
107
|
+
export function renderJudgmentPrompt(request: JudgmentRequest, options: { guardState?: boolean } = {}): TextPrompt {
|
|
108
|
+
const questions: RenderedQuestion[] = [];
|
|
109
|
+
for (const id in request.questions) {
|
|
110
|
+
const question = request.questions[id];
|
|
111
|
+
const rendered: RenderedQuestion = { id, instructions: question.instructions };
|
|
112
|
+
switch (question.type) {
|
|
113
|
+
case "choice": {
|
|
114
|
+
const options: RenderedQuestion["options"] = [];
|
|
115
|
+
for (const label in question.criteria) options.push({ label, description: question.criteria[label] });
|
|
116
|
+
rendered.options = options;
|
|
117
|
+
break;
|
|
118
|
+
}
|
|
119
|
+
case "score":
|
|
120
|
+
rendered.levels = question.criteria.map((description, index) => ({ index, description }));
|
|
121
|
+
break;
|
|
122
|
+
case "noul":
|
|
123
|
+
rendered.yesno = true;
|
|
124
|
+
rendered.yes = question.criteria?.true;
|
|
125
|
+
rendered.no = question.criteria?.false;
|
|
126
|
+
break;
|
|
127
|
+
}
|
|
128
|
+
questions.push(rendered);
|
|
129
|
+
}
|
|
130
|
+
const multi = questions.length > 1;
|
|
131
|
+
const guardState = options.guardState !== false;
|
|
132
|
+
return {
|
|
133
|
+
system: prompt.render(textJudgeTemplate, { questions, multi, guardState }),
|
|
134
|
+
user: prompt.render(textJudgeStateTemplate, {
|
|
135
|
+
...questions[0],
|
|
136
|
+
guardState,
|
|
137
|
+
multi,
|
|
138
|
+
state: renderJudgmentState(request.state),
|
|
139
|
+
}),
|
|
140
|
+
};
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
const WORD = /[\p{L}\p{N}_]/u;
|
|
144
|
+
|
|
145
|
+
/** Index of the earliest whole-word, case-insensitive occurrence of `needle` in `text`, or -1. */
|
|
146
|
+
function indexOfWord(text: string, needle: string): number {
|
|
147
|
+
const lower = text.toLowerCase();
|
|
148
|
+
const target = needle.toLowerCase();
|
|
149
|
+
let from = 0;
|
|
150
|
+
while (from <= lower.length - target.length) {
|
|
151
|
+
const at = lower.indexOf(target, from);
|
|
152
|
+
if (at < 0) return -1;
|
|
153
|
+
const before = at > 0 ? lower[at - 1] : "";
|
|
154
|
+
const after = lower[at + target.length] ?? "";
|
|
155
|
+
const boundedBefore = before === "" || !WORD.test(before) || !WORD.test(target[0]);
|
|
156
|
+
const boundedAfter = after === "" || !WORD.test(after) || !WORD.test(target[target.length - 1]);
|
|
157
|
+
if (boundedBefore && boundedAfter) return at;
|
|
158
|
+
from = at + 1;
|
|
159
|
+
}
|
|
160
|
+
return -1;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* Earliest option label mentioned in `text`; a longer label wins a tie at the
|
|
165
|
+
* same position (so `xhigh` beats `high` where both start together).
|
|
166
|
+
*/
|
|
167
|
+
export function parseChoiceReply<L extends string>(text: string, labels: readonly L[]): L | undefined {
|
|
168
|
+
let best: L | undefined;
|
|
169
|
+
let bestAt = Number.POSITIVE_INFINITY;
|
|
170
|
+
for (const label of labels) {
|
|
171
|
+
const at = indexOfWord(text, label);
|
|
172
|
+
if (at < 0) continue;
|
|
173
|
+
if (at < bestAt || (at === bestAt && best !== undefined && label.length > best.length)) {
|
|
174
|
+
best = label;
|
|
175
|
+
bestAt = at;
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
return best;
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/** `true` when a yes-word precedes any no-word, `false` for the reverse, `undefined` when neither appears. */
|
|
182
|
+
export function parseNoulReply(text: string): boolean | undefined {
|
|
183
|
+
const yes = [indexOfWord(text, "yes"), indexOfWord(text, "true")].filter(at => at >= 0);
|
|
184
|
+
const no = [indexOfWord(text, "no"), indexOfWord(text, "false")].filter(at => at >= 0);
|
|
185
|
+
const yesAt = yes.length > 0 ? Math.min(...yes) : -1;
|
|
186
|
+
const noAt = no.length > 0 ? Math.min(...no) : -1;
|
|
187
|
+
if (yesAt < 0 && noAt < 0) return undefined;
|
|
188
|
+
if (noAt < 0) return true;
|
|
189
|
+
if (yesAt < 0) return false;
|
|
190
|
+
return yesAt < noAt;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/** Standalone integers: not glued to a word (`v2`) and not one side of a decimal (`1.5`); a trailing period is fine. */
|
|
194
|
+
const INTEGER = /(?<![\p{L}\p{N}_])(?<!\d\.)(\d+)(?![\p{L}\p{N}_])(?!\.\d)/gu;
|
|
195
|
+
|
|
196
|
+
/** First standalone integer in `[0, levels)`, or `undefined`. */
|
|
197
|
+
export function parseScoreReply(text: string, levels: number): number | undefined {
|
|
198
|
+
for (const match of text.matchAll(INTEGER)) {
|
|
199
|
+
const value = Number(match[1]);
|
|
200
|
+
if (value < levels) return value;
|
|
201
|
+
}
|
|
202
|
+
return undefined;
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/**
|
|
206
|
+
* Split a multi-question reply into `id → answer text`. Lines are matched as
|
|
207
|
+
* `<id>: <answer>` (also `=` / `-` separators and quoted ids); unknown ids are
|
|
208
|
+
* ignored so a chatty preamble does not poison parsing.
|
|
209
|
+
*/
|
|
210
|
+
export function splitAnswerLines(text: string, ids: readonly string[]): Map<string, string> {
|
|
211
|
+
const byId = new Map<string, string>();
|
|
212
|
+
const wanted = new Set(ids);
|
|
213
|
+
for (const rawLine of text.split("\n")) {
|
|
214
|
+
const line = rawLine.replace(/^[\s\-*•]+/, "").trim();
|
|
215
|
+
const separator = line.search(/\s*[:=]\s*|\s+-\s+/);
|
|
216
|
+
if (separator <= 0) continue;
|
|
217
|
+
const id = line.slice(0, separator).replace(/^[`"']|[`"']$/g, "");
|
|
218
|
+
if (!wanted.has(id) || byId.has(id)) continue;
|
|
219
|
+
byId.set(id, line.slice(separator).replace(/^\s*[:=-]\s*/, ""));
|
|
220
|
+
}
|
|
221
|
+
return byId;
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
function oneHot<L extends string>(labels: readonly L[], chosen: L): Record<L, number> {
|
|
225
|
+
const probabilities = {} as Record<L, number>;
|
|
226
|
+
for (const label of labels) probabilities[label] = label === chosen ? 1 : 0;
|
|
227
|
+
return probabilities;
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
/** Parse one question's keyword reply into its typed, one-hot answer. */
|
|
231
|
+
export function parseAnswer(id: string, question: Question, reply: string): Answer {
|
|
232
|
+
switch (question.type) {
|
|
233
|
+
case "choice": {
|
|
234
|
+
const labels = Object.keys(question.criteria);
|
|
235
|
+
const choice = parseChoiceReply(reply, labels);
|
|
236
|
+
if (choice === undefined) throw new JudgmentParseError(id, reply, "no option label in reply");
|
|
237
|
+
return { type: "choice", choice, probabilities: oneHot(labels, choice), confidence: 1 };
|
|
238
|
+
}
|
|
239
|
+
case "noul": {
|
|
240
|
+
const verdict = parseNoulReply(reply);
|
|
241
|
+
if (verdict === undefined) throw new JudgmentParseError(id, reply, "no yes/no in reply");
|
|
242
|
+
return { type: "noul", noul: verdict ? 1 : 0 };
|
|
243
|
+
}
|
|
244
|
+
case "score": {
|
|
245
|
+
const level = parseScoreReply(reply, question.criteria.length);
|
|
246
|
+
if (level === undefined) throw new JudgmentParseError(id, reply, "no level number in reply");
|
|
247
|
+
const probabilities: Record<string, number> = {};
|
|
248
|
+
for (let index = 0; index < question.criteria.length; index++) {
|
|
249
|
+
probabilities[String(index)] = index === level ? 1 : 0;
|
|
250
|
+
}
|
|
251
|
+
return { type: "score", score: level, probabilities, confidence: 1 };
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
export class TextJudge implements Judge {
|
|
257
|
+
readonly label: string;
|
|
258
|
+
readonly #backend: TextBackend;
|
|
259
|
+
|
|
260
|
+
constructor(backend: TextBackend) {
|
|
261
|
+
this.#backend = backend;
|
|
262
|
+
this.label = `${backend.provider}/${backend.model}`;
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
async judge<Q extends Questions>(
|
|
266
|
+
request: JudgmentRequest<Q>,
|
|
267
|
+
options: JudgeOptions = {},
|
|
268
|
+
): Promise<JudgmentResult<Q>> {
|
|
269
|
+
const ids = Object.keys(request.questions);
|
|
270
|
+
if (ids.length === 0) throw new Error("judgment request has no questions");
|
|
271
|
+
const rendered = renderJudgmentPrompt(request, { guardState: this.#backend.guardState });
|
|
272
|
+
const retries = this.#backend.parseRetries ?? 0;
|
|
273
|
+
for (let attempt = 0; ; attempt++) {
|
|
274
|
+
const textPrompt =
|
|
275
|
+
attempt === 0
|
|
276
|
+
? rendered
|
|
277
|
+
: {
|
|
278
|
+
system: prompt.render(textJudgeRetryTemplate, {
|
|
279
|
+
system: rendered.system,
|
|
280
|
+
multi: ids.length > 1,
|
|
281
|
+
}),
|
|
282
|
+
user: rendered.user,
|
|
283
|
+
retry: true,
|
|
284
|
+
};
|
|
285
|
+
const completion = await this.#backend.complete(textPrompt, options);
|
|
286
|
+
try {
|
|
287
|
+
const replies =
|
|
288
|
+
ids.length === 1 ? new Map([[ids[0], completion.text]]) : splitAnswerLines(completion.text, ids);
|
|
289
|
+
const answers: Record<string, Answer> = {};
|
|
290
|
+
for (const id of ids) {
|
|
291
|
+
const reply = replies.get(id);
|
|
292
|
+
if (reply === undefined) {
|
|
293
|
+
throw new JudgmentParseError(id, completion.text, "no answer line for question");
|
|
294
|
+
}
|
|
295
|
+
answers[id] = parseAnswer(id, request.questions[id], reply);
|
|
296
|
+
}
|
|
297
|
+
return {
|
|
298
|
+
api: this.#backend.api,
|
|
299
|
+
provider: this.#backend.provider,
|
|
300
|
+
model: this.#backend.model,
|
|
301
|
+
answers: answers as JudgmentResult<Q>["answers"],
|
|
302
|
+
usage: completion.usage ?? tokenUsage(0, 0),
|
|
303
|
+
};
|
|
304
|
+
} catch (error) {
|
|
305
|
+
if (!(error instanceof JudgmentParseError) || attempt >= retries) throw error;
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
}
|
|
309
|
+
}
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Typed judgments: small, structured decisions over a JSON state.
|
|
3
|
+
*
|
|
4
|
+
* A {@link Judge} answers a map of named questions — {@link ChoiceQuestion}
|
|
5
|
+
* (one option from a fixed set), {@link NoulQuestion} (probability that a
|
|
6
|
+
* yes/no condition holds), {@link ScoreQuestion} (position on ordered levels)
|
|
7
|
+
* — about one {@link JudgmentState}. Every question in a request sees the same
|
|
8
|
+
* state and is answered independently, so callers batch independent questions
|
|
9
|
+
* into one call. The shape mirrors TypeSafe's System One API so the native
|
|
10
|
+
* backend ({@link TypeSafeJudge}) forwards requests verbatim, while the text
|
|
11
|
+
* bridge ({@link TextJudge}) renders the same questions into keyword prompts
|
|
12
|
+
* for an ordinary chat model.
|
|
13
|
+
*/
|
|
14
|
+
import type { Usage } from "../types";
|
|
15
|
+
|
|
16
|
+
/** JSON-compatible value; readonly containers are accepted so `as const` state passes through. */
|
|
17
|
+
export type JsonValue = string | number | boolean | null | readonly JsonValue[] | { readonly [key: string]: JsonValue };
|
|
18
|
+
|
|
19
|
+
/** The content a request evaluates: plain text, or a named-field object / array for multi-part context. */
|
|
20
|
+
export type JudgmentState = string | { readonly [key: string]: JsonValue } | readonly JsonValue[];
|
|
21
|
+
|
|
22
|
+
/** Pick one option from a fixed set. `criteria` maps option → rubric (`null` when the name suffices). */
|
|
23
|
+
export interface ChoiceQuestion<L extends string = string> {
|
|
24
|
+
type: "choice";
|
|
25
|
+
instructions: string;
|
|
26
|
+
criteria: Record<L, string | null>;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** Probability that a yes/no condition holds. `criteria` optionally spells out what yes and no mean. */
|
|
30
|
+
export interface NoulQuestion {
|
|
31
|
+
type: "noul";
|
|
32
|
+
instructions: string;
|
|
33
|
+
criteria?: { true?: string; false?: string };
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/** Position on ordered levels; `criteria` lists at least two level descriptions from lowest to highest. */
|
|
37
|
+
export interface ScoreQuestion {
|
|
38
|
+
type: "score";
|
|
39
|
+
instructions: string;
|
|
40
|
+
criteria: readonly [string, string, ...string[]];
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export type Question = ChoiceQuestion | NoulQuestion | ScoreQuestion;
|
|
44
|
+
|
|
45
|
+
/** Questions keyed by caller-chosen ids; answers come back under the same ids. */
|
|
46
|
+
export type Questions = Record<string, Question>;
|
|
47
|
+
|
|
48
|
+
export interface ChoiceAnswer<L extends string = string> {
|
|
49
|
+
type: "choice";
|
|
50
|
+
/** Highest-probability option. */
|
|
51
|
+
choice: L;
|
|
52
|
+
/** Every option mapped to its probability (sums to 1). */
|
|
53
|
+
probabilities: Record<L, number>;
|
|
54
|
+
/** Concentration of the distribution, 0–1. */
|
|
55
|
+
confidence: number;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export interface NoulAnswer {
|
|
59
|
+
type: "noul";
|
|
60
|
+
/** Probability of yes, 0–1. */
|
|
61
|
+
noul: number;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
export interface ScoreAnswer {
|
|
65
|
+
type: "score";
|
|
66
|
+
/** Probability-weighted level index; may land between levels. */
|
|
67
|
+
score: number;
|
|
68
|
+
/** Level index (as a string key) mapped to its probability. */
|
|
69
|
+
probabilities: Record<string, number>;
|
|
70
|
+
/** Concentration of the distribution, 0–1. */
|
|
71
|
+
confidence: number;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
export type Answer = ChoiceAnswer | NoulAnswer | ScoreAnswer;
|
|
75
|
+
|
|
76
|
+
/** The answer type for a question, preserving choice labels. */
|
|
77
|
+
export type AnswerFor<Q extends Question> =
|
|
78
|
+
Q extends ChoiceQuestion<infer L> ? ChoiceAnswer<L> : Q extends NoulQuestion ? NoulAnswer : ScoreAnswer;
|
|
79
|
+
|
|
80
|
+
export interface JudgmentRequest<Q extends Questions = Questions> {
|
|
81
|
+
state: JudgmentState;
|
|
82
|
+
questions: Q;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
export interface JudgmentResult<Q extends Questions = Questions> {
|
|
86
|
+
/** Transport that produced the answers (`typesafe`, or the chat model's api). */
|
|
87
|
+
api: string;
|
|
88
|
+
provider: string;
|
|
89
|
+
model: string;
|
|
90
|
+
answers: { [K in keyof Q]: AnswerFor<Q[K]> };
|
|
91
|
+
usage: Usage;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
export interface JudgeOptions {
|
|
95
|
+
signal?: AbortSignal;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/** Answers typed questions about a state. Implementations are stateless and safe to share. */
|
|
99
|
+
export interface Judge {
|
|
100
|
+
/** Backend description for logs, e.g. `typesafe/jev-latest`. */
|
|
101
|
+
readonly label: string;
|
|
102
|
+
judge<Q extends Questions>(request: JudgmentRequest<Q>, options?: JudgeOptions): Promise<JudgmentResult<Q>>;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** Thrown when a backend returned output that does not resolve to an answer for every question. */
|
|
106
|
+
export class JudgmentParseError extends Error {
|
|
107
|
+
readonly questionId: string;
|
|
108
|
+
readonly output: string;
|
|
109
|
+
|
|
110
|
+
constructor(questionId: string, output: string, detail: string) {
|
|
111
|
+
super(`judgment "${questionId}": ${detail}: ${JSON.stringify(output)}`);
|
|
112
|
+
this.name = "JudgmentParseError";
|
|
113
|
+
this.questionId = questionId;
|
|
114
|
+
this.output = output;
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/** Zero-cost usage for a request whose backend reports only token counts. */
|
|
119
|
+
export function tokenUsage(input: number, output: number): Usage {
|
|
120
|
+
return {
|
|
121
|
+
input,
|
|
122
|
+
output,
|
|
123
|
+
cacheRead: 0,
|
|
124
|
+
cacheWrite: 0,
|
|
125
|
+
totalTokens: input + output,
|
|
126
|
+
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
|
127
|
+
};
|
|
128
|
+
}
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* TypeSafe System One client: the native {@link Judge} backend.
|
|
3
|
+
*
|
|
4
|
+
* Forwards a {@link JudgmentRequest} verbatim to `POST /v1/systemone` and maps
|
|
5
|
+
* the typed answers back. Credentials flow through {@link withAuth}, so a
|
|
6
|
+
* stored key rotates on 401/403 exactly like chat providers; transient
|
|
7
|
+
* 429/5xx responses retry with bounded, `retry-after`-aware backoff.
|
|
8
|
+
*
|
|
9
|
+
* Environment (mirrors the official SDK): `TYPESAFE_API_KEY` is resolved by
|
|
10
|
+
* the auth registry (`rules/auth/typesafe.kdl`), `TYPESAFE_BASE_URL`
|
|
11
|
+
* overrides the API root, `TYPESAFE_DEFAULT_MODEL` the model.
|
|
12
|
+
*/
|
|
13
|
+
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
|
|
14
|
+
import { $env } from "@oh-my-pi/pi-utils";
|
|
15
|
+
import { type ApiKey, withAuth } from "../auth-retry";
|
|
16
|
+
import * as AIError from "../error";
|
|
17
|
+
import { getRetryAfterMsFromHeaders } from "../utils/retry-after";
|
|
18
|
+
import {
|
|
19
|
+
type Answer,
|
|
20
|
+
type Judge,
|
|
21
|
+
type JudgeOptions,
|
|
22
|
+
type JudgmentRequest,
|
|
23
|
+
type JudgmentResult,
|
|
24
|
+
type Questions,
|
|
25
|
+
tokenUsage,
|
|
26
|
+
} from "./types";
|
|
27
|
+
|
|
28
|
+
export const TYPESAFE_PROVIDER = "typesafe";
|
|
29
|
+
export const TYPESAFE_DEFAULT_BASE_URL = "https://api.typesafe.ai";
|
|
30
|
+
export const TYPESAFE_DEFAULT_MODEL = "jev-latest";
|
|
31
|
+
|
|
32
|
+
/** `TYPESAFE_BASE_URL` when set, else the public API root; trailing slashes stripped. */
|
|
33
|
+
export function typesafeBaseUrl(): string {
|
|
34
|
+
return ($env.TYPESAFE_BASE_URL?.trim() || TYPESAFE_DEFAULT_BASE_URL).replace(/\/+$/, "");
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** `TYPESAFE_DEFAULT_MODEL` when set, else {@link TYPESAFE_DEFAULT_MODEL}. */
|
|
38
|
+
export function typesafeModel(): string {
|
|
39
|
+
return $env.TYPESAFE_DEFAULT_MODEL?.trim() || TYPESAFE_DEFAULT_MODEL;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export interface TypeSafeJudgeOptions {
|
|
43
|
+
apiKey: ApiKey;
|
|
44
|
+
/** Defaults to {@link typesafeBaseUrl}. */
|
|
45
|
+
baseUrl?: string;
|
|
46
|
+
/** Defaults to {@link typesafeModel}. */
|
|
47
|
+
model?: string;
|
|
48
|
+
fetch?: FetchImpl;
|
|
49
|
+
/** Per-attempt timeout; defaults to {@link DEFAULT_TIMEOUT_MS}. */
|
|
50
|
+
timeoutMs?: number;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/** Non-2xx response from the TypeSafe API. */
|
|
54
|
+
export class TypeSafeApiError extends AIError.ProviderHttpError {
|
|
55
|
+
override readonly name = "TypeSafeApiError";
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
const DEFAULT_TIMEOUT_MS = 10_000;
|
|
59
|
+
const MAX_ATTEMPTS = 3;
|
|
60
|
+
const BACKOFF_BASE_MS = 500;
|
|
61
|
+
const BACKOFF_MAX_MS = 5_000;
|
|
62
|
+
|
|
63
|
+
/** Wire shape of `GET /v1/models`. */
|
|
64
|
+
export interface TypeSafeModelCard {
|
|
65
|
+
name: string;
|
|
66
|
+
description: string;
|
|
67
|
+
release_date: string;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
interface SystemOneResponse {
|
|
71
|
+
model: string;
|
|
72
|
+
answers: Record<string, Answer>;
|
|
73
|
+
usage: { input_tokens: number; output_tokens: number };
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** Server hint wins (capped); otherwise exponential backoff from {@link BACKOFF_BASE_MS}. */
|
|
77
|
+
function backoffMs(attempt: number, headers: Headers | undefined): number {
|
|
78
|
+
const hinted = headers === undefined ? undefined : getRetryAfterMsFromHeaders(headers);
|
|
79
|
+
if (hinted !== undefined) return Math.min(hinted, BACKOFF_MAX_MS);
|
|
80
|
+
return Math.min(BACKOFF_BASE_MS * 2 ** attempt, BACKOFF_MAX_MS);
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
export class TypeSafeJudge implements Judge {
|
|
84
|
+
readonly label: string;
|
|
85
|
+
readonly model: string;
|
|
86
|
+
readonly baseUrl: string;
|
|
87
|
+
readonly #apiKey: ApiKey;
|
|
88
|
+
readonly #fetch: FetchImpl;
|
|
89
|
+
readonly #timeoutMs: number;
|
|
90
|
+
|
|
91
|
+
constructor(options: TypeSafeJudgeOptions) {
|
|
92
|
+
this.#apiKey = options.apiKey;
|
|
93
|
+
this.baseUrl = (options.baseUrl ?? typesafeBaseUrl()).replace(/\/+$/, "");
|
|
94
|
+
this.model = options.model ?? typesafeModel();
|
|
95
|
+
this.#fetch = options.fetch ?? fetch;
|
|
96
|
+
this.#timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS;
|
|
97
|
+
this.label = `${TYPESAFE_PROVIDER}/${this.model}`;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
async judge<Q extends Questions>(request: JudgmentRequest<Q>, options?: JudgeOptions): Promise<JudgmentResult<Q>> {
|
|
101
|
+
const body = JSON.stringify({ state: request.state, model: this.model, questions: request.questions });
|
|
102
|
+
const response = await this.#request<SystemOneResponse>("POST", "/v1/systemone", body, options?.signal);
|
|
103
|
+
for (const id in request.questions) {
|
|
104
|
+
const answer = response.answers[id];
|
|
105
|
+
if (answer === undefined || answer.type !== request.questions[id].type) {
|
|
106
|
+
throw new AIError.ProviderResponseError(
|
|
107
|
+
`TypeSafe response is missing a "${request.questions[id].type}" answer for question "${id}"`,
|
|
108
|
+
{ provider: TYPESAFE_PROVIDER, kind: "envelope" },
|
|
109
|
+
);
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
return {
|
|
113
|
+
api: TYPESAFE_PROVIDER,
|
|
114
|
+
provider: TYPESAFE_PROVIDER,
|
|
115
|
+
model: response.model,
|
|
116
|
+
answers: response.answers as JudgmentResult<Q>["answers"],
|
|
117
|
+
usage: tokenUsage(response.usage.input_tokens, response.usage.output_tokens),
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/** Models available to the account (`GET /v1/models`); also the login validation probe. */
|
|
122
|
+
async listModels(signal?: AbortSignal): Promise<TypeSafeModelCard[]> {
|
|
123
|
+
const response = await this.#request<{ models: TypeSafeModelCard[] }>("GET", "/v1/models", undefined, signal);
|
|
124
|
+
if (!Array.isArray(response.models)) {
|
|
125
|
+
throw new AIError.ProviderResponseError("TypeSafe /v1/models response is missing `models`", {
|
|
126
|
+
provider: TYPESAFE_PROVIDER,
|
|
127
|
+
kind: "envelope",
|
|
128
|
+
});
|
|
129
|
+
}
|
|
130
|
+
return response.models;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
async #request<T>(method: "GET" | "POST", path: string, body: string | undefined, signal?: AbortSignal): Promise<T> {
|
|
134
|
+
return withAuth(this.#apiKey, key => this.#attempt<T>(method, path, body, key, signal), { signal });
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
async #attempt<T>(
|
|
138
|
+
method: "GET" | "POST",
|
|
139
|
+
path: string,
|
|
140
|
+
body: string | undefined,
|
|
141
|
+
key: string,
|
|
142
|
+
signal: AbortSignal | undefined,
|
|
143
|
+
): Promise<T> {
|
|
144
|
+
const url = `${this.baseUrl}${path}`;
|
|
145
|
+
const headers: Record<string, string> = { Authorization: `Bearer ${key}`, Accept: "application/json" };
|
|
146
|
+
if (body !== undefined) headers["Content-Type"] = "application/json";
|
|
147
|
+
for (let attempt = 0; ; attempt++) {
|
|
148
|
+
signal?.throwIfAborted();
|
|
149
|
+
const timeout = AbortSignal.timeout(this.#timeoutMs);
|
|
150
|
+
let response: Response;
|
|
151
|
+
try {
|
|
152
|
+
response = await this.#fetch(url, {
|
|
153
|
+
method,
|
|
154
|
+
headers,
|
|
155
|
+
body,
|
|
156
|
+
signal: signal ? AbortSignal.any([signal, timeout]) : timeout,
|
|
157
|
+
});
|
|
158
|
+
} catch (error) {
|
|
159
|
+
if (signal?.aborted || attempt + 1 >= MAX_ATTEMPTS) throw error;
|
|
160
|
+
await Bun.sleep(backoffMs(attempt, undefined));
|
|
161
|
+
continue;
|
|
162
|
+
}
|
|
163
|
+
if (response.ok) return (await response.json()) as T;
|
|
164
|
+
const text = await response.text();
|
|
165
|
+
const error = new TypeSafeApiError(`TypeSafe API error (${response.status}): ${text}`, response.status, {
|
|
166
|
+
headers: response.headers,
|
|
167
|
+
});
|
|
168
|
+
const transient = response.status === 408 || response.status === 429 || response.status >= 500;
|
|
169
|
+
if (!transient || attempt + 1 >= MAX_ATTEMPTS) throw error;
|
|
170
|
+
await Bun.sleep(backoffMs(attempt, response.headers));
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
}
|
|
@@ -29,7 +29,7 @@ import {
|
|
|
29
29
|
} from "../../providers/github-copilot-headers";
|
|
30
30
|
import * as AIError from "../../error";
|
|
31
31
|
import type { FetchImpl } from "../../types";
|
|
32
|
-
import type { OAuthController, OAuthCredentials } from "./types";
|
|
32
|
+
import type { OAuthController, OAuthCredentials, OAuthPrompt } from "./types";
|
|
33
33
|
|
|
34
34
|
const OPENCODE_CLIENT_ID = "Ov23li8tweQw6odWQebz";
|
|
35
35
|
const COPILOT_CLI_CLIENT_ID = "Ov23ctDVkRmgkPke0Mmm";
|
|
@@ -55,7 +55,7 @@ const SLOW_DOWN_POLL_INTERVAL_MULTIPLIER = 1.4;
|
|
|
55
55
|
|
|
56
56
|
type GitHubCopilotLoginOptions = {
|
|
57
57
|
onAuth: (url: string, instructions?: string) => void;
|
|
58
|
-
onPrompt: (prompt:
|
|
58
|
+
onPrompt: (prompt: OAuthPrompt) => Promise<string>;
|
|
59
59
|
onProgress?: (message: string) => void;
|
|
60
60
|
copilotIntegrationId?: unknown;
|
|
61
61
|
signal?: AbortSignal;
|
|
@@ -6,7 +6,7 @@ import * as crypto from "node:crypto";
|
|
|
6
6
|
import * as fs from "node:fs";
|
|
7
7
|
import * as os from "node:os";
|
|
8
8
|
import * as path from "node:path";
|
|
9
|
-
import { getAgentDir } from "@oh-my-pi/pi-utils";
|
|
9
|
+
import { getAgentDir, once } from "@oh-my-pi/pi-utils";
|
|
10
10
|
import packageJson from "../../../package.json" with { type: "json" };
|
|
11
11
|
|
|
12
12
|
const DEVICE_ID_FILENAME = "kimi-device-id";
|
|
@@ -57,7 +57,8 @@ function sanitizeHeaderValue(value: string, fallback = ""): string {
|
|
|
57
57
|
return sanitized || fallback;
|
|
58
58
|
}
|
|
59
59
|
|
|
60
|
-
|
|
60
|
+
/** Lazily resolve the process-stable device headers used by Kimi requests. */
|
|
61
|
+
export const getKimiCommonHeaders = once(() => {
|
|
61
62
|
const headers = Object.freeze({
|
|
62
63
|
"User-Agent": `KimiCLI/${packageJson.version}`,
|
|
63
64
|
"X-Msh-Platform": "kimi_cli",
|
|
@@ -67,6 +68,5 @@ export let getKimiCommonHeaders = () => {
|
|
|
67
68
|
"X-Msh-Os-Version": sanitizeHeaderValue(os.version(), "unknown"),
|
|
68
69
|
"X-Msh-Device-Id": sanitizeHeaderValue(getDeviceId(), "unknown"),
|
|
69
70
|
});
|
|
70
|
-
getKimiCommonHeaders = () => headers;
|
|
71
71
|
return headers;
|
|
72
|
-
};
|
|
72
|
+
});
|
|
@@ -37,6 +37,8 @@ export type OAuthPrompt = {
|
|
|
37
37
|
message: string;
|
|
38
38
|
placeholder?: string;
|
|
39
39
|
allowEmpty?: boolean;
|
|
40
|
+
/** Request masked entry from interactive hosts. Hosts that cannot hide input must reject the prompt. */
|
|
41
|
+
secret?: boolean;
|
|
40
42
|
};
|
|
41
43
|
|
|
42
44
|
export type OAuthAuthInfo = {
|