@ggui-ai/negotiator 0.20.0 → 0.21.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/llm-caller.d.ts +14 -8
- package/dist/llm-caller.d.ts.map +1 -1
- package/dist/llm-caller.js +9 -5
- package/dist/rerank-eval/run-probe-cli.js +6 -75
- package/dist/synth-bench/cli-llm.d.ts +5 -0
- package/dist/synth-bench/cli-llm.d.ts.map +1 -1
- package/dist/synth-bench/cli-llm.js +34 -7
- package/package.json +3 -3
- package/src/llm-caller.ts +14 -8
- package/src/rerank-eval/run-probe-cli.ts +10 -106
- package/src/synth-bench/cli-llm.ts +35 -7
package/dist/llm-caller.d.ts
CHANGED
|
@@ -25,11 +25,15 @@
|
|
|
25
25
|
* the caller didn't request them — the text path is used as a
|
|
26
26
|
* regex-JSON fallback.
|
|
27
27
|
* - `callStructured?<T>(...)` is OPTIONAL. When present, it MUST
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
28
|
+
* return the input of the supplied `ToolSchema`'s tool, parsed as
|
|
29
|
+
* `T`, or THROW — never return anything else. It forces the tool
|
|
30
|
+
* where the model allows that; a model that refuses a forced tool
|
|
31
|
+
* (the always-thinking family) is asked for it without forcing, so
|
|
32
|
+
* the call can end with no tool input, and that ending is a throw.
|
|
33
|
+
* Consumers validate what they get back either way. Implementations
|
|
34
|
+
* that can't produce tool input at all simply omit this method;
|
|
35
|
+
* consumers fall back to `call` + regex JSON extraction. Absence is
|
|
36
|
+
* not an error.
|
|
33
37
|
* - `ToolSchema.input_schema` follows the OpenAI tool-use JSON
|
|
34
38
|
* Schema convention. Implementations that use a different
|
|
35
39
|
* tool-use protocol (e.g., Anthropic's variant) MUST translate at
|
|
@@ -49,9 +53,11 @@ export interface LLMCaller {
|
|
|
49
53
|
*/
|
|
50
54
|
call(systemPrompt: string, userMessage: string, maxTokens?: number): Promise<string>;
|
|
51
55
|
/**
|
|
52
|
-
* Call
|
|
53
|
-
*
|
|
54
|
-
*
|
|
56
|
+
* Call for the supplied tool's input as structured JSON — forced
|
|
57
|
+
* where the model allows it, requested otherwise; throws when the
|
|
58
|
+
* turn ends without it (see the normative semantics above).
|
|
59
|
+
* Implementations that can't produce tool input at all should omit
|
|
60
|
+
* this method — consumers detect absence and fall back to regex JSON
|
|
55
61
|
* extraction on the text path.
|
|
56
62
|
*/
|
|
57
63
|
callStructured?<T>(systemPrompt: string, userMessage: string, tool: ToolSchema, maxTokens?: number): Promise<T>;
|
package/dist/llm-caller.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"llm-caller.d.ts","sourceRoot":"","sources":["../src/llm-caller.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"llm-caller.d.ts","sourceRoot":"","sources":["../src/llm-caller.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAwCG;AAEH,6DAA6D;AAC7D,MAAM,WAAW,UAAU;IACzB,IAAI,EAAE,MAAM,CAAC;IACb,WAAW,EAAE,MAAM,CAAC;IACpB,YAAY,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACvC;AAED,uFAAuF;AACvF,MAAM,WAAW,SAAS;IACxB;;;OAGG;IACH,IAAI,CACF,YAAY,EAAE,MAAM,EACpB,WAAW,EAAE,MAAM,EACnB,SAAS,CAAC,EAAE,MAAM,GACjB,OAAO,CAAC,MAAM,CAAC,CAAC;IAEnB;;;;;;;OAOG;IACH,cAAc,CAAC,CAAC,CAAC,EACf,YAAY,EAAE,MAAM,EACpB,WAAW,EAAE,MAAM,EACnB,IAAI,EAAE,UAAU,EAChB,SAAS,CAAC,EAAE,MAAM,GACjB,OAAO,CAAC,CAAC,CAAC,CAAC;CACf;AAED;;;;;GAKG;AACH,MAAM,WAAW,eAAe;IAC9B,QAAQ,EAAE,WAAW,GAAG,QAAQ,GAAG,QAAQ,GAAG,YAAY,GAAG,SAAS,CAAC;IACvE,KAAK,EAAE,MAAM,CAAC;IACd,MAAM,CAAC,EAAE,MAAM,CAAC;CACjB"}
|
package/dist/llm-caller.js
CHANGED
|
@@ -25,11 +25,15 @@
|
|
|
25
25
|
* the caller didn't request them — the text path is used as a
|
|
26
26
|
* regex-JSON fallback.
|
|
27
27
|
* - `callStructured?<T>(...)` is OPTIONAL. When present, it MUST
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
28
|
+
* return the input of the supplied `ToolSchema`'s tool, parsed as
|
|
29
|
+
* `T`, or THROW — never return anything else. It forces the tool
|
|
30
|
+
* where the model allows that; a model that refuses a forced tool
|
|
31
|
+
* (the always-thinking family) is asked for it without forcing, so
|
|
32
|
+
* the call can end with no tool input, and that ending is a throw.
|
|
33
|
+
* Consumers validate what they get back either way. Implementations
|
|
34
|
+
* that can't produce tool input at all simply omit this method;
|
|
35
|
+
* consumers fall back to `call` + regex JSON extraction. Absence is
|
|
36
|
+
* not an error.
|
|
33
37
|
* - `ToolSchema.input_schema` follows the OpenAI tool-use JSON
|
|
34
38
|
* Schema convention. Implementations that use a different
|
|
35
39
|
* tool-use protocol (e.g., Anthropic's variant) MUST translate at
|
|
@@ -14,79 +14,9 @@
|
|
|
14
14
|
*
|
|
15
15
|
* Eval-only — not exported from the package index.
|
|
16
16
|
*/
|
|
17
|
-
import { readFileSync } from 'node:fs';
|
|
18
|
-
import { homedir } from 'node:os';
|
|
19
|
-
import { resolve as pathResolve } from 'node:path';
|
|
20
17
|
import { runProbe, formatReport } from './run-probe.js';
|
|
21
|
-
|
|
18
|
+
import { buildAnthropicLlmCaller, getTokenUsage, resolveAnthropicKey, } from '../synth-bench/cli-llm.js';
|
|
22
19
|
const DEFAULT_MODEL = 'claude-haiku-4-5';
|
|
23
|
-
function resolveAnthropicKey() {
|
|
24
|
-
const envKey = process.env['ANTHROPIC_API_KEY'];
|
|
25
|
-
if (envKey && envKey.length > 0)
|
|
26
|
-
return envKey;
|
|
27
|
-
const credsPath = pathResolve(homedir(), '.ggui', 'credentials.json');
|
|
28
|
-
let parsed;
|
|
29
|
-
try {
|
|
30
|
-
parsed = JSON.parse(readFileSync(credsPath, 'utf8'));
|
|
31
|
-
}
|
|
32
|
-
catch (err) {
|
|
33
|
-
throw new Error(`probe-rerank: could not read ${credsPath} (${err instanceof Error ? err.message : String(err)}). Set ANTHROPIC_API_KEY env var or run \`ggui auth set anthropic\`.`);
|
|
34
|
-
}
|
|
35
|
-
const key = parsed.apps?.global?.anthropic;
|
|
36
|
-
if (typeof key !== 'string' || key.length === 0) {
|
|
37
|
-
throw new Error(`probe-rerank: no anthropic key found at apps.global.anthropic in ${credsPath}.`);
|
|
38
|
-
}
|
|
39
|
-
return key;
|
|
40
|
-
}
|
|
41
|
-
let totalInputTokens = 0;
|
|
42
|
-
let totalOutputTokens = 0;
|
|
43
|
-
function buildAnthropicLlmCaller(apiKey, model) {
|
|
44
|
-
return {
|
|
45
|
-
async call() {
|
|
46
|
-
throw new Error('probe-cli: text-mode not exercised — use callStructured');
|
|
47
|
-
},
|
|
48
|
-
async callStructured(systemPrompt, userMessage, tool, maxTokens) {
|
|
49
|
-
const body = {
|
|
50
|
-
model,
|
|
51
|
-
max_tokens: maxTokens ?? 1024,
|
|
52
|
-
system: systemPrompt,
|
|
53
|
-
messages: [{ role: 'user', content: userMessage }],
|
|
54
|
-
tools: [
|
|
55
|
-
{
|
|
56
|
-
name: tool.name,
|
|
57
|
-
description: tool.description,
|
|
58
|
-
input_schema: tool.input_schema,
|
|
59
|
-
},
|
|
60
|
-
],
|
|
61
|
-
tool_choice: { type: 'tool', name: tool.name },
|
|
62
|
-
};
|
|
63
|
-
const res = await fetch(ANTHROPIC_API, {
|
|
64
|
-
method: 'POST',
|
|
65
|
-
headers: {
|
|
66
|
-
'content-type': 'application/json',
|
|
67
|
-
'x-api-key': apiKey,
|
|
68
|
-
'anthropic-version': '2023-06-01',
|
|
69
|
-
},
|
|
70
|
-
body: JSON.stringify(body),
|
|
71
|
-
});
|
|
72
|
-
const json = (await res.json());
|
|
73
|
-
if (!res.ok) {
|
|
74
|
-
const errType = json.error?.type ?? 'unknown';
|
|
75
|
-
const errMsg = json.error?.message ?? `HTTP ${res.status}`;
|
|
76
|
-
throw new Error(`anthropic ${errType}: ${errMsg}`);
|
|
77
|
-
}
|
|
78
|
-
if (json.usage) {
|
|
79
|
-
totalInputTokens += json.usage.input_tokens ?? 0;
|
|
80
|
-
totalOutputTokens += json.usage.output_tokens ?? 0;
|
|
81
|
-
}
|
|
82
|
-
const toolBlock = json.content?.find((b) => b.type === 'tool_use');
|
|
83
|
-
if (!toolBlock || toolBlock.input === undefined) {
|
|
84
|
-
throw new Error(`anthropic: no tool_use block in response (stop_reason=${json.stop_reason ?? 'unknown'})`);
|
|
85
|
-
}
|
|
86
|
-
return toolBlock.input;
|
|
87
|
-
},
|
|
88
|
-
};
|
|
89
|
-
}
|
|
90
20
|
function parseArgs(argv) {
|
|
91
21
|
let limit;
|
|
92
22
|
let threshold;
|
|
@@ -116,7 +46,7 @@ const HAIKU_4_5_PRICE_INPUT_PER_TOKEN = 1.0 / 1_000_000;
|
|
|
116
46
|
const HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN = 5.0 / 1_000_000;
|
|
117
47
|
async function main() {
|
|
118
48
|
const args = parseArgs(process.argv.slice(2));
|
|
119
|
-
const apiKey = resolveAnthropicKey();
|
|
49
|
+
const apiKey = resolveAnthropicKey('probe-rerank');
|
|
120
50
|
const llm = buildAnthropicLlmCaller(apiKey, args.model);
|
|
121
51
|
process.stdout.write(`probe: model=${args.model}\n\n`);
|
|
122
52
|
const report = await runProbe({ llm }, {
|
|
@@ -132,11 +62,12 @@ async function main() {
|
|
|
132
62
|
process.stdout.write('\n');
|
|
133
63
|
process.stdout.write(formatReport(report));
|
|
134
64
|
process.stdout.write('\n\n');
|
|
135
|
-
const
|
|
136
|
-
|
|
65
|
+
const usage = getTokenUsage();
|
|
66
|
+
const totalCost = usage.input * HAIKU_4_5_PRICE_INPUT_PER_TOKEN +
|
|
67
|
+
usage.output * HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN;
|
|
137
68
|
const callsMade = report.outcomes.filter((o) => !/short-circuited/.test(o.decision.reason)).length;
|
|
138
69
|
const costPerCall = callsMade === 0 ? 0 : totalCost / callsMade;
|
|
139
|
-
process.stdout.write(`Tokens: input=${
|
|
70
|
+
process.stdout.write(`Tokens: input=${usage.input} · output=${usage.output}\n`);
|
|
140
71
|
process.stdout.write(`Cost: total=$${totalCost.toFixed(4)} · per-call=$${costPerCall.toFixed(4)}\n`);
|
|
141
72
|
process.stdout.write(` G4 cost ≤ $0.002/call → ${costPerCall <= 0.002 ? 'PASS' : 'FAIL'} ($${costPerCall.toFixed(4)})\n`);
|
|
142
73
|
}
|
|
@@ -16,5 +16,10 @@ export declare function getTokenUsage(): {
|
|
|
16
16
|
readonly input: number;
|
|
17
17
|
readonly output: number;
|
|
18
18
|
};
|
|
19
|
+
/**
|
|
20
|
+
* The dev caller the synth benches and the rerank probe CLI share
|
|
21
|
+
* (Haiku-pinned by default; every CLI takes `--model`). Text mode is not
|
|
22
|
+
* exercised — every consumer uses `callStructured`.
|
|
23
|
+
*/
|
|
19
24
|
export declare function buildAnthropicLlmCaller(apiKey: string, model: string): LLMCaller;
|
|
20
25
|
//# sourceMappingURL=cli-llm.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"cli-llm.d.ts","sourceRoot":"","sources":["../../src/synth-bench/cli-llm.ts"],"names":[],"mappings":"AAUA,OAAO,KAAK,EAAE,SAAS,EAAc,MAAM,kBAAkB,CAAC;
|
|
1
|
+
{"version":3,"file":"cli-llm.d.ts","sourceRoot":"","sources":["../../src/synth-bench/cli-llm.ts"],"names":[],"mappings":"AAUA,OAAO,KAAK,EAAE,SAAS,EAAc,MAAM,kBAAkB,CAAC;AAK9D,oFAAoF;AACpF,eAAO,MAAM,aAAa,qBAAqB,CAAC;AAEhD,mEAAmE;AACnE,eAAO,MAAM,+BAA+B,QAAkB,CAAC;AAC/D,eAAO,MAAM,gCAAgC,QAAkB,CAAC;AAMhE;;;;GAIG;AACH,wBAAgB,mBAAmB,CAAC,UAAU,EAAE,MAAM,GAAG,MAAM,CAmB9D;AAmBD;oDACoD;AACpD,wBAAgB,aAAa,IAAI;IAC/B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;CACzB,CAEA;AAUD;;;;GAIG;AACH,wBAAgB,uBAAuB,CACrC,MAAM,EAAE,MAAM,EACd,KAAK,EAAE,MAAM,GACZ,SAAS,CAyEX"}
|
|
@@ -8,6 +8,7 @@
|
|
|
8
8
|
import { readFileSync } from 'node:fs';
|
|
9
9
|
import { homedir } from 'node:os';
|
|
10
10
|
import { resolve as pathResolve } from 'node:path';
|
|
11
|
+
import { anthropicRejectsForcedToolChoice } from '@ggui-ai/protocol';
|
|
11
12
|
const ANTHROPIC_API = 'https://api.anthropic.com/v1/messages';
|
|
12
13
|
/** Default bench model — the model the `ui-gen-default` seed generator declares. */
|
|
13
14
|
export const DEFAULT_MODEL = 'claude-haiku-4-5';
|
|
@@ -44,20 +45,44 @@ let totalOutputTokens = 0;
|
|
|
44
45
|
export function getTokenUsage() {
|
|
45
46
|
return { input: totalInputTokens, output: totalOutputTokens };
|
|
46
47
|
}
|
|
48
|
+
/**
|
|
49
|
+
* Thinking budget added to the caller's answer budget for a model that
|
|
50
|
+
* refuses a forced tool (ggui#1264). Same value and reason as the
|
|
51
|
+
* negotiator's production caller in `@ggui-ai/mcp-server`; the three
|
|
52
|
+
* copies are consolidated under ggui#1271.
|
|
53
|
+
*/
|
|
54
|
+
const ALWAYS_THINKING_HEADROOM_TOKENS = 16_000;
|
|
55
|
+
/**
|
|
56
|
+
* The dev caller the synth benches and the rerank probe CLI share
|
|
57
|
+
* (Haiku-pinned by default; every CLI takes `--model`). Text mode is not
|
|
58
|
+
* exercised — every consumer uses `callStructured`.
|
|
59
|
+
*/
|
|
47
60
|
export function buildAnthropicLlmCaller(apiKey, model) {
|
|
48
61
|
return {
|
|
49
62
|
async call() {
|
|
50
|
-
throw new Error('
|
|
63
|
+
throw new Error('negotiator dev caller: text-mode not exercised — use callStructured');
|
|
51
64
|
},
|
|
52
65
|
async callStructured(systemPrompt, userMessage, tool, maxTokens) {
|
|
53
66
|
// `temperature` deprecated on Haiku 4.5+ — Anthropic rejects with
|
|
54
|
-
// HTTP 400.
|
|
55
|
-
//
|
|
56
|
-
//
|
|
67
|
+
// HTTP 400. Residual stochasticity stays bounded via canonical-key
|
|
68
|
+
// normalization downstream.
|
|
69
|
+
//
|
|
70
|
+
// ggui#1264 — a forced `tool_choice` only where the model accepts
|
|
71
|
+
// one. The always-thinking models (`@ggui-ai/protocol`
|
|
72
|
+
// `anthropicRejectsForcedToolChoice`) 400 on it, so they get
|
|
73
|
+
// `auto` + at most one call + the tool named in the system prompt,
|
|
74
|
+
// and thinking headroom on top of the answer budget (their thinking
|
|
75
|
+
// counts against `max_tokens` and comes first).
|
|
76
|
+
const refusesForcedTool = anthropicRejectsForcedToolChoice(model);
|
|
77
|
+
const answerBudget = maxTokens ?? 1024;
|
|
57
78
|
const body = {
|
|
58
79
|
model,
|
|
59
|
-
max_tokens:
|
|
60
|
-
|
|
80
|
+
max_tokens: refusesForcedTool
|
|
81
|
+
? answerBudget + ALWAYS_THINKING_HEADROOM_TOKENS
|
|
82
|
+
: answerBudget,
|
|
83
|
+
system: refusesForcedTool
|
|
84
|
+
? `${systemPrompt}\n\nAnswer by calling the \`${tool.name}\` tool exactly once. Do not answer in text.`
|
|
85
|
+
: systemPrompt,
|
|
61
86
|
messages: [{ role: 'user', content: userMessage }],
|
|
62
87
|
tools: [
|
|
63
88
|
{
|
|
@@ -66,7 +91,9 @@ export function buildAnthropicLlmCaller(apiKey, model) {
|
|
|
66
91
|
input_schema: tool.input_schema,
|
|
67
92
|
},
|
|
68
93
|
],
|
|
69
|
-
tool_choice:
|
|
94
|
+
tool_choice: refusesForcedTool
|
|
95
|
+
? { type: 'auto', disable_parallel_tool_use: true }
|
|
96
|
+
: { type: 'tool', name: tool.name },
|
|
70
97
|
};
|
|
71
98
|
const res = await fetch(ANTHROPIC_API, {
|
|
72
99
|
method: 'POST',
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ggui-ai/negotiator",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.21.0",
|
|
4
4
|
"description": "Contract-synthesis + match-judge engine for ggui's handshake. Synthesizes or repairs a conforming DataContract from an agent's draft, judges blueprint-match candidates for reuse, and validates contract structure + novelty — the primitives composed by decideHandshake in @ggui-ai/mcp-server-handlers. Deployment-agnostic: concrete embedding and vector-store bindings plug in via the storage interfaces from @ggui-ai/mcp-server-core.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"keywords": [
|
|
@@ -47,8 +47,8 @@
|
|
|
47
47
|
}
|
|
48
48
|
},
|
|
49
49
|
"dependencies": {
|
|
50
|
-
"@ggui-ai/
|
|
51
|
-
"@ggui-ai/
|
|
50
|
+
"@ggui-ai/protocol": "0.21.0",
|
|
51
|
+
"@ggui-ai/mcp-server-core": "0.21.0"
|
|
52
52
|
},
|
|
53
53
|
"devDependencies": {
|
|
54
54
|
"@types/node": "^24.0.0",
|
package/src/llm-caller.ts
CHANGED
|
@@ -25,11 +25,15 @@
|
|
|
25
25
|
* the caller didn't request them — the text path is used as a
|
|
26
26
|
* regex-JSON fallback.
|
|
27
27
|
* - `callStructured?<T>(...)` is OPTIONAL. When present, it MUST
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
28
|
+
* return the input of the supplied `ToolSchema`'s tool, parsed as
|
|
29
|
+
* `T`, or THROW — never return anything else. It forces the tool
|
|
30
|
+
* where the model allows that; a model that refuses a forced tool
|
|
31
|
+
* (the always-thinking family) is asked for it without forcing, so
|
|
32
|
+
* the call can end with no tool input, and that ending is a throw.
|
|
33
|
+
* Consumers validate what they get back either way. Implementations
|
|
34
|
+
* that can't produce tool input at all simply omit this method;
|
|
35
|
+
* consumers fall back to `call` + regex JSON extraction. Absence is
|
|
36
|
+
* not an error.
|
|
33
37
|
* - `ToolSchema.input_schema` follows the OpenAI tool-use JSON
|
|
34
38
|
* Schema convention. Implementations that use a different
|
|
35
39
|
* tool-use protocol (e.g., Anthropic's variant) MUST translate at
|
|
@@ -56,9 +60,11 @@ export interface LLMCaller {
|
|
|
56
60
|
): Promise<string>;
|
|
57
61
|
|
|
58
62
|
/**
|
|
59
|
-
* Call
|
|
60
|
-
*
|
|
61
|
-
*
|
|
63
|
+
* Call for the supplied tool's input as structured JSON — forced
|
|
64
|
+
* where the model allows it, requested otherwise; throws when the
|
|
65
|
+
* turn ends without it (see the normative semantics above).
|
|
66
|
+
* Implementations that can't produce tool input at all should omit
|
|
67
|
+
* this method — consumers detect absence and fall back to regex JSON
|
|
62
68
|
* extraction on the text path.
|
|
63
69
|
*/
|
|
64
70
|
callStructured?<T>(
|
|
@@ -14,112 +14,15 @@
|
|
|
14
14
|
*
|
|
15
15
|
* Eval-only — not exported from the package index.
|
|
16
16
|
*/
|
|
17
|
-
import { readFileSync } from 'node:fs';
|
|
18
|
-
import { homedir } from 'node:os';
|
|
19
|
-
import { resolve as pathResolve } from 'node:path';
|
|
20
17
|
import { runProbe, formatReport } from './run-probe.js';
|
|
21
|
-
import
|
|
18
|
+
import {
|
|
19
|
+
buildAnthropicLlmCaller,
|
|
20
|
+
getTokenUsage,
|
|
21
|
+
resolveAnthropicKey,
|
|
22
|
+
} from '../synth-bench/cli-llm.js';
|
|
22
23
|
|
|
23
|
-
const ANTHROPIC_API = 'https://api.anthropic.com/v1/messages';
|
|
24
24
|
const DEFAULT_MODEL = 'claude-haiku-4-5';
|
|
25
25
|
|
|
26
|
-
interface CredsFile {
|
|
27
|
-
apps?: { global?: { anthropic?: string } };
|
|
28
|
-
}
|
|
29
|
-
|
|
30
|
-
function resolveAnthropicKey(): string {
|
|
31
|
-
const envKey = process.env['ANTHROPIC_API_KEY'];
|
|
32
|
-
if (envKey && envKey.length > 0) return envKey;
|
|
33
|
-
const credsPath = pathResolve(homedir(), '.ggui', 'credentials.json');
|
|
34
|
-
let parsed: CredsFile;
|
|
35
|
-
try {
|
|
36
|
-
parsed = JSON.parse(readFileSync(credsPath, 'utf8')) as CredsFile;
|
|
37
|
-
} catch (err) {
|
|
38
|
-
throw new Error(
|
|
39
|
-
`probe-rerank: could not read ${credsPath} (${err instanceof Error ? err.message : String(err)}). Set ANTHROPIC_API_KEY env var or run \`ggui auth set anthropic\`.`,
|
|
40
|
-
);
|
|
41
|
-
}
|
|
42
|
-
const key = parsed.apps?.global?.anthropic;
|
|
43
|
-
if (typeof key !== 'string' || key.length === 0) {
|
|
44
|
-
throw new Error(
|
|
45
|
-
`probe-rerank: no anthropic key found at apps.global.anthropic in ${credsPath}.`,
|
|
46
|
-
);
|
|
47
|
-
}
|
|
48
|
-
return key;
|
|
49
|
-
}
|
|
50
|
-
|
|
51
|
-
interface AnthropicContentBlock {
|
|
52
|
-
type: string;
|
|
53
|
-
name?: string;
|
|
54
|
-
input?: unknown;
|
|
55
|
-
text?: string;
|
|
56
|
-
}
|
|
57
|
-
|
|
58
|
-
interface AnthropicResponse {
|
|
59
|
-
content?: AnthropicContentBlock[];
|
|
60
|
-
usage?: { input_tokens?: number; output_tokens?: number };
|
|
61
|
-
stop_reason?: string;
|
|
62
|
-
error?: { type?: string; message?: string };
|
|
63
|
-
}
|
|
64
|
-
|
|
65
|
-
let totalInputTokens = 0;
|
|
66
|
-
let totalOutputTokens = 0;
|
|
67
|
-
|
|
68
|
-
function buildAnthropicLlmCaller(apiKey: string, model: string): LLMCaller {
|
|
69
|
-
return {
|
|
70
|
-
async call(): Promise<string> {
|
|
71
|
-
throw new Error('probe-cli: text-mode not exercised — use callStructured');
|
|
72
|
-
},
|
|
73
|
-
async callStructured<T>(
|
|
74
|
-
systemPrompt: string,
|
|
75
|
-
userMessage: string,
|
|
76
|
-
tool: ToolSchema,
|
|
77
|
-
maxTokens?: number,
|
|
78
|
-
): Promise<T> {
|
|
79
|
-
const body = {
|
|
80
|
-
model,
|
|
81
|
-
max_tokens: maxTokens ?? 1024,
|
|
82
|
-
system: systemPrompt,
|
|
83
|
-
messages: [{ role: 'user', content: userMessage }],
|
|
84
|
-
tools: [
|
|
85
|
-
{
|
|
86
|
-
name: tool.name,
|
|
87
|
-
description: tool.description,
|
|
88
|
-
input_schema: tool.input_schema,
|
|
89
|
-
},
|
|
90
|
-
],
|
|
91
|
-
tool_choice: { type: 'tool', name: tool.name },
|
|
92
|
-
};
|
|
93
|
-
const res = await fetch(ANTHROPIC_API, {
|
|
94
|
-
method: 'POST',
|
|
95
|
-
headers: {
|
|
96
|
-
'content-type': 'application/json',
|
|
97
|
-
'x-api-key': apiKey,
|
|
98
|
-
'anthropic-version': '2023-06-01',
|
|
99
|
-
},
|
|
100
|
-
body: JSON.stringify(body),
|
|
101
|
-
});
|
|
102
|
-
const json = (await res.json()) as AnthropicResponse;
|
|
103
|
-
if (!res.ok) {
|
|
104
|
-
const errType = json.error?.type ?? 'unknown';
|
|
105
|
-
const errMsg = json.error?.message ?? `HTTP ${res.status}`;
|
|
106
|
-
throw new Error(`anthropic ${errType}: ${errMsg}`);
|
|
107
|
-
}
|
|
108
|
-
if (json.usage) {
|
|
109
|
-
totalInputTokens += json.usage.input_tokens ?? 0;
|
|
110
|
-
totalOutputTokens += json.usage.output_tokens ?? 0;
|
|
111
|
-
}
|
|
112
|
-
const toolBlock = json.content?.find((b) => b.type === 'tool_use');
|
|
113
|
-
if (!toolBlock || toolBlock.input === undefined) {
|
|
114
|
-
throw new Error(
|
|
115
|
-
`anthropic: no tool_use block in response (stop_reason=${json.stop_reason ?? 'unknown'})`,
|
|
116
|
-
);
|
|
117
|
-
}
|
|
118
|
-
return toolBlock.input as T;
|
|
119
|
-
},
|
|
120
|
-
};
|
|
121
|
-
}
|
|
122
|
-
|
|
123
26
|
function parseArgs(argv: readonly string[]): { limit?: number; threshold?: number; model: string } {
|
|
124
27
|
let limit: number | undefined;
|
|
125
28
|
let threshold: number | undefined;
|
|
@@ -147,7 +50,7 @@ const HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN = 5.0 / 1_000_000;
|
|
|
147
50
|
|
|
148
51
|
async function main(): Promise<void> {
|
|
149
52
|
const args = parseArgs(process.argv.slice(2));
|
|
150
|
-
const apiKey = resolveAnthropicKey();
|
|
53
|
+
const apiKey = resolveAnthropicKey('probe-rerank');
|
|
151
54
|
const llm = buildAnthropicLlmCaller(apiKey, args.model);
|
|
152
55
|
|
|
153
56
|
process.stdout.write(`probe: model=${args.model}\n\n`);
|
|
@@ -171,15 +74,16 @@ async function main(): Promise<void> {
|
|
|
171
74
|
process.stdout.write(formatReport(report));
|
|
172
75
|
process.stdout.write('\n\n');
|
|
173
76
|
|
|
77
|
+
const usage = getTokenUsage();
|
|
174
78
|
const totalCost =
|
|
175
|
-
|
|
176
|
-
|
|
79
|
+
usage.input * HAIKU_4_5_PRICE_INPUT_PER_TOKEN +
|
|
80
|
+
usage.output * HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN;
|
|
177
81
|
const callsMade = report.outcomes.filter(
|
|
178
82
|
(o) => !/short-circuited/.test(o.decision.reason),
|
|
179
83
|
).length;
|
|
180
84
|
const costPerCall = callsMade === 0 ? 0 : totalCost / callsMade;
|
|
181
85
|
process.stdout.write(
|
|
182
|
-
`Tokens: input=${
|
|
86
|
+
`Tokens: input=${usage.input} · output=${usage.output}\n`,
|
|
183
87
|
);
|
|
184
88
|
process.stdout.write(
|
|
185
89
|
`Cost: total=$${totalCost.toFixed(4)} · per-call=$${costPerCall.toFixed(4)}\n`,
|
|
@@ -9,6 +9,7 @@ import { readFileSync } from 'node:fs';
|
|
|
9
9
|
import { homedir } from 'node:os';
|
|
10
10
|
import { resolve as pathResolve } from 'node:path';
|
|
11
11
|
import type { LLMCaller, ToolSchema } from '../llm-caller.js';
|
|
12
|
+
import { anthropicRejectsForcedToolChoice } from '@ggui-ai/protocol';
|
|
12
13
|
|
|
13
14
|
const ANTHROPIC_API = 'https://api.anthropic.com/v1/messages';
|
|
14
15
|
|
|
@@ -75,6 +76,19 @@ export function getTokenUsage(): {
|
|
|
75
76
|
return { input: totalInputTokens, output: totalOutputTokens };
|
|
76
77
|
}
|
|
77
78
|
|
|
79
|
+
/**
|
|
80
|
+
* Thinking budget added to the caller's answer budget for a model that
|
|
81
|
+
* refuses a forced tool (ggui#1264). Same value and reason as the
|
|
82
|
+
* negotiator's production caller in `@ggui-ai/mcp-server`; the three
|
|
83
|
+
* copies are consolidated under ggui#1271.
|
|
84
|
+
*/
|
|
85
|
+
const ALWAYS_THINKING_HEADROOM_TOKENS = 16_000;
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* The dev caller the synth benches and the rerank probe CLI share
|
|
89
|
+
* (Haiku-pinned by default; every CLI takes `--model`). Text mode is not
|
|
90
|
+
* exercised — every consumer uses `callStructured`.
|
|
91
|
+
*/
|
|
78
92
|
export function buildAnthropicLlmCaller(
|
|
79
93
|
apiKey: string,
|
|
80
94
|
model: string,
|
|
@@ -82,7 +96,7 @@ export function buildAnthropicLlmCaller(
|
|
|
82
96
|
return {
|
|
83
97
|
async call(): Promise<string> {
|
|
84
98
|
throw new Error(
|
|
85
|
-
'
|
|
99
|
+
'negotiator dev caller: text-mode not exercised — use callStructured',
|
|
86
100
|
);
|
|
87
101
|
},
|
|
88
102
|
async callStructured<T>(
|
|
@@ -92,13 +106,25 @@ export function buildAnthropicLlmCaller(
|
|
|
92
106
|
maxTokens?: number,
|
|
93
107
|
): Promise<T> {
|
|
94
108
|
// `temperature` deprecated on Haiku 4.5+ — Anthropic rejects with
|
|
95
|
-
// HTTP 400.
|
|
96
|
-
//
|
|
97
|
-
//
|
|
109
|
+
// HTTP 400. Residual stochasticity stays bounded via canonical-key
|
|
110
|
+
// normalization downstream.
|
|
111
|
+
//
|
|
112
|
+
// ggui#1264 — a forced `tool_choice` only where the model accepts
|
|
113
|
+
// one. The always-thinking models (`@ggui-ai/protocol`
|
|
114
|
+
// `anthropicRejectsForcedToolChoice`) 400 on it, so they get
|
|
115
|
+
// `auto` + at most one call + the tool named in the system prompt,
|
|
116
|
+
// and thinking headroom on top of the answer budget (their thinking
|
|
117
|
+
// counts against `max_tokens` and comes first).
|
|
118
|
+
const refusesForcedTool = anthropicRejectsForcedToolChoice(model);
|
|
119
|
+
const answerBudget = maxTokens ?? 1024;
|
|
98
120
|
const body = {
|
|
99
121
|
model,
|
|
100
|
-
max_tokens:
|
|
101
|
-
|
|
122
|
+
max_tokens: refusesForcedTool
|
|
123
|
+
? answerBudget + ALWAYS_THINKING_HEADROOM_TOKENS
|
|
124
|
+
: answerBudget,
|
|
125
|
+
system: refusesForcedTool
|
|
126
|
+
? `${systemPrompt}\n\nAnswer by calling the \`${tool.name}\` tool exactly once. Do not answer in text.`
|
|
127
|
+
: systemPrompt,
|
|
102
128
|
messages: [{ role: 'user', content: userMessage }],
|
|
103
129
|
tools: [
|
|
104
130
|
{
|
|
@@ -107,7 +133,9 @@ export function buildAnthropicLlmCaller(
|
|
|
107
133
|
input_schema: tool.input_schema,
|
|
108
134
|
},
|
|
109
135
|
],
|
|
110
|
-
tool_choice:
|
|
136
|
+
tool_choice: refusesForcedTool
|
|
137
|
+
? { type: 'auto', disable_parallel_tool_use: true }
|
|
138
|
+
: { type: 'tool', name: tool.name },
|
|
111
139
|
};
|
|
112
140
|
const res = await fetch(ANTHROPIC_API, {
|
|
113
141
|
method: 'POST',
|