@ggui-ai/negotiator 0.1.0-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +49 -0
- package/dist/contract-hash.d.ts +54 -0
- package/dist/contract-hash.d.ts.map +1 -0
- package/dist/contract-hash.js +96 -0
- package/dist/contract-validators.d.ts +171 -0
- package/dist/contract-validators.d.ts.map +1 -0
- package/dist/contract-validators.js +478 -0
- package/dist/decision-input.d.ts +48 -0
- package/dist/decision-input.d.ts.map +1 -0
- package/dist/decision-input.js +14 -0
- package/dist/decision.d.ts +54 -0
- package/dist/decision.d.ts.map +1 -0
- package/dist/decision.js +500 -0
- package/dist/index.d.ts +36 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +25 -0
- package/dist/intent.d.ts +22 -0
- package/dist/intent.d.ts.map +1 -0
- package/dist/intent.js +28 -0
- package/dist/llm-caller.d.ts +70 -0
- package/dist/llm-caller.d.ts.map +1 -0
- package/dist/llm-caller.js +38 -0
- package/dist/llm-rerank.d.ts +101 -0
- package/dist/llm-rerank.d.ts.map +1 -0
- package/dist/llm-rerank.js +178 -0
- package/dist/negotiate.d.ts +141 -0
- package/dist/negotiate.d.ts.map +1 -0
- package/dist/negotiate.js +161 -0
- package/dist/normalize-schema.d.ts +22 -0
- package/dist/normalize-schema.d.ts.map +1 -0
- package/dist/normalize-schema.js +191 -0
- package/dist/pure.d.ts +30 -0
- package/dist/pure.d.ts.map +1 -0
- package/dist/pure.js +43 -0
- package/dist/rag-search.d.ts +73 -0
- package/dist/rag-search.d.ts.map +1 -0
- package/dist/rag-search.js +192 -0
- package/dist/rerank-eval/pairs.d.ts +28 -0
- package/dist/rerank-eval/pairs.d.ts.map +1 -0
- package/dist/rerank-eval/pairs.js +531 -0
- package/dist/rerank-eval/run-probe-cli.d.ts +3 -0
- package/dist/rerank-eval/run-probe-cli.d.ts.map +1 -0
- package/dist/rerank-eval/run-probe-cli.js +146 -0
- package/dist/rerank-eval/run-probe.d.ts +68 -0
- package/dist/rerank-eval/run-probe.d.ts.map +1 -0
- package/dist/rerank-eval/run-probe.js +113 -0
- package/dist/session.d.ts +42 -0
- package/dist/session.d.ts.map +1 -0
- package/dist/session.js +21 -0
- package/dist/suggestion.d.ts +38 -0
- package/dist/suggestion.d.ts.map +1 -0
- package/dist/suggestion.js +47 -0
- package/dist/synth-bench/corpus.d.ts +106 -0
- package/dist/synth-bench/corpus.d.ts.map +1 -0
- package/dist/synth-bench/corpus.js +994 -0
- package/dist/synth-bench/run-bench-cli.d.ts +3 -0
- package/dist/synth-bench/run-bench-cli.d.ts.map +1 -0
- package/dist/synth-bench/run-bench-cli.js +181 -0
- package/dist/synth-bench/run-bench.d.ts +101 -0
- package/dist/synth-bench/run-bench.d.ts.map +1 -0
- package/dist/synth-bench/run-bench.js +374 -0
- package/dist/synthesize-contract.d.ts +131 -0
- package/dist/synthesize-contract.d.ts.map +1 -0
- package/dist/synthesize-contract.js +948 -0
- package/dist/types.d.ts +30 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +13 -0
- package/package.json +74 -0
- package/src/contract-hash.ts +102 -0
- package/src/contract-validators.ts +604 -0
- package/src/decision-input.ts +49 -0
- package/src/decision.ts +581 -0
- package/src/index.ts +63 -0
- package/src/intent.ts +37 -0
- package/src/llm-caller.ts +82 -0
- package/src/llm-rerank.ts +280 -0
- package/src/negotiate.ts +312 -0
- package/src/normalize-schema.ts +193 -0
- package/src/pure.ts +46 -0
- package/src/rag-search.ts +274 -0
- package/src/rerank-eval/pairs.ts +624 -0
- package/src/rerank-eval/run-probe-cli.ts +197 -0
- package/src/rerank-eval/run-probe.ts +198 -0
- package/src/session.ts +41 -0
- package/src/suggestion.ts +73 -0
- package/src/synth-bench/corpus.ts +1126 -0
- package/src/synth-bench/run-bench-cli.ts +237 -0
- package/src/synth-bench/run-bench.ts +525 -0
- package/src/synthesize-contract.ts +1161 -0
- package/src/types.ts +31 -0
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* Rerank quality probe CLI.
|
|
4
|
+
*
|
|
5
|
+
* Reads `~/.ggui/credentials.json` for the Anthropic API key, runs
|
|
6
|
+
* the probe against `claude-haiku-4-5`, prints the report.
|
|
7
|
+
*
|
|
8
|
+
* Usage:
|
|
9
|
+
* pnpm -F @ggui-ai/negotiator probe-rerank
|
|
10
|
+
* ANTHROPIC_API_KEY=sk-... pnpm -F @ggui-ai/negotiator probe-rerank
|
|
11
|
+
* pnpm -F @ggui-ai/negotiator probe-rerank -- --limit 5
|
|
12
|
+
*
|
|
13
|
+
* Cost: ~$0.025 for the full 25-pair run with Haiku 4.5.
|
|
14
|
+
*
|
|
15
|
+
* Eval-only — not exported from the package index.
|
|
16
|
+
*/
|
|
17
|
+
import { readFileSync } from 'node:fs';
|
|
18
|
+
import { homedir } from 'node:os';
|
|
19
|
+
import { resolve as pathResolve } from 'node:path';
|
|
20
|
+
import { runProbe, formatReport } from './run-probe.js';
|
|
21
|
+
import type { LLMCaller, ToolSchema } from '../llm-caller.js';
|
|
22
|
+
|
|
23
|
+
const ANTHROPIC_API = 'https://api.anthropic.com/v1/messages';
|
|
24
|
+
const DEFAULT_MODEL = 'claude-haiku-4-5';
|
|
25
|
+
|
|
26
|
+
interface CredsFile {
|
|
27
|
+
apps?: { global?: { anthropic?: string } };
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
function resolveAnthropicKey(): string {
|
|
31
|
+
const envKey = process.env['ANTHROPIC_API_KEY'];
|
|
32
|
+
if (envKey && envKey.length > 0) return envKey;
|
|
33
|
+
const credsPath = pathResolve(homedir(), '.ggui', 'credentials.json');
|
|
34
|
+
let parsed: CredsFile;
|
|
35
|
+
try {
|
|
36
|
+
parsed = JSON.parse(readFileSync(credsPath, 'utf8')) as CredsFile;
|
|
37
|
+
} catch (err) {
|
|
38
|
+
throw new Error(
|
|
39
|
+
`probe-rerank: could not read ${credsPath} (${err instanceof Error ? err.message : String(err)}). Set ANTHROPIC_API_KEY env var or run \`ggui auth set anthropic\`.`,
|
|
40
|
+
);
|
|
41
|
+
}
|
|
42
|
+
const key = parsed.apps?.global?.anthropic;
|
|
43
|
+
if (typeof key !== 'string' || key.length === 0) {
|
|
44
|
+
throw new Error(
|
|
45
|
+
`probe-rerank: no anthropic key found at apps.global.anthropic in ${credsPath}.`,
|
|
46
|
+
);
|
|
47
|
+
}
|
|
48
|
+
return key;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
interface AnthropicContentBlock {
|
|
52
|
+
type: string;
|
|
53
|
+
name?: string;
|
|
54
|
+
input?: unknown;
|
|
55
|
+
text?: string;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
interface AnthropicResponse {
|
|
59
|
+
content?: AnthropicContentBlock[];
|
|
60
|
+
usage?: { input_tokens?: number; output_tokens?: number };
|
|
61
|
+
stop_reason?: string;
|
|
62
|
+
error?: { type?: string; message?: string };
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
let totalInputTokens = 0;
|
|
66
|
+
let totalOutputTokens = 0;
|
|
67
|
+
|
|
68
|
+
function buildAnthropicLlmCaller(apiKey: string, model: string): LLMCaller {
|
|
69
|
+
return {
|
|
70
|
+
async call(): Promise<string> {
|
|
71
|
+
throw new Error('probe-cli: text-mode not exercised — use callStructured');
|
|
72
|
+
},
|
|
73
|
+
async callStructured<T>(
|
|
74
|
+
systemPrompt: string,
|
|
75
|
+
userMessage: string,
|
|
76
|
+
tool: ToolSchema,
|
|
77
|
+
maxTokens?: number,
|
|
78
|
+
): Promise<T> {
|
|
79
|
+
const body = {
|
|
80
|
+
model,
|
|
81
|
+
max_tokens: maxTokens ?? 1024,
|
|
82
|
+
system: systemPrompt,
|
|
83
|
+
messages: [{ role: 'user', content: userMessage }],
|
|
84
|
+
tools: [
|
|
85
|
+
{
|
|
86
|
+
name: tool.name,
|
|
87
|
+
description: tool.description,
|
|
88
|
+
input_schema: tool.input_schema,
|
|
89
|
+
},
|
|
90
|
+
],
|
|
91
|
+
tool_choice: { type: 'tool', name: tool.name },
|
|
92
|
+
};
|
|
93
|
+
const res = await fetch(ANTHROPIC_API, {
|
|
94
|
+
method: 'POST',
|
|
95
|
+
headers: {
|
|
96
|
+
'content-type': 'application/json',
|
|
97
|
+
'x-api-key': apiKey,
|
|
98
|
+
'anthropic-version': '2023-06-01',
|
|
99
|
+
},
|
|
100
|
+
body: JSON.stringify(body),
|
|
101
|
+
});
|
|
102
|
+
const json = (await res.json()) as AnthropicResponse;
|
|
103
|
+
if (!res.ok) {
|
|
104
|
+
const errType = json.error?.type ?? 'unknown';
|
|
105
|
+
const errMsg = json.error?.message ?? `HTTP ${res.status}`;
|
|
106
|
+
throw new Error(`anthropic ${errType}: ${errMsg}`);
|
|
107
|
+
}
|
|
108
|
+
if (json.usage) {
|
|
109
|
+
totalInputTokens += json.usage.input_tokens ?? 0;
|
|
110
|
+
totalOutputTokens += json.usage.output_tokens ?? 0;
|
|
111
|
+
}
|
|
112
|
+
const toolBlock = json.content?.find((b) => b.type === 'tool_use');
|
|
113
|
+
if (!toolBlock || toolBlock.input === undefined) {
|
|
114
|
+
throw new Error(
|
|
115
|
+
`anthropic: no tool_use block in response (stop_reason=${json.stop_reason ?? 'unknown'})`,
|
|
116
|
+
);
|
|
117
|
+
}
|
|
118
|
+
return toolBlock.input as T;
|
|
119
|
+
},
|
|
120
|
+
};
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function parseArgs(argv: readonly string[]): { limit?: number; threshold?: number; model: string } {
|
|
124
|
+
let limit: number | undefined;
|
|
125
|
+
let threshold: number | undefined;
|
|
126
|
+
let model = DEFAULT_MODEL;
|
|
127
|
+
for (let i = 0; i < argv.length; i++) {
|
|
128
|
+
const a = argv[i];
|
|
129
|
+
if (a === '--limit' && argv[i + 1]) {
|
|
130
|
+
limit = Number(argv[++i]);
|
|
131
|
+
} else if (a === '--threshold' && argv[i + 1]) {
|
|
132
|
+
threshold = Number(argv[++i]);
|
|
133
|
+
} else if (a === '--model' && argv[i + 1]) {
|
|
134
|
+
model = argv[++i]!;
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
const result: { limit?: number; threshold?: number; model: string } = { model };
|
|
138
|
+
if (limit !== undefined) result.limit = limit;
|
|
139
|
+
if (threshold !== undefined) result.threshold = threshold;
|
|
140
|
+
return result;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
// Approximate Haiku 4.5 pricing as of 2026-05.
|
|
144
|
+
// Input: $1.00 / Mtok, output: $5.00 / Mtok.
|
|
145
|
+
const HAIKU_4_5_PRICE_INPUT_PER_TOKEN = 1.0 / 1_000_000;
|
|
146
|
+
const HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN = 5.0 / 1_000_000;
|
|
147
|
+
|
|
148
|
+
async function main(): Promise<void> {
|
|
149
|
+
const args = parseArgs(process.argv.slice(2));
|
|
150
|
+
const apiKey = resolveAnthropicKey();
|
|
151
|
+
const llm = buildAnthropicLlmCaller(apiKey, args.model);
|
|
152
|
+
|
|
153
|
+
process.stdout.write(`probe: model=${args.model}\n\n`);
|
|
154
|
+
const report = await runProbe(
|
|
155
|
+
{ llm },
|
|
156
|
+
{
|
|
157
|
+
...(args.limit !== undefined ? { limit: args.limit } : {}),
|
|
158
|
+
...(args.threshold !== undefined ? { threshold: args.threshold } : {}),
|
|
159
|
+
onProgress: (outcome, idx, total) => {
|
|
160
|
+
const status = outcome.correct ? 'OK ' : 'NO ';
|
|
161
|
+
const conf = outcome.decision.confidence.toFixed(2);
|
|
162
|
+
const id = outcome.pair.id.padEnd(24);
|
|
163
|
+
process.stdout.write(
|
|
164
|
+
`[${status}] ${(idx + 1).toString().padStart(2)}/${total} ${id} conf=${conf} pred=${outcome.predictedMatchId ?? 'null'} gold=${outcome.pair.goldMatchId ?? 'null'}\n`,
|
|
165
|
+
);
|
|
166
|
+
},
|
|
167
|
+
},
|
|
168
|
+
);
|
|
169
|
+
|
|
170
|
+
process.stdout.write('\n');
|
|
171
|
+
process.stdout.write(formatReport(report));
|
|
172
|
+
process.stdout.write('\n\n');
|
|
173
|
+
|
|
174
|
+
const totalCost =
|
|
175
|
+
totalInputTokens * HAIKU_4_5_PRICE_INPUT_PER_TOKEN +
|
|
176
|
+
totalOutputTokens * HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN;
|
|
177
|
+
const callsMade = report.outcomes.filter(
|
|
178
|
+
(o) => !/short-circuited/.test(o.decision.reason),
|
|
179
|
+
).length;
|
|
180
|
+
const costPerCall = callsMade === 0 ? 0 : totalCost / callsMade;
|
|
181
|
+
process.stdout.write(
|
|
182
|
+
`Tokens: input=${totalInputTokens} · output=${totalOutputTokens}\n`,
|
|
183
|
+
);
|
|
184
|
+
process.stdout.write(
|
|
185
|
+
`Cost: total=$${totalCost.toFixed(4)} · per-call=$${costPerCall.toFixed(4)}\n`,
|
|
186
|
+
);
|
|
187
|
+
process.stdout.write(
|
|
188
|
+
` G4 cost ≤ $0.002/call → ${costPerCall <= 0.002 ? 'PASS' : 'FAIL'} ($${costPerCall.toFixed(4)})\n`,
|
|
189
|
+
);
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
main().catch((err) => {
|
|
193
|
+
process.stderr.write(
|
|
194
|
+
`probe-rerank failed: ${err instanceof Error ? err.message : String(err)}\n`,
|
|
195
|
+
);
|
|
196
|
+
process.exit(1);
|
|
197
|
+
});
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Rerank quality probe.
|
|
3
|
+
*
|
|
4
|
+
* Runs the configured `LLMCaller` over the {@link EVAL_PAIRS} eval
|
|
5
|
+
* set and returns precision@1 + adversarial false-positive +
|
|
6
|
+
* latency p95. The caller compares these against the quality gates
|
|
7
|
+
* (precision, adversarial false-positive rate, latency, cost)
|
|
8
|
+
* reported by the probe CLI.
|
|
9
|
+
*
|
|
10
|
+
* Eval-only — not exported from the package index.
|
|
11
|
+
*/
|
|
12
|
+
import { rerankCandidates, type RerankDecision } from '../llm-rerank.js';
|
|
13
|
+
import type { LLMCaller } from '../llm-caller.js';
|
|
14
|
+
import { EVAL_PAIRS, type EvalPair } from './pairs.js';
|
|
15
|
+
|
|
16
|
+
export interface ProbeOutcome {
|
|
17
|
+
readonly pair: EvalPair;
|
|
18
|
+
readonly decision: RerankDecision;
|
|
19
|
+
/** Confidence threshold applied to compute correctness. */
|
|
20
|
+
readonly threshold: number;
|
|
21
|
+
/** Predicted matchId AFTER threshold gate (null if confidence below). */
|
|
22
|
+
readonly predictedMatchId: string | null;
|
|
23
|
+
/** Was the prediction correct vs gold? */
|
|
24
|
+
readonly correct: boolean;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
export interface ProbeReport {
|
|
28
|
+
readonly outcomes: readonly ProbeOutcome[];
|
|
29
|
+
readonly totals: {
|
|
30
|
+
readonly all: number;
|
|
31
|
+
readonly correct: number;
|
|
32
|
+
readonly precision: number;
|
|
33
|
+
};
|
|
34
|
+
readonly byKind: Readonly<{
|
|
35
|
+
[K in EvalPair['kind']]: {
|
|
36
|
+
readonly all: number;
|
|
37
|
+
readonly correct: number;
|
|
38
|
+
readonly precision: number;
|
|
39
|
+
};
|
|
40
|
+
}>;
|
|
41
|
+
readonly adversarialFalsePositiveRate: number;
|
|
42
|
+
readonly latency: { readonly p50Ms: number; readonly p95Ms: number };
|
|
43
|
+
readonly threshold: number;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
export interface RunProbeOptions {
|
|
47
|
+
/** Use a subset for smoke; default = all 25 pairs. */
|
|
48
|
+
readonly limit?: number;
|
|
49
|
+
/** Confidence threshold for considering a non-null match a hit. Default 0.6. */
|
|
50
|
+
readonly threshold?: number;
|
|
51
|
+
/**
|
|
52
|
+
* Optional progress callback. Fired AFTER each pair completes — the
|
|
53
|
+
* CLI uses this to print a per-pair line so the operator sees live
|
|
54
|
+
* progress instead of waiting for the full sweep to finish.
|
|
55
|
+
*/
|
|
56
|
+
readonly onProgress?: (outcome: ProbeOutcome, index: number, total: number) => void;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Run the probe against `pairs` (defaults to {@link EVAL_PAIRS}).
|
|
61
|
+
* Sequential by design — small N (~25), and concurrent calls would
|
|
62
|
+
* complicate latency measurement without changing the gate decision.
|
|
63
|
+
*/
|
|
64
|
+
export async function runProbe(
|
|
65
|
+
deps: { readonly llm: LLMCaller },
|
|
66
|
+
options: RunProbeOptions = {},
|
|
67
|
+
pairs: readonly EvalPair[] = EVAL_PAIRS,
|
|
68
|
+
): Promise<ProbeReport> {
|
|
69
|
+
const threshold = options.threshold ?? 0.6;
|
|
70
|
+
const subset = options.limit ? pairs.slice(0, options.limit) : pairs;
|
|
71
|
+
const outcomes: ProbeOutcome[] = [];
|
|
72
|
+
|
|
73
|
+
for (let i = 0; i < subset.length; i++) {
|
|
74
|
+
const pair = subset[i]!;
|
|
75
|
+
const decision = await rerankCandidates(deps, pair.query, pair.candidates);
|
|
76
|
+
const predictedMatchId =
|
|
77
|
+
decision.confidence >= threshold ? decision.matchId : null;
|
|
78
|
+
const correct = predictedMatchId === pair.goldMatchId;
|
|
79
|
+
const outcome: ProbeOutcome = {
|
|
80
|
+
pair,
|
|
81
|
+
decision,
|
|
82
|
+
threshold,
|
|
83
|
+
predictedMatchId,
|
|
84
|
+
correct,
|
|
85
|
+
};
|
|
86
|
+
outcomes.push(outcome);
|
|
87
|
+
options.onProgress?.(outcome, i, subset.length);
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
const totals = aggregate(outcomes);
|
|
91
|
+
const byKind = {
|
|
92
|
+
'should-match': aggregate(outcomes.filter((o) => o.pair.kind === 'should-match')),
|
|
93
|
+
'no-match': aggregate(outcomes.filter((o) => o.pair.kind === 'no-match')),
|
|
94
|
+
adversarial: aggregate(outcomes.filter((o) => o.pair.kind === 'adversarial')),
|
|
95
|
+
};
|
|
96
|
+
// Adversarial false-positive: judge accepted a candidate when gold
|
|
97
|
+
// says null. Compute as `(adversarial-pairs-with-prediction-when-
|
|
98
|
+
// gold-was-null) / (adversarial-pairs-with-gold-null)`.
|
|
99
|
+
const adversarial = outcomes.filter((o) => o.pair.kind === 'adversarial');
|
|
100
|
+
const adversarialNullGold = adversarial.filter((o) => o.pair.goldMatchId === null);
|
|
101
|
+
const adversarialFalsePositives = adversarialNullGold.filter(
|
|
102
|
+
(o) => o.predictedMatchId !== null,
|
|
103
|
+
);
|
|
104
|
+
const adversarialFalsePositiveRate =
|
|
105
|
+
adversarialNullGold.length === 0
|
|
106
|
+
? 0
|
|
107
|
+
: adversarialFalsePositives.length / adversarialNullGold.length;
|
|
108
|
+
|
|
109
|
+
const latencies = outcomes
|
|
110
|
+
.map((o) => o.decision.latencyMs)
|
|
111
|
+
.sort((a, b) => a - b);
|
|
112
|
+
const latency = {
|
|
113
|
+
p50Ms: percentile(latencies, 0.5),
|
|
114
|
+
p95Ms: percentile(latencies, 0.95),
|
|
115
|
+
};
|
|
116
|
+
|
|
117
|
+
return {
|
|
118
|
+
outcomes,
|
|
119
|
+
totals,
|
|
120
|
+
byKind,
|
|
121
|
+
adversarialFalsePositiveRate,
|
|
122
|
+
latency,
|
|
123
|
+
threshold,
|
|
124
|
+
};
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function aggregate(outcomes: readonly ProbeOutcome[]): {
|
|
128
|
+
readonly all: number;
|
|
129
|
+
readonly correct: number;
|
|
130
|
+
readonly precision: number;
|
|
131
|
+
} {
|
|
132
|
+
const all = outcomes.length;
|
|
133
|
+
const correct = outcomes.filter((o) => o.correct).length;
|
|
134
|
+
const precision = all === 0 ? 0 : correct / all;
|
|
135
|
+
return { all, correct, precision };
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
function percentile(sorted: readonly number[], p: number): number {
|
|
139
|
+
if (sorted.length === 0) return 0;
|
|
140
|
+
const idx = Math.min(
|
|
141
|
+
sorted.length - 1,
|
|
142
|
+
Math.floor(p * sorted.length),
|
|
143
|
+
);
|
|
144
|
+
return sorted[idx] ?? 0;
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/** Pretty-print a {@link ProbeReport} for terminal output. */
|
|
148
|
+
export function formatReport(report: ProbeReport): string {
|
|
149
|
+
const lines: string[] = [];
|
|
150
|
+
lines.push('━━━ rerank quality probe ━━━');
|
|
151
|
+
lines.push('');
|
|
152
|
+
lines.push(`Threshold: confidence ≥ ${report.threshold}`);
|
|
153
|
+
lines.push('');
|
|
154
|
+
lines.push('Precision by kind:');
|
|
155
|
+
for (const [kind, stats] of Object.entries(report.byKind) as Array<[
|
|
156
|
+
EvalPair['kind'],
|
|
157
|
+
{ all: number; correct: number; precision: number },
|
|
158
|
+
]>) {
|
|
159
|
+
const pct = (stats.precision * 100).toFixed(1);
|
|
160
|
+
lines.push(` ${kind.padEnd(14)} ${stats.correct}/${stats.all} (${pct}%)`);
|
|
161
|
+
}
|
|
162
|
+
lines.push('');
|
|
163
|
+
lines.push(
|
|
164
|
+
`Overall: ${report.totals.correct}/${report.totals.all} (${(report.totals.precision * 100).toFixed(1)}%)`,
|
|
165
|
+
);
|
|
166
|
+
lines.push(
|
|
167
|
+
`Adversarial FP: ${(report.adversarialFalsePositiveRate * 100).toFixed(1)}%`,
|
|
168
|
+
);
|
|
169
|
+
lines.push(
|
|
170
|
+
`Latency: p50=${report.latency.p50Ms}ms · p95=${report.latency.p95Ms}ms`,
|
|
171
|
+
);
|
|
172
|
+
lines.push('');
|
|
173
|
+
lines.push('Gates:');
|
|
174
|
+
lines.push(
|
|
175
|
+
` G1 precision@1 ≥ 85% → ${report.totals.precision >= 0.85 ? 'PASS' : 'FAIL'} (${(report.totals.precision * 100).toFixed(1)}%)`,
|
|
176
|
+
);
|
|
177
|
+
lines.push(
|
|
178
|
+
` G2 adversarial FP ≤ 5% → ${report.adversarialFalsePositiveRate <= 0.05 ? 'PASS' : 'FAIL'} (${(report.adversarialFalsePositiveRate * 100).toFixed(1)}%)`,
|
|
179
|
+
);
|
|
180
|
+
lines.push(
|
|
181
|
+
` G3 latency p95 ≤ 600ms → ${report.latency.p95Ms <= 600 ? 'PASS' : 'FAIL'} (p95=${report.latency.p95Ms}ms)`,
|
|
182
|
+
);
|
|
183
|
+
|
|
184
|
+
// Per-pair detail for failed predictions
|
|
185
|
+
const failed = report.outcomes.filter((o) => !o.correct);
|
|
186
|
+
if (failed.length > 0) {
|
|
187
|
+
lines.push('');
|
|
188
|
+
lines.push('Failed predictions:');
|
|
189
|
+
for (const o of failed) {
|
|
190
|
+
lines.push(
|
|
191
|
+
` [${o.pair.kind}] ${o.pair.id}: predicted=${o.predictedMatchId ?? 'null'}, gold=${o.pair.goldMatchId ?? 'null'}, conf=${o.decision.confidence.toFixed(2)}`,
|
|
192
|
+
);
|
|
193
|
+
lines.push(` reason: ${o.decision.reason}`);
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
return lines.join('\n');
|
|
198
|
+
}
|
package/src/session.ts
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `SessionState` — what the decision engine sees about the live
|
|
3
|
+
* session at the moment of a negotiation call.
|
|
4
|
+
*
|
|
5
|
+
* Captures the UI stack (pages previously pushed into this
|
|
6
|
+
* session), recent conversation, and optional interface context
|
|
7
|
+
* (viewport, device class). Consumed by `NegotiatorDecisionInput`
|
|
8
|
+
* to drive `create / update / compose / replace` decisions — the
|
|
9
|
+
* stack tells the engine whether something already on screen can
|
|
10
|
+
* absorb this push as an `update`, and the conversation history
|
|
11
|
+
* feeds the decision LLM.
|
|
12
|
+
*
|
|
13
|
+
* Kept in `@ggui-ai/negotiator` (not `mcp-server-core`): this is
|
|
14
|
+
* the decision engine's input shape, not a storage seam. An MCP
|
|
15
|
+
* server implementer binding against the public `Negotiator`
|
|
16
|
+
* interface sees `NegotiatorInput` / `NegotiatorResult` — never
|
|
17
|
+
* this internal input shape. Community adapters that want to
|
|
18
|
+
* call `makeDecision` directly (bypassing the `Negotiator` wrapper)
|
|
19
|
+
* import this type.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import type { DataContract, InterfaceContext } from '@ggui-ai/protocol';
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* One entry on the session's UI stack — a page the agent previously
|
|
26
|
+
* pushed. Minimal shape: every field the decision engine actually
|
|
27
|
+
* reads to decide reuse vs. create.
|
|
28
|
+
*/
|
|
29
|
+
export interface SessionStackEntry {
|
|
30
|
+
id: string;
|
|
31
|
+
prompt?: string;
|
|
32
|
+
contract?: DataContract;
|
|
33
|
+
componentCode: string;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/** Current state of a session — stack + conversation history. */
|
|
37
|
+
export interface SessionState {
|
|
38
|
+
stack: SessionStackEntry[];
|
|
39
|
+
conversationHistory: Array<{ role: 'user' | 'agent'; content: string }>;
|
|
40
|
+
interfaceContext?: InterfaceContext;
|
|
41
|
+
}
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Suggestion engine — detects structured data patterns in free-form
|
|
3
|
+
* agent text and proposes a ggui_push call that would show them
|
|
4
|
+
* interactively.
|
|
5
|
+
*
|
|
6
|
+
* Pure regex heuristics — zero I/O, zero LLM. The agent can call this
|
|
7
|
+
* opportunistically on streamed output to decide whether to surface a
|
|
8
|
+
* UI suggestion. Callers must pass a deduplication set to avoid
|
|
9
|
+
* re-suggesting the same UI repeatedly in a live session.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { computeIntentId } from './intent.js';
|
|
13
|
+
|
|
14
|
+
/** Suggestion event sent to the agent. */
|
|
15
|
+
export interface NegotiatorSuggestion {
|
|
16
|
+
type: 'negotiator:suggest';
|
|
17
|
+
trigger: 'stream-data-detected' | 'session-context' | 'user-pattern';
|
|
18
|
+
message: string;
|
|
19
|
+
suggestedAction: {
|
|
20
|
+
tool: 'ggui_push';
|
|
21
|
+
input: { data?: Record<string, unknown>; prompt?: string };
|
|
22
|
+
};
|
|
23
|
+
confidence: number;
|
|
24
|
+
intentId: string;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
interface DetectionPattern {
|
|
28
|
+
name: string;
|
|
29
|
+
regex: RegExp;
|
|
30
|
+
uiType: string;
|
|
31
|
+
confidence: number;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const PATTERNS: DetectionPattern[] = [
|
|
35
|
+
{ name: 'temperature', regex: /\b\d+\s*°[CF]\b/i, uiType: 'weather display', confidence: 0.8 },
|
|
36
|
+
{ name: 'percentage', regex: /\b\d+(\.\d+)?\s*%/, uiType: 'metrics display', confidence: 0.6 },
|
|
37
|
+
{ name: 'currency', regex: /\$\s*[\d,]+(\.\d{2})?|\b\d+(\.\d{2})?\s*(USD|EUR|GBP|JPY|KRW)\b/i, uiType: 'financial display', confidence: 0.7 },
|
|
38
|
+
{ name: 'speed-or-distance', regex: /\b\d+(\.\d+)?\s*(km\/h|mph|km|mi|m\/s)\b/i, uiType: 'data display', confidence: 0.6 },
|
|
39
|
+
{ name: 'numbered-list', regex: /(?:^|\n)\s*[1-9]\.\s+\S/m, uiType: 'step flow or list', confidence: 0.5 },
|
|
40
|
+
{ name: 'comparison', regex: /\b(vs\.?|versus|compared to|on the other hand|alternatively)\b/i, uiType: 'comparison view', confidence: 0.6 },
|
|
41
|
+
{ name: 'table-like', regex: /\|.*\|.*\|/, uiType: 'table or grid', confidence: 0.8 },
|
|
42
|
+
];
|
|
43
|
+
|
|
44
|
+
/** Detect structured data patterns in streamed text. */
|
|
45
|
+
export function detectDataPatterns(text: string): Array<{ pattern: string; uiType: string; confidence: number }> {
|
|
46
|
+
const matches: Array<{ pattern: string; uiType: string; confidence: number }> = [];
|
|
47
|
+
for (const p of PATTERNS) {
|
|
48
|
+
if (p.regex.test(text)) {
|
|
49
|
+
matches.push({ pattern: p.name, uiType: p.uiType, confidence: p.confidence });
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
return matches;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/** Build a suggestion event from detected patterns. */
|
|
56
|
+
export function buildSuggestion(
|
|
57
|
+
sessionId: string,
|
|
58
|
+
detections: Array<{ pattern: string; uiType: string; confidence: number }>,
|
|
59
|
+
activeIntentIds: Set<string>,
|
|
60
|
+
): NegotiatorSuggestion | null {
|
|
61
|
+
if (detections.length === 0) return null;
|
|
62
|
+
const best = detections.reduce((a, b) => (a.confidence > b.confidence ? a : b));
|
|
63
|
+
const intentId = computeIntentId(sessionId, { detectedPattern: best.pattern }, 'create');
|
|
64
|
+
if (activeIntentIds.has(intentId)) return null;
|
|
65
|
+
return {
|
|
66
|
+
type: 'negotiator:suggest',
|
|
67
|
+
trigger: 'stream-data-detected',
|
|
68
|
+
message: `I noticed structured data in your response (${best.pattern}). Consider calling ggui_push for an interactive ${best.uiType} instead.`,
|
|
69
|
+
suggestedAction: { tool: 'ggui_push', input: { prompt: best.uiType } },
|
|
70
|
+
confidence: best.confidence,
|
|
71
|
+
intentId,
|
|
72
|
+
};
|
|
73
|
+
}
|