@ggui-ai/negotiator 0.1.0-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +49 -0
- package/dist/contract-hash.d.ts +54 -0
- package/dist/contract-hash.d.ts.map +1 -0
- package/dist/contract-hash.js +96 -0
- package/dist/contract-validators.d.ts +171 -0
- package/dist/contract-validators.d.ts.map +1 -0
- package/dist/contract-validators.js +478 -0
- package/dist/decision-input.d.ts +48 -0
- package/dist/decision-input.d.ts.map +1 -0
- package/dist/decision-input.js +14 -0
- package/dist/decision.d.ts +54 -0
- package/dist/decision.d.ts.map +1 -0
- package/dist/decision.js +500 -0
- package/dist/index.d.ts +36 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +25 -0
- package/dist/intent.d.ts +22 -0
- package/dist/intent.d.ts.map +1 -0
- package/dist/intent.js +28 -0
- package/dist/llm-caller.d.ts +70 -0
- package/dist/llm-caller.d.ts.map +1 -0
- package/dist/llm-caller.js +38 -0
- package/dist/llm-rerank.d.ts +101 -0
- package/dist/llm-rerank.d.ts.map +1 -0
- package/dist/llm-rerank.js +178 -0
- package/dist/negotiate.d.ts +141 -0
- package/dist/negotiate.d.ts.map +1 -0
- package/dist/negotiate.js +161 -0
- package/dist/normalize-schema.d.ts +22 -0
- package/dist/normalize-schema.d.ts.map +1 -0
- package/dist/normalize-schema.js +191 -0
- package/dist/pure.d.ts +30 -0
- package/dist/pure.d.ts.map +1 -0
- package/dist/pure.js +43 -0
- package/dist/rag-search.d.ts +73 -0
- package/dist/rag-search.d.ts.map +1 -0
- package/dist/rag-search.js +192 -0
- package/dist/rerank-eval/pairs.d.ts +28 -0
- package/dist/rerank-eval/pairs.d.ts.map +1 -0
- package/dist/rerank-eval/pairs.js +531 -0
- package/dist/rerank-eval/run-probe-cli.d.ts +3 -0
- package/dist/rerank-eval/run-probe-cli.d.ts.map +1 -0
- package/dist/rerank-eval/run-probe-cli.js +146 -0
- package/dist/rerank-eval/run-probe.d.ts +68 -0
- package/dist/rerank-eval/run-probe.d.ts.map +1 -0
- package/dist/rerank-eval/run-probe.js +113 -0
- package/dist/session.d.ts +42 -0
- package/dist/session.d.ts.map +1 -0
- package/dist/session.js +21 -0
- package/dist/suggestion.d.ts +38 -0
- package/dist/suggestion.d.ts.map +1 -0
- package/dist/suggestion.js +47 -0
- package/dist/synth-bench/corpus.d.ts +106 -0
- package/dist/synth-bench/corpus.d.ts.map +1 -0
- package/dist/synth-bench/corpus.js +994 -0
- package/dist/synth-bench/run-bench-cli.d.ts +3 -0
- package/dist/synth-bench/run-bench-cli.d.ts.map +1 -0
- package/dist/synth-bench/run-bench-cli.js +181 -0
- package/dist/synth-bench/run-bench.d.ts +101 -0
- package/dist/synth-bench/run-bench.d.ts.map +1 -0
- package/dist/synth-bench/run-bench.js +374 -0
- package/dist/synthesize-contract.d.ts +131 -0
- package/dist/synthesize-contract.d.ts.map +1 -0
- package/dist/synthesize-contract.js +948 -0
- package/dist/types.d.ts +30 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +13 -0
- package/package.json +74 -0
- package/src/contract-hash.ts +102 -0
- package/src/contract-validators.ts +604 -0
- package/src/decision-input.ts +49 -0
- package/src/decision.ts +581 -0
- package/src/index.ts +63 -0
- package/src/intent.ts +37 -0
- package/src/llm-caller.ts +82 -0
- package/src/llm-rerank.ts +280 -0
- package/src/negotiate.ts +312 -0
- package/src/normalize-schema.ts +193 -0
- package/src/pure.ts +46 -0
- package/src/rag-search.ts +274 -0
- package/src/rerank-eval/pairs.ts +624 -0
- package/src/rerank-eval/run-probe-cli.ts +197 -0
- package/src/rerank-eval/run-probe.ts +198 -0
- package/src/session.ts +41 -0
- package/src/suggestion.ts +73 -0
- package/src/synth-bench/corpus.ts +1126 -0
- package/src/synth-bench/run-bench-cli.ts +237 -0
- package/src/synth-bench/run-bench.ts +525 -0
- package/src/synthesize-contract.ts +1161 -0
- package/src/types.ts +31 -0
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `LLMCaller` — the decision engine's LLM dispatcher.
|
|
3
|
+
*
|
|
4
|
+
* Narrow abstraction over "call a chat model, optionally with a forced
|
|
5
|
+
* tool-use schema for guaranteed-JSON structured output." Kept public
|
|
6
|
+
* so OSS consumers of `@ggui-ai/negotiator` can bring their own LLM
|
|
7
|
+
* provider (Anthropic direct, OpenAI, Google, a local model, a
|
|
8
|
+
* community LiteLLM wrapper) without touching the decision-engine
|
|
9
|
+
* source.
|
|
10
|
+
*
|
|
11
|
+
* **Why this lives in `@ggui-ai/negotiator`, not
|
|
12
|
+
* `@ggui-ai/mcp-server-core`.** `mcp-server-core` contains the
|
|
13
|
+
* storage + runtime seams an MCP server implementer binds against
|
|
14
|
+
* (`VectorStore`, `EmbeddingProvider`, `KeyValueStore`,
|
|
15
|
+
* `BlueprintProvider`, `Negotiator`). `LLMCaller` is an
|
|
16
|
+
* engine-internal dispatcher — one level below `Negotiator` — so
|
|
17
|
+
* lifting it to `mcp-server-core` would grow the public seam count
|
|
18
|
+
* speculatively. If a second consumer outside the negotiator
|
|
19
|
+
* surfaces later, the "where does `LLMCaller` live?" question can be
|
|
20
|
+
* re-opened at that point.
|
|
21
|
+
*
|
|
22
|
+
* Normative semantics:
|
|
23
|
+
* - `call(systemPrompt, userMessage, maxTokens?)` returns the raw
|
|
24
|
+
* model text. Implementations MUST NOT inject tool-use blocks when
|
|
25
|
+
* the caller didn't request them — the text path is used as a
|
|
26
|
+
* regex-JSON fallback.
|
|
27
|
+
* - `callStructured?<T>(...)` is OPTIONAL. When present, it MUST
|
|
28
|
+
* force tool use against the supplied `ToolSchema` and return the
|
|
29
|
+
* tool input, parsed as `T`. Implementations that don't support
|
|
30
|
+
* forced structured output simply omit this method; consumers
|
|
31
|
+
* fall back to `call` + regex JSON extraction. Absence is not an
|
|
32
|
+
* error.
|
|
33
|
+
* - `ToolSchema.input_schema` follows the OpenAI tool-use JSON
|
|
34
|
+
* Schema convention. Implementations that use a different
|
|
35
|
+
* tool-use protocol (e.g., Anthropic's variant) MUST translate at
|
|
36
|
+
* the adapter boundary.
|
|
37
|
+
*/
|
|
38
|
+
|
|
39
|
+
/** Tool schema for structured output via forced tool use. */
|
|
40
|
+
export interface ToolSchema {
|
|
41
|
+
name: string;
|
|
42
|
+
description: string;
|
|
43
|
+
input_schema: Record<string, unknown>;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** Chat-model dispatcher consumed by the negotiator decision engine. */
|
|
47
|
+
export interface LLMCaller {
|
|
48
|
+
/**
|
|
49
|
+
* Call the model in plain-text mode. `maxTokens` defaults to
|
|
50
|
+
* something implementation-appropriate (usually 2048).
|
|
51
|
+
*/
|
|
52
|
+
call(
|
|
53
|
+
systemPrompt: string,
|
|
54
|
+
userMessage: string,
|
|
55
|
+
maxTokens?: number,
|
|
56
|
+
): Promise<string>;
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Call with forced tool use for guaranteed structured JSON output.
|
|
60
|
+
* Implementations that can't force tool use should omit this
|
|
61
|
+
* method — consumers detect absence and fall back to regex JSON
|
|
62
|
+
* extraction on the text path.
|
|
63
|
+
*/
|
|
64
|
+
callStructured?<T>(
|
|
65
|
+
systemPrompt: string,
|
|
66
|
+
userMessage: string,
|
|
67
|
+
tool: ToolSchema,
|
|
68
|
+
maxTokens?: number,
|
|
69
|
+
): Promise<T>;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Provider + model selector for factory-style LLM caller
|
|
74
|
+
* construction. The `provider` enum stays narrow to the ones
|
|
75
|
+
* ggui supports today; community adapters can extend by widening
|
|
76
|
+
* the union at their own boundary.
|
|
77
|
+
*/
|
|
78
|
+
export interface LLMCallerConfig {
|
|
79
|
+
provider: 'anthropic' | 'openai' | 'google' | 'openrouter' | 'bedrock';
|
|
80
|
+
model: string;
|
|
81
|
+
apiKey?: string;
|
|
82
|
+
}
|
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* LLM rerank — Tier-2 precision oracle for the blueprint registry.
|
|
3
|
+
*
|
|
4
|
+
* Given a user's UI request (intent + contract structure) and a set
|
|
5
|
+
* of candidate cached blueprints retrieved by RAG, ask a fast LLM
|
|
6
|
+
* (Haiku 4.5) which candidate (if any) matches. Returns a structured
|
|
7
|
+
* decision so the caller can branch deterministically.
|
|
8
|
+
*
|
|
9
|
+
* This module is the precision half of the blueprint-first
|
|
10
|
+
* architecture: RAG retrieval is high-recall but low-precision (bge-
|
|
11
|
+
* small confuses topic-similar but UI-divergent prompts); the LLM
|
|
12
|
+
* judge restores precision. Combined break-even hit rate is ~10%;
|
|
13
|
+
* realistic workloads observe 30-70%.
|
|
14
|
+
*/
|
|
15
|
+
import { summarizeContract } from '@ggui-ai/protocol';
|
|
16
|
+
import type { LLMCaller, ToolSchema } from './llm-caller.js';
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* One candidate blueprint for the LLM judge to consider.
|
|
20
|
+
*
|
|
21
|
+
* Keep this struct narrow — the prompt sees only what's necessary
|
|
22
|
+
* to decide match-vs-no-match. componentCode is intentionally
|
|
23
|
+
* absent (huge, distracting, doesn't change the decision).
|
|
24
|
+
*/
|
|
25
|
+
export interface RerankCandidate {
|
|
26
|
+
/** Stable blueprint id — echoed back as `matchId` on a hit. */
|
|
27
|
+
readonly id: string;
|
|
28
|
+
/** The intent prose that originally produced this blueprint. */
|
|
29
|
+
readonly cachedIntent: string;
|
|
30
|
+
/**
|
|
31
|
+
* One-line summary of the blueprint's contract surface. Format
|
|
32
|
+
* matches `summarizeContract()` below — `slots=...; actions=...;
|
|
33
|
+
* streams=...; props=...` so the judge sees the structural shape
|
|
34
|
+
* without the JSON noise.
|
|
35
|
+
*/
|
|
36
|
+
readonly cachedContractSummary: string;
|
|
37
|
+
/**
|
|
38
|
+
* Optional retrieval signal. Higher cosine biases the prior, but
|
|
39
|
+
* the judge's decision is the source of truth. Pass-through so
|
|
40
|
+
* the prompt can include "candidate retrieved at cosine X" if the
|
|
41
|
+
* judge would benefit (default: omit).
|
|
42
|
+
*/
|
|
43
|
+
readonly cosine?: number;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** Decision returned by the LLM judge. */
|
|
47
|
+
export interface RerankDecision {
|
|
48
|
+
/**
|
|
49
|
+
* The matched candidate's id, or `null` if no candidate matches.
|
|
50
|
+
* `null` means "all candidates rejected — generate fresh."
|
|
51
|
+
*/
|
|
52
|
+
readonly matchId: string | null;
|
|
53
|
+
/**
|
|
54
|
+
* Confidence on `[0, 1]`. Caller compares against a threshold (e.g.
|
|
55
|
+
* 0.6) before treating the decision as a hit. Returned even when
|
|
56
|
+
* matchId is null so callers can log "judge declined with
|
|
57
|
+
* confidence X."
|
|
58
|
+
*/
|
|
59
|
+
readonly confidence: number;
|
|
60
|
+
/**
|
|
61
|
+
* Free-text reason from the judge. Surface in trace logs so
|
|
62
|
+
* operators can debug "why didn't this hit." Truncate at the
|
|
63
|
+
* persistence boundary if cardinality is a concern.
|
|
64
|
+
*/
|
|
65
|
+
readonly reason: string;
|
|
66
|
+
/** Wall-clock latency of the LLM call. */
|
|
67
|
+
readonly latencyMs: number;
|
|
68
|
+
/**
|
|
69
|
+
* Token cost of the call — for the cache-trace sink and cost
|
|
70
|
+
* accounting. Implementations that can't surface token counts may
|
|
71
|
+
* report `{input: 0, output: 0}` and the cost-per-call gate will
|
|
72
|
+
* have to be measured externally.
|
|
73
|
+
*/
|
|
74
|
+
readonly tokenCost: { readonly input: number; readonly output: number };
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** Query the user's request the judge is matching against. */
|
|
78
|
+
export interface RerankQuery {
|
|
79
|
+
readonly intent: string;
|
|
80
|
+
readonly contractSummary: string;
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
const RERANK_SYSTEM_PROMPT = `You match user UI requests against previously-generated UI blueprints. Each blueprint was produced for a past request and stored. Decide whether any candidate produces the SAME USEFUL UI for the current request.
|
|
84
|
+
|
|
85
|
+
MATCH means the candidate would correctly satisfy the user's current request — same UI shape (component types, layout pattern), same wire surface (slot names, action names), same intended user task, and same load-bearing parameters (dates, months, ranges, enum values).
|
|
86
|
+
|
|
87
|
+
NO-MATCH means the candidate would NOT satisfy the user's current request — different task (haiku composer vs tweet draft, login vs signup), different UI shape (form vs list vs dashboard), or load-bearing parameters differ (calendar-Jan vs calendar-Mar — same contract, different value).
|
|
88
|
+
|
|
89
|
+
Visual style differences alone (minimal vs ornate, dense vs spacious) DO NOT block a match — the user can refine those after they get a working UI.
|
|
90
|
+
|
|
91
|
+
Output exactly ONE tool call with your decision. confidence is a number in [0, 1]. matchId is a string from the candidate ids, or null when no candidate matches. reason is a short sentence the operator can use to debug.`;
|
|
92
|
+
|
|
93
|
+
const RERANK_TOOL: ToolSchema = {
|
|
94
|
+
name: 'submit_rerank_decision',
|
|
95
|
+
description:
|
|
96
|
+
'Submit your match-vs-no-match decision over the candidates.',
|
|
97
|
+
input_schema: {
|
|
98
|
+
type: 'object',
|
|
99
|
+
additionalProperties: false,
|
|
100
|
+
properties: {
|
|
101
|
+
matchId: {
|
|
102
|
+
type: ['string', 'null'],
|
|
103
|
+
description:
|
|
104
|
+
"ID of the matching candidate, or null when no candidate matches. MUST be one of the candidate ids supplied in the user message, or null.",
|
|
105
|
+
},
|
|
106
|
+
confidence: {
|
|
107
|
+
type: 'number',
|
|
108
|
+
minimum: 0,
|
|
109
|
+
maximum: 1,
|
|
110
|
+
description:
|
|
111
|
+
'Confidence in the decision on [0, 1]. Caller will compare against a threshold before treating it as a hit.',
|
|
112
|
+
},
|
|
113
|
+
reason: {
|
|
114
|
+
type: 'string',
|
|
115
|
+
description:
|
|
116
|
+
'Brief explanation — one sentence — that the operator can use to debug match decisions.',
|
|
117
|
+
},
|
|
118
|
+
},
|
|
119
|
+
required: ['matchId', 'confidence', 'reason'],
|
|
120
|
+
},
|
|
121
|
+
};
|
|
122
|
+
|
|
123
|
+
function buildUserMessage(
|
|
124
|
+
query: RerankQuery,
|
|
125
|
+
candidates: readonly RerankCandidate[],
|
|
126
|
+
): string {
|
|
127
|
+
const lines: string[] = [];
|
|
128
|
+
lines.push('CURRENT REQUEST');
|
|
129
|
+
lines.push(` intent: ${query.intent}`);
|
|
130
|
+
lines.push(` contract: ${query.contractSummary}`);
|
|
131
|
+
lines.push('');
|
|
132
|
+
lines.push(`CANDIDATES (${candidates.length})`);
|
|
133
|
+
for (const c of candidates) {
|
|
134
|
+
lines.push('---');
|
|
135
|
+
lines.push(` id: ${c.id}`);
|
|
136
|
+
lines.push(` intent: ${truncate(c.cachedIntent, 280)}`);
|
|
137
|
+
lines.push(` contract: ${c.cachedContractSummary}`);
|
|
138
|
+
if (typeof c.cosine === 'number') {
|
|
139
|
+
lines.push(` cosine: ${c.cosine.toFixed(3)}`);
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
lines.push('');
|
|
143
|
+
lines.push(
|
|
144
|
+
'Decide: does any candidate match the current request? Submit via the tool.',
|
|
145
|
+
);
|
|
146
|
+
return lines.join('\n');
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
function truncate(text: string, max: number): string {
|
|
150
|
+
if (text.length <= max) return text;
|
|
151
|
+
return `${text.slice(0, max - 1)}…`;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
// Re-export `summarizeContract` from protocol for backwards-compatible
|
|
155
|
+
// access through `@ggui-ai/negotiator/llm-rerank` consumers (the
|
|
156
|
+
// canonical home is `@ggui-ai/protocol` — both the registry storage
|
|
157
|
+
// and the rerank prompt depend on it). Kept as a re-export so the
|
|
158
|
+
// existing test imports here keep working.
|
|
159
|
+
export { summarizeContract };
|
|
160
|
+
|
|
161
|
+
interface RerankToolInput {
|
|
162
|
+
matchId: string | null;
|
|
163
|
+
confidence: number;
|
|
164
|
+
reason: string;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
function clampConfidence(value: unknown): number {
|
|
168
|
+
if (typeof value !== 'number' || !Number.isFinite(value)) return 0;
|
|
169
|
+
if (value < 0) return 0;
|
|
170
|
+
if (value > 1) return 1;
|
|
171
|
+
return value;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
function parseToolInput(
|
|
175
|
+
raw: unknown,
|
|
176
|
+
candidateIds: ReadonlySet<string>,
|
|
177
|
+
): { matchId: string | null; confidence: number; reason: string } {
|
|
178
|
+
if (raw === null || typeof raw !== 'object') {
|
|
179
|
+
return { matchId: null, confidence: 0, reason: 'parse-failed: non-object tool input' };
|
|
180
|
+
}
|
|
181
|
+
const obj = raw as Record<string, unknown>;
|
|
182
|
+
const reason = typeof obj['reason'] === 'string' ? obj['reason'] : '';
|
|
183
|
+
const confidence = clampConfidence(obj['confidence']);
|
|
184
|
+
const rawMatchId = obj['matchId'];
|
|
185
|
+
if (rawMatchId === null || rawMatchId === undefined) {
|
|
186
|
+
return { matchId: null, confidence, reason };
|
|
187
|
+
}
|
|
188
|
+
if (typeof rawMatchId !== 'string') {
|
|
189
|
+
return {
|
|
190
|
+
matchId: null,
|
|
191
|
+
confidence: 0,
|
|
192
|
+
reason: `parse-failed: matchId was ${typeof rawMatchId}, not string|null`,
|
|
193
|
+
};
|
|
194
|
+
}
|
|
195
|
+
if (!candidateIds.has(rawMatchId)) {
|
|
196
|
+
return {
|
|
197
|
+
matchId: null,
|
|
198
|
+
confidence: 0,
|
|
199
|
+
reason: `judge returned matchId='${rawMatchId}' but that id is not in the candidate set — treating as no-match`,
|
|
200
|
+
};
|
|
201
|
+
}
|
|
202
|
+
return { matchId: rawMatchId, confidence, reason };
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/**
|
|
206
|
+
* Run the rerank judge against a query + candidate list.
|
|
207
|
+
*
|
|
208
|
+
* Empty candidate list short-circuits without an LLM call — the
|
|
209
|
+
* caller's RAG step decided there's nothing close enough; we don't
|
|
210
|
+
* burn tokens to confirm it.
|
|
211
|
+
*
|
|
212
|
+
* Operational errors (LLM throws, parse fails) collapse to a `null`
|
|
213
|
+
* match with confidence=0 and a diagnostic reason. The caller treats
|
|
214
|
+
* confidence-below-threshold as no-match anyway, so the failure mode
|
|
215
|
+
* lands on the cold-gen path automatically.
|
|
216
|
+
*/
|
|
217
|
+
export async function rerankCandidates(
|
|
218
|
+
deps: { readonly llm: LLMCaller },
|
|
219
|
+
query: RerankQuery,
|
|
220
|
+
candidates: readonly RerankCandidate[],
|
|
221
|
+
): Promise<RerankDecision> {
|
|
222
|
+
const startedAt = Date.now();
|
|
223
|
+
if (candidates.length === 0) {
|
|
224
|
+
return {
|
|
225
|
+
matchId: null,
|
|
226
|
+
confidence: 0,
|
|
227
|
+
reason: 'no candidates — short-circuited without LLM call',
|
|
228
|
+
latencyMs: Date.now() - startedAt,
|
|
229
|
+
tokenCost: { input: 0, output: 0 },
|
|
230
|
+
};
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
const userMessage = buildUserMessage(query, candidates);
|
|
234
|
+
const candidateIds = new Set(candidates.map((c) => c.id));
|
|
235
|
+
|
|
236
|
+
if (typeof deps.llm.callStructured !== 'function') {
|
|
237
|
+
return {
|
|
238
|
+
matchId: null,
|
|
239
|
+
confidence: 0,
|
|
240
|
+
reason:
|
|
241
|
+
'llm-rerank: provider does not support callStructured (forced tool-use). Bind a structured-capable LLMCaller (Anthropic adapter via mcp-server) for rerank.',
|
|
242
|
+
latencyMs: Date.now() - startedAt,
|
|
243
|
+
tokenCost: { input: 0, output: 0 },
|
|
244
|
+
};
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
let toolInput: unknown;
|
|
248
|
+
try {
|
|
249
|
+
toolInput = await deps.llm.callStructured<RerankToolInput>(
|
|
250
|
+
RERANK_SYSTEM_PROMPT,
|
|
251
|
+
userMessage,
|
|
252
|
+
RERANK_TOOL,
|
|
253
|
+
512,
|
|
254
|
+
);
|
|
255
|
+
} catch (err) {
|
|
256
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
257
|
+
return {
|
|
258
|
+
matchId: null,
|
|
259
|
+
confidence: 0,
|
|
260
|
+
reason: `llm-rerank: callStructured threw — ${message}`,
|
|
261
|
+
latencyMs: Date.now() - startedAt,
|
|
262
|
+
tokenCost: { input: 0, output: 0 },
|
|
263
|
+
};
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
const parsed = parseToolInput(toolInput, candidateIds);
|
|
267
|
+
return {
|
|
268
|
+
matchId: parsed.matchId,
|
|
269
|
+
confidence: parsed.confidence,
|
|
270
|
+
reason: parsed.reason,
|
|
271
|
+
latencyMs: Date.now() - startedAt,
|
|
272
|
+
// Token cost surfacing requires LLMCaller-level instrumentation
|
|
273
|
+
// we don't have today. Default to zero; the cost gate is measured
|
|
274
|
+
// out-of-band from billing data during the probe.
|
|
275
|
+
tokenCost: { input: 0, output: 0 },
|
|
276
|
+
};
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
// Re-exports for the eval harness — keep public surface explicit.
|
|
280
|
+
export { RERANK_SYSTEM_PROMPT, RERANK_TOOL };
|
package/src/negotiate.ts
ADDED
|
@@ -0,0 +1,312 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Negotiator — top-level orchestrator over the public storage +
|
|
3
|
+
* decision-engine seams.
|
|
4
|
+
*
|
|
5
|
+
* Pipeline:
|
|
6
|
+
* 1. RAG search (per-scope + optional shared pool) via
|
|
7
|
+
* {@link ragSearch} — composes `EmbeddingProvider.embed` +
|
|
8
|
+
* `VectorStore.query` from `@ggui-ai/mcp-server-core`.
|
|
9
|
+
* 2. Read session state (optional injectable).
|
|
10
|
+
* 3. Fast-path for exact blueprint hits — skip the decision LLM.
|
|
11
|
+
* 4. Otherwise call {@link makeDecision} with the RAG candidates and
|
|
12
|
+
* session stack; fold the picked blueprint's pool provenance into
|
|
13
|
+
* the return value.
|
|
14
|
+
*
|
|
15
|
+
* Timing logs (stable format — consumed by benchmarks):
|
|
16
|
+
* [negotiate] embedding: Xms | search: Xms | candidates: N
|
|
17
|
+
* [negotiate] candidate: <id8> | <description> | hash=<h16>
|
|
18
|
+
* [negotiate] FAST PATH: blueprint=<id8> hash=<h12> | LLM decision: 0ms | total: Xms
|
|
19
|
+
* [negotiate] LLM PATH: action=X blueprint=<id8> hash=<h12> | LLM decision: Xms | total: Xms
|
|
20
|
+
*
|
|
21
|
+
* ### Public surface + semver weight
|
|
22
|
+
*
|
|
23
|
+
* Exported:
|
|
24
|
+
* - `negotiate(deps, input)` — runtime orchestrator.
|
|
25
|
+
* - `NegotiateDeps` — injection shape (embedding / vectors / llm +
|
|
26
|
+
* optional session-state reader + optional progress callback).
|
|
27
|
+
* - `NegotiateInput` — agent signal + config.
|
|
28
|
+
* - `NegotiateConfig` — minimum fields the orchestrator actually
|
|
29
|
+
* reads. Pool selection is expressed as `includeSharedPool:
|
|
30
|
+
* boolean` rather than a `poolMode` enum, so callers decide
|
|
31
|
+
* shared-pool inclusion explicitly at each call site.
|
|
32
|
+
* - `NegotiateResult` — the decision result returned to callers.
|
|
33
|
+
*
|
|
34
|
+
* ### Why this package, not `mcp-server-core`
|
|
35
|
+
*
|
|
36
|
+
* `mcp-server-core` locks storage/runtime seams MCP server
|
|
37
|
+
* implementers bind against (`EmbeddingProvider`, `VectorStore`,
|
|
38
|
+
* `Negotiator`). `negotiate()` is a *composition* over those seams —
|
|
39
|
+
* decision-engine semantics, not a new seam. Adding it to
|
|
40
|
+
* `mcp-server-core` would drag the LLM prompts + tool-schemas into a
|
|
41
|
+
* package whose job is to stay minimal and runtime-agnostic.
|
|
42
|
+
*/
|
|
43
|
+
|
|
44
|
+
import type {
|
|
45
|
+
DataContract,
|
|
46
|
+
NegotiatorAlternative,
|
|
47
|
+
NegotiatorDecision,
|
|
48
|
+
} from '@ggui-ai/protocol';
|
|
49
|
+
import type {
|
|
50
|
+
EmbeddingProvider,
|
|
51
|
+
VectorStore,
|
|
52
|
+
} from '@ggui-ai/mcp-server-core';
|
|
53
|
+
import type { LLMCaller } from './llm-caller.js';
|
|
54
|
+
import type { SessionState } from './session.js';
|
|
55
|
+
import type { NegotiatorDecisionInput } from './decision-input.js';
|
|
56
|
+
import { ragSearch } from './rag-search.js';
|
|
57
|
+
import { makeDecision } from './decision.js';
|
|
58
|
+
|
|
59
|
+
/** Empty session state for cold starts and benchmarks. */
|
|
60
|
+
const EMPTY_SESSION: SessionState = {
|
|
61
|
+
stack: [],
|
|
62
|
+
conversationHistory: [],
|
|
63
|
+
};
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Minimum config the orchestrator reads.
|
|
67
|
+
*
|
|
68
|
+
* `appId` is the primary RAG scope (per-app registered UIs live
|
|
69
|
+
* here). `sessionId` keys the optional `readSessionState` callback.
|
|
70
|
+
* `includeSharedPool` (default `false`) gates whether to also search
|
|
71
|
+
* the global `"shared"` pool in parallel and fold its hits into the
|
|
72
|
+
* candidate set.
|
|
73
|
+
*/
|
|
74
|
+
export interface NegotiateConfig {
|
|
75
|
+
appId: string;
|
|
76
|
+
sessionId: string;
|
|
77
|
+
/** Also search the shared (global) pool in parallel. Default `false`. */
|
|
78
|
+
includeSharedPool?: boolean;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Agent signal + config for a single negotiation call. Bundled as one
|
|
83
|
+
* object so future additive fields don't force yet another positional
|
|
84
|
+
* arg on the public API.
|
|
85
|
+
*/
|
|
86
|
+
export interface NegotiateInput {
|
|
87
|
+
agent: {
|
|
88
|
+
/** Raw data the agent wants to render. */
|
|
89
|
+
data?: Record<string, unknown>;
|
|
90
|
+
/** Natural-language prompt. Used as RAG query when present. */
|
|
91
|
+
prompt?: string;
|
|
92
|
+
/** Free-form context string or structured map. Forwarded to the decision LLM. */
|
|
93
|
+
context?: string | Record<string, unknown>;
|
|
94
|
+
/**
|
|
95
|
+
* Names of MCP tools the agent invokes (catalog seed). Merged
|
|
96
|
+
* into the decision's `agentCapabilities.tools` deterministically
|
|
97
|
+
* (see {@link makeDecision}).
|
|
98
|
+
*/
|
|
99
|
+
agentTools?: string[];
|
|
100
|
+
/**
|
|
101
|
+
* Browser-capability gadget catalog the app exposes (default
|
|
102
|
+
* `STDLIB_GADGETS`, operator-extensible). Forwarded to
|
|
103
|
+
* the decision LLM so it knows which gadget bindings are
|
|
104
|
+
* available; canonical entries enrich partial LLM output
|
|
105
|
+
* downstream (see {@link mergeGadgets}).
|
|
106
|
+
*/
|
|
107
|
+
gadgets?: readonly import('@ggui-ai/protocol').GadgetDescriptor[];
|
|
108
|
+
};
|
|
109
|
+
config: NegotiateConfig;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Dependencies injected into {@link negotiate}. Public field names
|
|
114
|
+
* (`embedding` / `vectors` / `llm`) align with {@link ragSearch}'s
|
|
115
|
+
* already-shipped deps shape.
|
|
116
|
+
*/
|
|
117
|
+
export interface NegotiateDeps {
|
|
118
|
+
/**
|
|
119
|
+
* Embedding provider for the RAG search step. **Optional** —
|
|
120
|
+
* when omitted (paired with omitted `vectors`), the negotiator
|
|
121
|
+
* skips RAG entirely and runs the decision LLM against an empty
|
|
122
|
+
* candidate list. OSS without vector-store infrastructure binds
|
|
123
|
+
* with `embedding: undefined, vectors: undefined` and still gets
|
|
124
|
+
* useful negotiation via the decision LLM alone.
|
|
125
|
+
*
|
|
126
|
+
* `embedding` and `vectors` are paired — both must be present for
|
|
127
|
+
* RAG to fire, or both must be absent. A half-bound config (one
|
|
128
|
+
* present, one missing) is a configuration bug; the search step
|
|
129
|
+
* skips and logs a warn rather than erroring, but consumers
|
|
130
|
+
* should fix the binding.
|
|
131
|
+
*/
|
|
132
|
+
embedding?: EmbeddingProvider;
|
|
133
|
+
vectors?: VectorStore;
|
|
134
|
+
llm: LLMCaller;
|
|
135
|
+
/** Optional: read current session state for stack-aware decisions. */
|
|
136
|
+
readSessionState?: (sessionId: string) => Promise<SessionState | null>;
|
|
137
|
+
/** Optional: surface pipeline progress to consumers. */
|
|
138
|
+
onProgress?: (phase: string, summary: string) => void;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* Output shape — mirrors the legacy `DecisionResult` so the
|
|
143
|
+
* back-compat shim can return the object verbatim. See `NegotiateDeps`
|
|
144
|
+
* semver note.
|
|
145
|
+
*/
|
|
146
|
+
export interface NegotiateResult {
|
|
147
|
+
decision: NegotiatorDecision;
|
|
148
|
+
alternatives: NegotiatorAlternative[];
|
|
149
|
+
/** Stored contract hash from blueprint match — deterministic pool key. */
|
|
150
|
+
storedContractHash?: string;
|
|
151
|
+
/** Which pool the matched blueprint's code lives in. */
|
|
152
|
+
storedPoolSource?: 'shared' | 'private';
|
|
153
|
+
embeddingLatencyMs: number;
|
|
154
|
+
searchLatencyMs: number;
|
|
155
|
+
decisionLatencyMs: number;
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
/**
|
|
159
|
+
* Orchestrate one negotiation call — RAG search, session read, fast
|
|
160
|
+
* path, decision LLM. See module docstring for the pipeline outline.
|
|
161
|
+
*/
|
|
162
|
+
export async function negotiate(
|
|
163
|
+
deps: NegotiateDeps,
|
|
164
|
+
input: NegotiateInput,
|
|
165
|
+
): Promise<NegotiateResult> {
|
|
166
|
+
const { agent, config } = input;
|
|
167
|
+
const negotiateStart = Date.now();
|
|
168
|
+
deps.onProgress?.('negotiating', 'Analyzing request...');
|
|
169
|
+
|
|
170
|
+
// Step 1: RAG search (per-app + optional shared pool in parallel).
|
|
171
|
+
// `embedding` + `vectors` are optional. When either is missing,
|
|
172
|
+
// ragSearch is a no-op (returns empty options + 0 latency); the
|
|
173
|
+
// decision LLM runs against zero candidates and falls back to
|
|
174
|
+
// "create" via its standard branching. OSS without RAG infrastructure
|
|
175
|
+
// gets useful negotiation from the decision LLM alone.
|
|
176
|
+
const queryText = agent.prompt ?? JSON.stringify(agent.data ?? {});
|
|
177
|
+
const emptyResult = { options: [], embeddingLatencyMs: 0, searchLatencyMs: 0 };
|
|
178
|
+
const ragDeps =
|
|
179
|
+
deps.embedding !== undefined && deps.vectors !== undefined
|
|
180
|
+
? { embedding: deps.embedding, vectors: deps.vectors }
|
|
181
|
+
: undefined;
|
|
182
|
+
const [appResult, sharedResult] = await Promise.all([
|
|
183
|
+
ragDeps
|
|
184
|
+
? ragSearch(ragDeps, { prompt: queryText, scope: config.appId })
|
|
185
|
+
: Promise.resolve(emptyResult),
|
|
186
|
+
ragDeps && config.includeSharedPool
|
|
187
|
+
? ragSearch(ragDeps, { prompt: queryText, scope: 'shared' })
|
|
188
|
+
: Promise.resolve(emptyResult),
|
|
189
|
+
]);
|
|
190
|
+
|
|
191
|
+
const ragResult = {
|
|
192
|
+
options: [...appResult.options, ...sharedResult.options],
|
|
193
|
+
embeddingLatencyMs: Math.max(
|
|
194
|
+
appResult.embeddingLatencyMs,
|
|
195
|
+
sharedResult.embeddingLatencyMs,
|
|
196
|
+
),
|
|
197
|
+
searchLatencyMs: Math.max(
|
|
198
|
+
appResult.searchLatencyMs,
|
|
199
|
+
sharedResult.searchLatencyMs,
|
|
200
|
+
),
|
|
201
|
+
};
|
|
202
|
+
|
|
203
|
+
// eslint-disable-next-line no-console
|
|
204
|
+
console.log(
|
|
205
|
+
`[negotiate] embedding: ${ragResult.embeddingLatencyMs}ms | search: ${ragResult.searchLatencyMs}ms | candidates: ${ragResult.options.length}`,
|
|
206
|
+
);
|
|
207
|
+
for (const opt of ragResult.options) {
|
|
208
|
+
// eslint-disable-next-line no-console
|
|
209
|
+
console.log(
|
|
210
|
+
`[negotiate] candidate: ${opt.blueprintId?.slice(-8) ?? 'none'} | ${opt.description.slice(0, 80)} | hash=${opt.contractHash?.slice(0, 16) ?? 'none'}`,
|
|
211
|
+
);
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
deps.onProgress?.(
|
|
215
|
+
'blueprint_search',
|
|
216
|
+
ragResult.options.length > 0
|
|
217
|
+
? `Found ${ragResult.options.length} blueprint candidate${ragResult.options.length > 1 ? 's' : ''}`
|
|
218
|
+
: 'No blueprints found',
|
|
219
|
+
);
|
|
220
|
+
|
|
221
|
+
// Step 2: Read session state (falls back to EMPTY_SESSION).
|
|
222
|
+
const sessionState = deps.readSessionState
|
|
223
|
+
? ((await deps.readSessionState(config.sessionId)) ?? EMPTY_SESSION)
|
|
224
|
+
: EMPTY_SESSION;
|
|
225
|
+
|
|
226
|
+
// Step 3: Fast-path for high-confidence exact matches — skip decision LLM.
|
|
227
|
+
const exactOpt = ragResult.options.find((opt) =>
|
|
228
|
+
opt.description.includes('exact'),
|
|
229
|
+
);
|
|
230
|
+
if (exactOpt) {
|
|
231
|
+
const stackHasSameType = sessionState.stack.some(
|
|
232
|
+
(item) => item.prompt && agent.prompt && item.prompt === agent.prompt,
|
|
233
|
+
);
|
|
234
|
+
const action = stackHasSameType ? ('update' as const) : ('create' as const);
|
|
235
|
+
const totalMs = Date.now() - negotiateStart;
|
|
236
|
+
|
|
237
|
+
// eslint-disable-next-line no-console
|
|
238
|
+
console.log(
|
|
239
|
+
`[negotiate] FAST PATH: blueprint=${exactOpt.blueprintId?.slice(-8)} hash=${exactOpt.contractHash?.slice(0, 12) ?? 'none'} | LLM decision: 0ms | total: ${totalMs}ms`,
|
|
240
|
+
);
|
|
241
|
+
deps.onProgress?.('deciding', 'Exact match found — fast path');
|
|
242
|
+
|
|
243
|
+
// The agent's prompt is the outer-pipeline intent (`intent` is not
|
|
244
|
+
// a contract field). Fallback contract is the empty contract — the
|
|
245
|
+
// four-spec surface is omitted entirely (no props, no actions, no
|
|
246
|
+
// streams, no context).
|
|
247
|
+
const fallbackContract: DataContract = {};
|
|
248
|
+
return {
|
|
249
|
+
decision: {
|
|
250
|
+
action,
|
|
251
|
+
reasoning: `Exact blueprint match (high confidence). ${action === 'update' ? 'Updating existing view.' : 'Creating new view.'}`,
|
|
252
|
+
blueprintId: exactOpt.blueprintId,
|
|
253
|
+
contract: exactOpt.contract ?? fallbackContract,
|
|
254
|
+
},
|
|
255
|
+
alternatives: [],
|
|
256
|
+
storedContractHash: exactOpt.contractHash,
|
|
257
|
+
storedPoolSource: exactOpt.poolSource,
|
|
258
|
+
embeddingLatencyMs: ragResult.embeddingLatencyMs,
|
|
259
|
+
searchLatencyMs: ragResult.searchLatencyMs,
|
|
260
|
+
decisionLatencyMs: 0,
|
|
261
|
+
};
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
// Step 4: Build decision input + call the LLM.
|
|
265
|
+
const decisionInput: NegotiatorDecisionInput = {
|
|
266
|
+
agentData: agent.data,
|
|
267
|
+
agentPrompt: agent.prompt,
|
|
268
|
+
agentContext: agent.context,
|
|
269
|
+
agentTools: agent.agentTools,
|
|
270
|
+
...(agent.gadgets
|
|
271
|
+
? { gadgets: agent.gadgets }
|
|
272
|
+
: {}),
|
|
273
|
+
sessionState,
|
|
274
|
+
blueprintCandidates: ragResult.options.map((opt) => ({
|
|
275
|
+
blueprintId: opt.blueprintId ?? opt.id,
|
|
276
|
+
description: opt.description,
|
|
277
|
+
contract: opt.contract,
|
|
278
|
+
similarity:
|
|
279
|
+
parseFloat(opt.description.match(/similarity: (\d+)%/)?.[1] ?? '0') / 100,
|
|
280
|
+
verdict: (opt.description.includes('exact') ? 'exact' : 'partial') as
|
|
281
|
+
| 'exact'
|
|
282
|
+
| 'partial',
|
|
283
|
+
})),
|
|
284
|
+
};
|
|
285
|
+
|
|
286
|
+
deps.onProgress?.('deciding', 'Choosing the best UI approach...');
|
|
287
|
+
const decisionStart = Date.now();
|
|
288
|
+
const { decision, alternatives } = await makeDecision(decisionInput, deps.llm);
|
|
289
|
+
const decisionLatencyMs = Date.now() - decisionStart;
|
|
290
|
+
const totalMs = Date.now() - negotiateStart;
|
|
291
|
+
|
|
292
|
+
const pickedBlueprint = decision.blueprintId
|
|
293
|
+
? ragResult.options.find(
|
|
294
|
+
(opt) => (opt.blueprintId ?? opt.id) === decision.blueprintId,
|
|
295
|
+
)
|
|
296
|
+
: undefined;
|
|
297
|
+
|
|
298
|
+
// eslint-disable-next-line no-console
|
|
299
|
+
console.log(
|
|
300
|
+
`[negotiate] LLM PATH: action=${decision.action} blueprint=${decision.blueprintId?.slice(-8) ?? 'none'} hash=${pickedBlueprint?.contractHash?.slice(0, 12) ?? 'none'} | LLM decision: ${decisionLatencyMs}ms | total: ${totalMs}ms`,
|
|
301
|
+
);
|
|
302
|
+
|
|
303
|
+
return {
|
|
304
|
+
decision,
|
|
305
|
+
alternatives,
|
|
306
|
+
storedContractHash: pickedBlueprint?.contractHash,
|
|
307
|
+
storedPoolSource: pickedBlueprint?.poolSource,
|
|
308
|
+
embeddingLatencyMs: ragResult.embeddingLatencyMs,
|
|
309
|
+
searchLatencyMs: ragResult.searchLatencyMs,
|
|
310
|
+
decisionLatencyMs,
|
|
311
|
+
};
|
|
312
|
+
}
|