@ggui-ai/negotiator 0.1.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +49 -0
  3. package/dist/contract-hash.d.ts +54 -0
  4. package/dist/contract-hash.d.ts.map +1 -0
  5. package/dist/contract-hash.js +96 -0
  6. package/dist/contract-validators.d.ts +171 -0
  7. package/dist/contract-validators.d.ts.map +1 -0
  8. package/dist/contract-validators.js +478 -0
  9. package/dist/decision-input.d.ts +48 -0
  10. package/dist/decision-input.d.ts.map +1 -0
  11. package/dist/decision-input.js +14 -0
  12. package/dist/decision.d.ts +54 -0
  13. package/dist/decision.d.ts.map +1 -0
  14. package/dist/decision.js +500 -0
  15. package/dist/index.d.ts +36 -0
  16. package/dist/index.d.ts.map +1 -0
  17. package/dist/index.js +25 -0
  18. package/dist/intent.d.ts +22 -0
  19. package/dist/intent.d.ts.map +1 -0
  20. package/dist/intent.js +28 -0
  21. package/dist/llm-caller.d.ts +70 -0
  22. package/dist/llm-caller.d.ts.map +1 -0
  23. package/dist/llm-caller.js +38 -0
  24. package/dist/llm-rerank.d.ts +101 -0
  25. package/dist/llm-rerank.d.ts.map +1 -0
  26. package/dist/llm-rerank.js +178 -0
  27. package/dist/negotiate.d.ts +141 -0
  28. package/dist/negotiate.d.ts.map +1 -0
  29. package/dist/negotiate.js +161 -0
  30. package/dist/normalize-schema.d.ts +22 -0
  31. package/dist/normalize-schema.d.ts.map +1 -0
  32. package/dist/normalize-schema.js +191 -0
  33. package/dist/pure.d.ts +30 -0
  34. package/dist/pure.d.ts.map +1 -0
  35. package/dist/pure.js +43 -0
  36. package/dist/rag-search.d.ts +73 -0
  37. package/dist/rag-search.d.ts.map +1 -0
  38. package/dist/rag-search.js +192 -0
  39. package/dist/rerank-eval/pairs.d.ts +28 -0
  40. package/dist/rerank-eval/pairs.d.ts.map +1 -0
  41. package/dist/rerank-eval/pairs.js +531 -0
  42. package/dist/rerank-eval/run-probe-cli.d.ts +3 -0
  43. package/dist/rerank-eval/run-probe-cli.d.ts.map +1 -0
  44. package/dist/rerank-eval/run-probe-cli.js +146 -0
  45. package/dist/rerank-eval/run-probe.d.ts +68 -0
  46. package/dist/rerank-eval/run-probe.d.ts.map +1 -0
  47. package/dist/rerank-eval/run-probe.js +113 -0
  48. package/dist/session.d.ts +42 -0
  49. package/dist/session.d.ts.map +1 -0
  50. package/dist/session.js +21 -0
  51. package/dist/suggestion.d.ts +38 -0
  52. package/dist/suggestion.d.ts.map +1 -0
  53. package/dist/suggestion.js +47 -0
  54. package/dist/synth-bench/corpus.d.ts +106 -0
  55. package/dist/synth-bench/corpus.d.ts.map +1 -0
  56. package/dist/synth-bench/corpus.js +994 -0
  57. package/dist/synth-bench/run-bench-cli.d.ts +3 -0
  58. package/dist/synth-bench/run-bench-cli.d.ts.map +1 -0
  59. package/dist/synth-bench/run-bench-cli.js +181 -0
  60. package/dist/synth-bench/run-bench.d.ts +101 -0
  61. package/dist/synth-bench/run-bench.d.ts.map +1 -0
  62. package/dist/synth-bench/run-bench.js +374 -0
  63. package/dist/synthesize-contract.d.ts +131 -0
  64. package/dist/synthesize-contract.d.ts.map +1 -0
  65. package/dist/synthesize-contract.js +948 -0
  66. package/dist/types.d.ts +30 -0
  67. package/dist/types.d.ts.map +1 -0
  68. package/dist/types.js +13 -0
  69. package/package.json +74 -0
  70. package/src/contract-hash.ts +102 -0
  71. package/src/contract-validators.ts +604 -0
  72. package/src/decision-input.ts +49 -0
  73. package/src/decision.ts +581 -0
  74. package/src/index.ts +63 -0
  75. package/src/intent.ts +37 -0
  76. package/src/llm-caller.ts +82 -0
  77. package/src/llm-rerank.ts +280 -0
  78. package/src/negotiate.ts +312 -0
  79. package/src/normalize-schema.ts +193 -0
  80. package/src/pure.ts +46 -0
  81. package/src/rag-search.ts +274 -0
  82. package/src/rerank-eval/pairs.ts +624 -0
  83. package/src/rerank-eval/run-probe-cli.ts +197 -0
  84. package/src/rerank-eval/run-probe.ts +198 -0
  85. package/src/session.ts +41 -0
  86. package/src/suggestion.ts +73 -0
  87. package/src/synth-bench/corpus.ts +1126 -0
  88. package/src/synth-bench/run-bench-cli.ts +237 -0
  89. package/src/synth-bench/run-bench.ts +525 -0
  90. package/src/synthesize-contract.ts +1161 -0
  91. package/src/types.ts +31 -0
@@ -0,0 +1,82 @@
1
+ /**
2
+ * `LLMCaller` — the decision engine's LLM dispatcher.
3
+ *
4
+ * Narrow abstraction over "call a chat model, optionally with a forced
5
+ * tool-use schema for guaranteed-JSON structured output." Kept public
6
+ * so OSS consumers of `@ggui-ai/negotiator` can bring their own LLM
7
+ * provider (Anthropic direct, OpenAI, Google, a local model, a
8
+ * community LiteLLM wrapper) without touching the decision-engine
9
+ * source.
10
+ *
11
+ * **Why this lives in `@ggui-ai/negotiator`, not
12
+ * `@ggui-ai/mcp-server-core`.** `mcp-server-core` contains the
13
+ * storage + runtime seams an MCP server implementer binds against
14
+ * (`VectorStore`, `EmbeddingProvider`, `KeyValueStore`,
15
+ * `BlueprintProvider`, `Negotiator`). `LLMCaller` is an
16
+ * engine-internal dispatcher — one level below `Negotiator` — so
17
+ * lifting it to `mcp-server-core` would grow the public seam count
18
+ * speculatively. If a second consumer outside the negotiator
19
+ * surfaces later, the "where does `LLMCaller` live?" question can be
20
+ * re-opened at that point.
21
+ *
22
+ * Normative semantics:
23
+ * - `call(systemPrompt, userMessage, maxTokens?)` returns the raw
24
+ * model text. Implementations MUST NOT inject tool-use blocks when
25
+ * the caller didn't request them — the text path is used as a
26
+ * regex-JSON fallback.
27
+ * - `callStructured?<T>(...)` is OPTIONAL. When present, it MUST
28
+ * force tool use against the supplied `ToolSchema` and return the
29
+ * tool input, parsed as `T`. Implementations that don't support
30
+ * forced structured output simply omit this method; consumers
31
+ * fall back to `call` + regex JSON extraction. Absence is not an
32
+ * error.
33
+ * - `ToolSchema.input_schema` follows the OpenAI tool-use JSON
34
+ * Schema convention. Implementations that use a different
35
+ * tool-use protocol (e.g., Anthropic's variant) MUST translate at
36
+ * the adapter boundary.
37
+ */
38
+
39
+ /** Tool schema for structured output via forced tool use. */
40
+ export interface ToolSchema {
41
+ name: string;
42
+ description: string;
43
+ input_schema: Record<string, unknown>;
44
+ }
45
+
46
+ /** Chat-model dispatcher consumed by the negotiator decision engine. */
47
+ export interface LLMCaller {
48
+ /**
49
+ * Call the model in plain-text mode. `maxTokens` defaults to
50
+ * something implementation-appropriate (usually 2048).
51
+ */
52
+ call(
53
+ systemPrompt: string,
54
+ userMessage: string,
55
+ maxTokens?: number,
56
+ ): Promise<string>;
57
+
58
+ /**
59
+ * Call with forced tool use for guaranteed structured JSON output.
60
+ * Implementations that can't force tool use should omit this
61
+ * method — consumers detect absence and fall back to regex JSON
62
+ * extraction on the text path.
63
+ */
64
+ callStructured?<T>(
65
+ systemPrompt: string,
66
+ userMessage: string,
67
+ tool: ToolSchema,
68
+ maxTokens?: number,
69
+ ): Promise<T>;
70
+ }
71
+
72
+ /**
73
+ * Provider + model selector for factory-style LLM caller
74
+ * construction. The `provider` enum stays narrow to the ones
75
+ * ggui supports today; community adapters can extend by widening
76
+ * the union at their own boundary.
77
+ */
78
+ export interface LLMCallerConfig {
79
+ provider: 'anthropic' | 'openai' | 'google' | 'openrouter' | 'bedrock';
80
+ model: string;
81
+ apiKey?: string;
82
+ }
@@ -0,0 +1,280 @@
1
+ /**
2
+ * LLM rerank — Tier-2 precision oracle for the blueprint registry.
3
+ *
4
+ * Given a user's UI request (intent + contract structure) and a set
5
+ * of candidate cached blueprints retrieved by RAG, ask a fast LLM
6
+ * (Haiku 4.5) which candidate (if any) matches. Returns a structured
7
+ * decision so the caller can branch deterministically.
8
+ *
9
+ * This module is the precision half of the blueprint-first
10
+ * architecture: RAG retrieval is high-recall but low-precision (bge-
11
+ * small confuses topic-similar but UI-divergent prompts); the LLM
12
+ * judge restores precision. Combined break-even hit rate is ~10%;
13
+ * realistic workloads observe 30-70%.
14
+ */
15
+ import { summarizeContract } from '@ggui-ai/protocol';
16
+ import type { LLMCaller, ToolSchema } from './llm-caller.js';
17
+
18
+ /**
19
+ * One candidate blueprint for the LLM judge to consider.
20
+ *
21
+ * Keep this struct narrow — the prompt sees only what's necessary
22
+ * to decide match-vs-no-match. componentCode is intentionally
23
+ * absent (huge, distracting, doesn't change the decision).
24
+ */
25
+ export interface RerankCandidate {
26
+ /** Stable blueprint id — echoed back as `matchId` on a hit. */
27
+ readonly id: string;
28
+ /** The intent prose that originally produced this blueprint. */
29
+ readonly cachedIntent: string;
30
+ /**
31
+ * One-line summary of the blueprint's contract surface. Format
32
+ * matches `summarizeContract()` below — `slots=...; actions=...;
33
+ * streams=...; props=...` so the judge sees the structural shape
34
+ * without the JSON noise.
35
+ */
36
+ readonly cachedContractSummary: string;
37
+ /**
38
+ * Optional retrieval signal. Higher cosine biases the prior, but
39
+ * the judge's decision is the source of truth. Pass-through so
40
+ * the prompt can include "candidate retrieved at cosine X" if the
41
+ * judge would benefit (default: omit).
42
+ */
43
+ readonly cosine?: number;
44
+ }
45
+
46
+ /** Decision returned by the LLM judge. */
47
+ export interface RerankDecision {
48
+ /**
49
+ * The matched candidate's id, or `null` if no candidate matches.
50
+ * `null` means "all candidates rejected — generate fresh."
51
+ */
52
+ readonly matchId: string | null;
53
+ /**
54
+ * Confidence on `[0, 1]`. Caller compares against a threshold (e.g.
55
+ * 0.6) before treating the decision as a hit. Returned even when
56
+ * matchId is null so callers can log "judge declined with
57
+ * confidence X."
58
+ */
59
+ readonly confidence: number;
60
+ /**
61
+ * Free-text reason from the judge. Surface in trace logs so
62
+ * operators can debug "why didn't this hit." Truncate at the
63
+ * persistence boundary if cardinality is a concern.
64
+ */
65
+ readonly reason: string;
66
+ /** Wall-clock latency of the LLM call. */
67
+ readonly latencyMs: number;
68
+ /**
69
+ * Token cost of the call — for the cache-trace sink and cost
70
+ * accounting. Implementations that can't surface token counts may
71
+ * report `{input: 0, output: 0}` and the cost-per-call gate will
72
+ * have to be measured externally.
73
+ */
74
+ readonly tokenCost: { readonly input: number; readonly output: number };
75
+ }
76
+
77
+ /** Query the user's request the judge is matching against. */
78
+ export interface RerankQuery {
79
+ readonly intent: string;
80
+ readonly contractSummary: string;
81
+ }
82
+
83
+ const RERANK_SYSTEM_PROMPT = `You match user UI requests against previously-generated UI blueprints. Each blueprint was produced for a past request and stored. Decide whether any candidate produces the SAME USEFUL UI for the current request.
84
+
85
+ MATCH means the candidate would correctly satisfy the user's current request — same UI shape (component types, layout pattern), same wire surface (slot names, action names), same intended user task, and same load-bearing parameters (dates, months, ranges, enum values).
86
+
87
+ NO-MATCH means the candidate would NOT satisfy the user's current request — different task (haiku composer vs tweet draft, login vs signup), different UI shape (form vs list vs dashboard), or load-bearing parameters differ (calendar-Jan vs calendar-Mar — same contract, different value).
88
+
89
+ Visual style differences alone (minimal vs ornate, dense vs spacious) DO NOT block a match — the user can refine those after they get a working UI.
90
+
91
+ Output exactly ONE tool call with your decision. confidence is a number in [0, 1]. matchId is a string from the candidate ids, or null when no candidate matches. reason is a short sentence the operator can use to debug.`;
92
+
93
+ const RERANK_TOOL: ToolSchema = {
94
+ name: 'submit_rerank_decision',
95
+ description:
96
+ 'Submit your match-vs-no-match decision over the candidates.',
97
+ input_schema: {
98
+ type: 'object',
99
+ additionalProperties: false,
100
+ properties: {
101
+ matchId: {
102
+ type: ['string', 'null'],
103
+ description:
104
+ "ID of the matching candidate, or null when no candidate matches. MUST be one of the candidate ids supplied in the user message, or null.",
105
+ },
106
+ confidence: {
107
+ type: 'number',
108
+ minimum: 0,
109
+ maximum: 1,
110
+ description:
111
+ 'Confidence in the decision on [0, 1]. Caller will compare against a threshold before treating it as a hit.',
112
+ },
113
+ reason: {
114
+ type: 'string',
115
+ description:
116
+ 'Brief explanation — one sentence — that the operator can use to debug match decisions.',
117
+ },
118
+ },
119
+ required: ['matchId', 'confidence', 'reason'],
120
+ },
121
+ };
122
+
123
+ function buildUserMessage(
124
+ query: RerankQuery,
125
+ candidates: readonly RerankCandidate[],
126
+ ): string {
127
+ const lines: string[] = [];
128
+ lines.push('CURRENT REQUEST');
129
+ lines.push(` intent: ${query.intent}`);
130
+ lines.push(` contract: ${query.contractSummary}`);
131
+ lines.push('');
132
+ lines.push(`CANDIDATES (${candidates.length})`);
133
+ for (const c of candidates) {
134
+ lines.push('---');
135
+ lines.push(` id: ${c.id}`);
136
+ lines.push(` intent: ${truncate(c.cachedIntent, 280)}`);
137
+ lines.push(` contract: ${c.cachedContractSummary}`);
138
+ if (typeof c.cosine === 'number') {
139
+ lines.push(` cosine: ${c.cosine.toFixed(3)}`);
140
+ }
141
+ }
142
+ lines.push('');
143
+ lines.push(
144
+ 'Decide: does any candidate match the current request? Submit via the tool.',
145
+ );
146
+ return lines.join('\n');
147
+ }
148
+
149
+ function truncate(text: string, max: number): string {
150
+ if (text.length <= max) return text;
151
+ return `${text.slice(0, max - 1)}…`;
152
+ }
153
+
154
+ // Re-export `summarizeContract` from protocol for backwards-compatible
155
+ // access through `@ggui-ai/negotiator/llm-rerank` consumers (the
156
+ // canonical home is `@ggui-ai/protocol` — both the registry storage
157
+ // and the rerank prompt depend on it). Kept as a re-export so the
158
+ // existing test imports here keep working.
159
+ export { summarizeContract };
160
+
161
+ interface RerankToolInput {
162
+ matchId: string | null;
163
+ confidence: number;
164
+ reason: string;
165
+ }
166
+
167
+ function clampConfidence(value: unknown): number {
168
+ if (typeof value !== 'number' || !Number.isFinite(value)) return 0;
169
+ if (value < 0) return 0;
170
+ if (value > 1) return 1;
171
+ return value;
172
+ }
173
+
174
+ function parseToolInput(
175
+ raw: unknown,
176
+ candidateIds: ReadonlySet<string>,
177
+ ): { matchId: string | null; confidence: number; reason: string } {
178
+ if (raw === null || typeof raw !== 'object') {
179
+ return { matchId: null, confidence: 0, reason: 'parse-failed: non-object tool input' };
180
+ }
181
+ const obj = raw as Record<string, unknown>;
182
+ const reason = typeof obj['reason'] === 'string' ? obj['reason'] : '';
183
+ const confidence = clampConfidence(obj['confidence']);
184
+ const rawMatchId = obj['matchId'];
185
+ if (rawMatchId === null || rawMatchId === undefined) {
186
+ return { matchId: null, confidence, reason };
187
+ }
188
+ if (typeof rawMatchId !== 'string') {
189
+ return {
190
+ matchId: null,
191
+ confidence: 0,
192
+ reason: `parse-failed: matchId was ${typeof rawMatchId}, not string|null`,
193
+ };
194
+ }
195
+ if (!candidateIds.has(rawMatchId)) {
196
+ return {
197
+ matchId: null,
198
+ confidence: 0,
199
+ reason: `judge returned matchId='${rawMatchId}' but that id is not in the candidate set — treating as no-match`,
200
+ };
201
+ }
202
+ return { matchId: rawMatchId, confidence, reason };
203
+ }
204
+
205
+ /**
206
+ * Run the rerank judge against a query + candidate list.
207
+ *
208
+ * Empty candidate list short-circuits without an LLM call — the
209
+ * caller's RAG step decided there's nothing close enough; we don't
210
+ * burn tokens to confirm it.
211
+ *
212
+ * Operational errors (LLM throws, parse fails) collapse to a `null`
213
+ * match with confidence=0 and a diagnostic reason. The caller treats
214
+ * confidence-below-threshold as no-match anyway, so the failure mode
215
+ * lands on the cold-gen path automatically.
216
+ */
217
+ export async function rerankCandidates(
218
+ deps: { readonly llm: LLMCaller },
219
+ query: RerankQuery,
220
+ candidates: readonly RerankCandidate[],
221
+ ): Promise<RerankDecision> {
222
+ const startedAt = Date.now();
223
+ if (candidates.length === 0) {
224
+ return {
225
+ matchId: null,
226
+ confidence: 0,
227
+ reason: 'no candidates — short-circuited without LLM call',
228
+ latencyMs: Date.now() - startedAt,
229
+ tokenCost: { input: 0, output: 0 },
230
+ };
231
+ }
232
+
233
+ const userMessage = buildUserMessage(query, candidates);
234
+ const candidateIds = new Set(candidates.map((c) => c.id));
235
+
236
+ if (typeof deps.llm.callStructured !== 'function') {
237
+ return {
238
+ matchId: null,
239
+ confidence: 0,
240
+ reason:
241
+ 'llm-rerank: provider does not support callStructured (forced tool-use). Bind a structured-capable LLMCaller (Anthropic adapter via mcp-server) for rerank.',
242
+ latencyMs: Date.now() - startedAt,
243
+ tokenCost: { input: 0, output: 0 },
244
+ };
245
+ }
246
+
247
+ let toolInput: unknown;
248
+ try {
249
+ toolInput = await deps.llm.callStructured<RerankToolInput>(
250
+ RERANK_SYSTEM_PROMPT,
251
+ userMessage,
252
+ RERANK_TOOL,
253
+ 512,
254
+ );
255
+ } catch (err) {
256
+ const message = err instanceof Error ? err.message : String(err);
257
+ return {
258
+ matchId: null,
259
+ confidence: 0,
260
+ reason: `llm-rerank: callStructured threw — ${message}`,
261
+ latencyMs: Date.now() - startedAt,
262
+ tokenCost: { input: 0, output: 0 },
263
+ };
264
+ }
265
+
266
+ const parsed = parseToolInput(toolInput, candidateIds);
267
+ return {
268
+ matchId: parsed.matchId,
269
+ confidence: parsed.confidence,
270
+ reason: parsed.reason,
271
+ latencyMs: Date.now() - startedAt,
272
+ // Token cost surfacing requires LLMCaller-level instrumentation
273
+ // we don't have today. Default to zero; the cost gate is measured
274
+ // out-of-band from billing data during the probe.
275
+ tokenCost: { input: 0, output: 0 },
276
+ };
277
+ }
278
+
279
+ // Re-exports for the eval harness — keep public surface explicit.
280
+ export { RERANK_SYSTEM_PROMPT, RERANK_TOOL };
@@ -0,0 +1,312 @@
1
+ /**
2
+ * Negotiator — top-level orchestrator over the public storage +
3
+ * decision-engine seams.
4
+ *
5
+ * Pipeline:
6
+ * 1. RAG search (per-scope + optional shared pool) via
7
+ * {@link ragSearch} — composes `EmbeddingProvider.embed` +
8
+ * `VectorStore.query` from `@ggui-ai/mcp-server-core`.
9
+ * 2. Read session state (optional injectable).
10
+ * 3. Fast-path for exact blueprint hits — skip the decision LLM.
11
+ * 4. Otherwise call {@link makeDecision} with the RAG candidates and
12
+ * session stack; fold the picked blueprint's pool provenance into
13
+ * the return value.
14
+ *
15
+ * Timing logs (stable format — consumed by benchmarks):
16
+ * [negotiate] embedding: Xms | search: Xms | candidates: N
17
+ * [negotiate] candidate: <id8> | <description> | hash=<h16>
18
+ * [negotiate] FAST PATH: blueprint=<id8> hash=<h12> | LLM decision: 0ms | total: Xms
19
+ * [negotiate] LLM PATH: action=X blueprint=<id8> hash=<h12> | LLM decision: Xms | total: Xms
20
+ *
21
+ * ### Public surface + semver weight
22
+ *
23
+ * Exported:
24
+ * - `negotiate(deps, input)` — runtime orchestrator.
25
+ * - `NegotiateDeps` — injection shape (embedding / vectors / llm +
26
+ * optional session-state reader + optional progress callback).
27
+ * - `NegotiateInput` — agent signal + config.
28
+ * - `NegotiateConfig` — minimum fields the orchestrator actually
29
+ * reads. Pool selection is expressed as `includeSharedPool:
30
+ * boolean` rather than a `poolMode` enum, so callers decide
31
+ * shared-pool inclusion explicitly at each call site.
32
+ * - `NegotiateResult` — the decision result returned to callers.
33
+ *
34
+ * ### Why this package, not `mcp-server-core`
35
+ *
36
+ * `mcp-server-core` locks storage/runtime seams MCP server
37
+ * implementers bind against (`EmbeddingProvider`, `VectorStore`,
38
+ * `Negotiator`). `negotiate()` is a *composition* over those seams —
39
+ * decision-engine semantics, not a new seam. Adding it to
40
+ * `mcp-server-core` would drag the LLM prompts + tool-schemas into a
41
+ * package whose job is to stay minimal and runtime-agnostic.
42
+ */
43
+
44
+ import type {
45
+ DataContract,
46
+ NegotiatorAlternative,
47
+ NegotiatorDecision,
48
+ } from '@ggui-ai/protocol';
49
+ import type {
50
+ EmbeddingProvider,
51
+ VectorStore,
52
+ } from '@ggui-ai/mcp-server-core';
53
+ import type { LLMCaller } from './llm-caller.js';
54
+ import type { SessionState } from './session.js';
55
+ import type { NegotiatorDecisionInput } from './decision-input.js';
56
+ import { ragSearch } from './rag-search.js';
57
+ import { makeDecision } from './decision.js';
58
+
59
+ /** Empty session state for cold starts and benchmarks. */
60
+ const EMPTY_SESSION: SessionState = {
61
+ stack: [],
62
+ conversationHistory: [],
63
+ };
64
+
65
+ /**
66
+ * Minimum config the orchestrator reads.
67
+ *
68
+ * `appId` is the primary RAG scope (per-app registered UIs live
69
+ * here). `sessionId` keys the optional `readSessionState` callback.
70
+ * `includeSharedPool` (default `false`) gates whether to also search
71
+ * the global `"shared"` pool in parallel and fold its hits into the
72
+ * candidate set.
73
+ */
74
+ export interface NegotiateConfig {
75
+ appId: string;
76
+ sessionId: string;
77
+ /** Also search the shared (global) pool in parallel. Default `false`. */
78
+ includeSharedPool?: boolean;
79
+ }
80
+
81
+ /**
82
+ * Agent signal + config for a single negotiation call. Bundled as one
83
+ * object so future additive fields don't force yet another positional
84
+ * arg on the public API.
85
+ */
86
+ export interface NegotiateInput {
87
+ agent: {
88
+ /** Raw data the agent wants to render. */
89
+ data?: Record<string, unknown>;
90
+ /** Natural-language prompt. Used as RAG query when present. */
91
+ prompt?: string;
92
+ /** Free-form context string or structured map. Forwarded to the decision LLM. */
93
+ context?: string | Record<string, unknown>;
94
+ /**
95
+ * Names of MCP tools the agent invokes (catalog seed). Merged
96
+ * into the decision's `agentCapabilities.tools` deterministically
97
+ * (see {@link makeDecision}).
98
+ */
99
+ agentTools?: string[];
100
+ /**
101
+ * Browser-capability gadget catalog the app exposes (default
102
+ * `STDLIB_GADGETS`, operator-extensible). Forwarded to
103
+ * the decision LLM so it knows which gadget bindings are
104
+ * available; canonical entries enrich partial LLM output
105
+ * downstream (see {@link mergeGadgets}).
106
+ */
107
+ gadgets?: readonly import('@ggui-ai/protocol').GadgetDescriptor[];
108
+ };
109
+ config: NegotiateConfig;
110
+ }
111
+
112
+ /**
113
+ * Dependencies injected into {@link negotiate}. Public field names
114
+ * (`embedding` / `vectors` / `llm`) align with {@link ragSearch}'s
115
+ * already-shipped deps shape.
116
+ */
117
+ export interface NegotiateDeps {
118
+ /**
119
+ * Embedding provider for the RAG search step. **Optional** —
120
+ * when omitted (paired with omitted `vectors`), the negotiator
121
+ * skips RAG entirely and runs the decision LLM against an empty
122
+ * candidate list. OSS without vector-store infrastructure binds
123
+ * with `embedding: undefined, vectors: undefined` and still gets
124
+ * useful negotiation via the decision LLM alone.
125
+ *
126
+ * `embedding` and `vectors` are paired — both must be present for
127
+ * RAG to fire, or both must be absent. A half-bound config (one
128
+ * present, one missing) is a configuration bug; the search step
129
+ * skips and logs a warn rather than erroring, but consumers
130
+ * should fix the binding.
131
+ */
132
+ embedding?: EmbeddingProvider;
133
+ vectors?: VectorStore;
134
+ llm: LLMCaller;
135
+ /** Optional: read current session state for stack-aware decisions. */
136
+ readSessionState?: (sessionId: string) => Promise<SessionState | null>;
137
+ /** Optional: surface pipeline progress to consumers. */
138
+ onProgress?: (phase: string, summary: string) => void;
139
+ }
140
+
141
+ /**
142
+ * Output shape — mirrors the legacy `DecisionResult` so the
143
+ * back-compat shim can return the object verbatim. See `NegotiateDeps`
144
+ * semver note.
145
+ */
146
+ export interface NegotiateResult {
147
+ decision: NegotiatorDecision;
148
+ alternatives: NegotiatorAlternative[];
149
+ /** Stored contract hash from blueprint match — deterministic pool key. */
150
+ storedContractHash?: string;
151
+ /** Which pool the matched blueprint's code lives in. */
152
+ storedPoolSource?: 'shared' | 'private';
153
+ embeddingLatencyMs: number;
154
+ searchLatencyMs: number;
155
+ decisionLatencyMs: number;
156
+ }
157
+
158
+ /**
159
+ * Orchestrate one negotiation call — RAG search, session read, fast
160
+ * path, decision LLM. See module docstring for the pipeline outline.
161
+ */
162
+ export async function negotiate(
163
+ deps: NegotiateDeps,
164
+ input: NegotiateInput,
165
+ ): Promise<NegotiateResult> {
166
+ const { agent, config } = input;
167
+ const negotiateStart = Date.now();
168
+ deps.onProgress?.('negotiating', 'Analyzing request...');
169
+
170
+ // Step 1: RAG search (per-app + optional shared pool in parallel).
171
+ // `embedding` + `vectors` are optional. When either is missing,
172
+ // ragSearch is a no-op (returns empty options + 0 latency); the
173
+ // decision LLM runs against zero candidates and falls back to
174
+ // "create" via its standard branching. OSS without RAG infrastructure
175
+ // gets useful negotiation from the decision LLM alone.
176
+ const queryText = agent.prompt ?? JSON.stringify(agent.data ?? {});
177
+ const emptyResult = { options: [], embeddingLatencyMs: 0, searchLatencyMs: 0 };
178
+ const ragDeps =
179
+ deps.embedding !== undefined && deps.vectors !== undefined
180
+ ? { embedding: deps.embedding, vectors: deps.vectors }
181
+ : undefined;
182
+ const [appResult, sharedResult] = await Promise.all([
183
+ ragDeps
184
+ ? ragSearch(ragDeps, { prompt: queryText, scope: config.appId })
185
+ : Promise.resolve(emptyResult),
186
+ ragDeps && config.includeSharedPool
187
+ ? ragSearch(ragDeps, { prompt: queryText, scope: 'shared' })
188
+ : Promise.resolve(emptyResult),
189
+ ]);
190
+
191
+ const ragResult = {
192
+ options: [...appResult.options, ...sharedResult.options],
193
+ embeddingLatencyMs: Math.max(
194
+ appResult.embeddingLatencyMs,
195
+ sharedResult.embeddingLatencyMs,
196
+ ),
197
+ searchLatencyMs: Math.max(
198
+ appResult.searchLatencyMs,
199
+ sharedResult.searchLatencyMs,
200
+ ),
201
+ };
202
+
203
+ // eslint-disable-next-line no-console
204
+ console.log(
205
+ `[negotiate] embedding: ${ragResult.embeddingLatencyMs}ms | search: ${ragResult.searchLatencyMs}ms | candidates: ${ragResult.options.length}`,
206
+ );
207
+ for (const opt of ragResult.options) {
208
+ // eslint-disable-next-line no-console
209
+ console.log(
210
+ `[negotiate] candidate: ${opt.blueprintId?.slice(-8) ?? 'none'} | ${opt.description.slice(0, 80)} | hash=${opt.contractHash?.slice(0, 16) ?? 'none'}`,
211
+ );
212
+ }
213
+
214
+ deps.onProgress?.(
215
+ 'blueprint_search',
216
+ ragResult.options.length > 0
217
+ ? `Found ${ragResult.options.length} blueprint candidate${ragResult.options.length > 1 ? 's' : ''}`
218
+ : 'No blueprints found',
219
+ );
220
+
221
+ // Step 2: Read session state (falls back to EMPTY_SESSION).
222
+ const sessionState = deps.readSessionState
223
+ ? ((await deps.readSessionState(config.sessionId)) ?? EMPTY_SESSION)
224
+ : EMPTY_SESSION;
225
+
226
+ // Step 3: Fast-path for high-confidence exact matches — skip decision LLM.
227
+ const exactOpt = ragResult.options.find((opt) =>
228
+ opt.description.includes('exact'),
229
+ );
230
+ if (exactOpt) {
231
+ const stackHasSameType = sessionState.stack.some(
232
+ (item) => item.prompt && agent.prompt && item.prompt === agent.prompt,
233
+ );
234
+ const action = stackHasSameType ? ('update' as const) : ('create' as const);
235
+ const totalMs = Date.now() - negotiateStart;
236
+
237
+ // eslint-disable-next-line no-console
238
+ console.log(
239
+ `[negotiate] FAST PATH: blueprint=${exactOpt.blueprintId?.slice(-8)} hash=${exactOpt.contractHash?.slice(0, 12) ?? 'none'} | LLM decision: 0ms | total: ${totalMs}ms`,
240
+ );
241
+ deps.onProgress?.('deciding', 'Exact match found — fast path');
242
+
243
+ // The agent's prompt is the outer-pipeline intent (`intent` is not
244
+ // a contract field). Fallback contract is the empty contract — the
245
+ // four-spec surface is omitted entirely (no props, no actions, no
246
+ // streams, no context).
247
+ const fallbackContract: DataContract = {};
248
+ return {
249
+ decision: {
250
+ action,
251
+ reasoning: `Exact blueprint match (high confidence). ${action === 'update' ? 'Updating existing view.' : 'Creating new view.'}`,
252
+ blueprintId: exactOpt.blueprintId,
253
+ contract: exactOpt.contract ?? fallbackContract,
254
+ },
255
+ alternatives: [],
256
+ storedContractHash: exactOpt.contractHash,
257
+ storedPoolSource: exactOpt.poolSource,
258
+ embeddingLatencyMs: ragResult.embeddingLatencyMs,
259
+ searchLatencyMs: ragResult.searchLatencyMs,
260
+ decisionLatencyMs: 0,
261
+ };
262
+ }
263
+
264
+ // Step 4: Build decision input + call the LLM.
265
+ const decisionInput: NegotiatorDecisionInput = {
266
+ agentData: agent.data,
267
+ agentPrompt: agent.prompt,
268
+ agentContext: agent.context,
269
+ agentTools: agent.agentTools,
270
+ ...(agent.gadgets
271
+ ? { gadgets: agent.gadgets }
272
+ : {}),
273
+ sessionState,
274
+ blueprintCandidates: ragResult.options.map((opt) => ({
275
+ blueprintId: opt.blueprintId ?? opt.id,
276
+ description: opt.description,
277
+ contract: opt.contract,
278
+ similarity:
279
+ parseFloat(opt.description.match(/similarity: (\d+)%/)?.[1] ?? '0') / 100,
280
+ verdict: (opt.description.includes('exact') ? 'exact' : 'partial') as
281
+ | 'exact'
282
+ | 'partial',
283
+ })),
284
+ };
285
+
286
+ deps.onProgress?.('deciding', 'Choosing the best UI approach...');
287
+ const decisionStart = Date.now();
288
+ const { decision, alternatives } = await makeDecision(decisionInput, deps.llm);
289
+ const decisionLatencyMs = Date.now() - decisionStart;
290
+ const totalMs = Date.now() - negotiateStart;
291
+
292
+ const pickedBlueprint = decision.blueprintId
293
+ ? ragResult.options.find(
294
+ (opt) => (opt.blueprintId ?? opt.id) === decision.blueprintId,
295
+ )
296
+ : undefined;
297
+
298
+ // eslint-disable-next-line no-console
299
+ console.log(
300
+ `[negotiate] LLM PATH: action=${decision.action} blueprint=${decision.blueprintId?.slice(-8) ?? 'none'} hash=${pickedBlueprint?.contractHash?.slice(0, 12) ?? 'none'} | LLM decision: ${decisionLatencyMs}ms | total: ${totalMs}ms`,
301
+ );
302
+
303
+ return {
304
+ decision,
305
+ alternatives,
306
+ storedContractHash: pickedBlueprint?.contractHash,
307
+ storedPoolSource: pickedBlueprint?.poolSource,
308
+ embeddingLatencyMs: ragResult.embeddingLatencyMs,
309
+ searchLatencyMs: ragResult.searchLatencyMs,
310
+ decisionLatencyMs,
311
+ };
312
+ }