@ggui-ai/negotiator 0.1.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +49 -0
  3. package/dist/contract-hash.d.ts +54 -0
  4. package/dist/contract-hash.d.ts.map +1 -0
  5. package/dist/contract-hash.js +96 -0
  6. package/dist/contract-validators.d.ts +171 -0
  7. package/dist/contract-validators.d.ts.map +1 -0
  8. package/dist/contract-validators.js +478 -0
  9. package/dist/decision-input.d.ts +48 -0
  10. package/dist/decision-input.d.ts.map +1 -0
  11. package/dist/decision-input.js +14 -0
  12. package/dist/decision.d.ts +54 -0
  13. package/dist/decision.d.ts.map +1 -0
  14. package/dist/decision.js +500 -0
  15. package/dist/index.d.ts +36 -0
  16. package/dist/index.d.ts.map +1 -0
  17. package/dist/index.js +25 -0
  18. package/dist/intent.d.ts +22 -0
  19. package/dist/intent.d.ts.map +1 -0
  20. package/dist/intent.js +28 -0
  21. package/dist/llm-caller.d.ts +70 -0
  22. package/dist/llm-caller.d.ts.map +1 -0
  23. package/dist/llm-caller.js +38 -0
  24. package/dist/llm-rerank.d.ts +101 -0
  25. package/dist/llm-rerank.d.ts.map +1 -0
  26. package/dist/llm-rerank.js +178 -0
  27. package/dist/negotiate.d.ts +141 -0
  28. package/dist/negotiate.d.ts.map +1 -0
  29. package/dist/negotiate.js +161 -0
  30. package/dist/normalize-schema.d.ts +22 -0
  31. package/dist/normalize-schema.d.ts.map +1 -0
  32. package/dist/normalize-schema.js +191 -0
  33. package/dist/pure.d.ts +30 -0
  34. package/dist/pure.d.ts.map +1 -0
  35. package/dist/pure.js +43 -0
  36. package/dist/rag-search.d.ts +73 -0
  37. package/dist/rag-search.d.ts.map +1 -0
  38. package/dist/rag-search.js +192 -0
  39. package/dist/rerank-eval/pairs.d.ts +28 -0
  40. package/dist/rerank-eval/pairs.d.ts.map +1 -0
  41. package/dist/rerank-eval/pairs.js +531 -0
  42. package/dist/rerank-eval/run-probe-cli.d.ts +3 -0
  43. package/dist/rerank-eval/run-probe-cli.d.ts.map +1 -0
  44. package/dist/rerank-eval/run-probe-cli.js +146 -0
  45. package/dist/rerank-eval/run-probe.d.ts +68 -0
  46. package/dist/rerank-eval/run-probe.d.ts.map +1 -0
  47. package/dist/rerank-eval/run-probe.js +113 -0
  48. package/dist/session.d.ts +42 -0
  49. package/dist/session.d.ts.map +1 -0
  50. package/dist/session.js +21 -0
  51. package/dist/suggestion.d.ts +38 -0
  52. package/dist/suggestion.d.ts.map +1 -0
  53. package/dist/suggestion.js +47 -0
  54. package/dist/synth-bench/corpus.d.ts +106 -0
  55. package/dist/synth-bench/corpus.d.ts.map +1 -0
  56. package/dist/synth-bench/corpus.js +994 -0
  57. package/dist/synth-bench/run-bench-cli.d.ts +3 -0
  58. package/dist/synth-bench/run-bench-cli.d.ts.map +1 -0
  59. package/dist/synth-bench/run-bench-cli.js +181 -0
  60. package/dist/synth-bench/run-bench.d.ts +101 -0
  61. package/dist/synth-bench/run-bench.d.ts.map +1 -0
  62. package/dist/synth-bench/run-bench.js +374 -0
  63. package/dist/synthesize-contract.d.ts +131 -0
  64. package/dist/synthesize-contract.d.ts.map +1 -0
  65. package/dist/synthesize-contract.js +948 -0
  66. package/dist/types.d.ts +30 -0
  67. package/dist/types.d.ts.map +1 -0
  68. package/dist/types.js +13 -0
  69. package/package.json +74 -0
  70. package/src/contract-hash.ts +102 -0
  71. package/src/contract-validators.ts +604 -0
  72. package/src/decision-input.ts +49 -0
  73. package/src/decision.ts +581 -0
  74. package/src/index.ts +63 -0
  75. package/src/intent.ts +37 -0
  76. package/src/llm-caller.ts +82 -0
  77. package/src/llm-rerank.ts +280 -0
  78. package/src/negotiate.ts +312 -0
  79. package/src/normalize-schema.ts +193 -0
  80. package/src/pure.ts +46 -0
  81. package/src/rag-search.ts +274 -0
  82. package/src/rerank-eval/pairs.ts +624 -0
  83. package/src/rerank-eval/run-probe-cli.ts +197 -0
  84. package/src/rerank-eval/run-probe.ts +198 -0
  85. package/src/session.ts +41 -0
  86. package/src/suggestion.ts +73 -0
  87. package/src/synth-bench/corpus.ts +1126 -0
  88. package/src/synth-bench/run-bench-cli.ts +237 -0
  89. package/src/synth-bench/run-bench.ts +525 -0
  90. package/src/synthesize-contract.ts +1161 -0
  91. package/src/types.ts +31 -0
@@ -0,0 +1,70 @@
1
+ /**
2
+ * `LLMCaller` — the decision engine's LLM dispatcher.
3
+ *
4
+ * Narrow abstraction over "call a chat model, optionally with a forced
5
+ * tool-use schema for guaranteed-JSON structured output." Kept public
6
+ * so OSS consumers of `@ggui-ai/negotiator` can bring their own LLM
7
+ * provider (Anthropic direct, OpenAI, Google, a local model, a
8
+ * community LiteLLM wrapper) without touching the decision-engine
9
+ * source.
10
+ *
11
+ * **Why this lives in `@ggui-ai/negotiator`, not
12
+ * `@ggui-ai/mcp-server-core`.** `mcp-server-core` contains the
13
+ * storage + runtime seams an MCP server implementer binds against
14
+ * (`VectorStore`, `EmbeddingProvider`, `KeyValueStore`,
15
+ * `BlueprintProvider`, `Negotiator`). `LLMCaller` is an
16
+ * engine-internal dispatcher — one level below `Negotiator` — so
17
+ * lifting it to `mcp-server-core` would grow the public seam count
18
+ * speculatively. If a second consumer outside the negotiator
19
+ * surfaces later, the "where does `LLMCaller` live?" question can be
20
+ * re-opened at that point.
21
+ *
22
+ * Normative semantics:
23
+ * - `call(systemPrompt, userMessage, maxTokens?)` returns the raw
24
+ * model text. Implementations MUST NOT inject tool-use blocks when
25
+ * the caller didn't request them — the text path is used as a
26
+ * regex-JSON fallback.
27
+ * - `callStructured?<T>(...)` is OPTIONAL. When present, it MUST
28
+ * force tool use against the supplied `ToolSchema` and return the
29
+ * tool input, parsed as `T`. Implementations that don't support
30
+ * forced structured output simply omit this method; consumers
31
+ * fall back to `call` + regex JSON extraction. Absence is not an
32
+ * error.
33
+ * - `ToolSchema.input_schema` follows the OpenAI tool-use JSON
34
+ * Schema convention. Implementations that use a different
35
+ * tool-use protocol (e.g., Anthropic's variant) MUST translate at
36
+ * the adapter boundary.
37
+ */
38
+ /** Tool schema for structured output via forced tool use. */
39
+ export interface ToolSchema {
40
+ name: string;
41
+ description: string;
42
+ input_schema: Record<string, unknown>;
43
+ }
44
+ /** Chat-model dispatcher consumed by the negotiator decision engine. */
45
+ export interface LLMCaller {
46
+ /**
47
+ * Call the model in plain-text mode. `maxTokens` defaults to
48
+ * something implementation-appropriate (usually 2048).
49
+ */
50
+ call(systemPrompt: string, userMessage: string, maxTokens?: number): Promise<string>;
51
+ /**
52
+ * Call with forced tool use for guaranteed structured JSON output.
53
+ * Implementations that can't force tool use should omit this
54
+ * method — consumers detect absence and fall back to regex JSON
55
+ * extraction on the text path.
56
+ */
57
+ callStructured?<T>(systemPrompt: string, userMessage: string, tool: ToolSchema, maxTokens?: number): Promise<T>;
58
+ }
59
+ /**
60
+ * Provider + model selector for factory-style LLM caller
61
+ * construction. The `provider` enum stays narrow to the ones
62
+ * ggui supports today; community adapters can extend by widening
63
+ * the union at their own boundary.
64
+ */
65
+ export interface LLMCallerConfig {
66
+ provider: 'anthropic' | 'openai' | 'google' | 'openrouter' | 'bedrock';
67
+ model: string;
68
+ apiKey?: string;
69
+ }
70
+ //# sourceMappingURL=llm-caller.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"llm-caller.d.ts","sourceRoot":"","sources":["../src/llm-caller.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAoCG;AAEH,6DAA6D;AAC7D,MAAM,WAAW,UAAU;IACzB,IAAI,EAAE,MAAM,CAAC;IACb,WAAW,EAAE,MAAM,CAAC;IACpB,YAAY,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACvC;AAED,wEAAwE;AACxE,MAAM,WAAW,SAAS;IACxB;;;OAGG;IACH,IAAI,CACF,YAAY,EAAE,MAAM,EACpB,WAAW,EAAE,MAAM,EACnB,SAAS,CAAC,EAAE,MAAM,GACjB,OAAO,CAAC,MAAM,CAAC,CAAC;IAEnB;;;;;OAKG;IACH,cAAc,CAAC,CAAC,CAAC,EACf,YAAY,EAAE,MAAM,EACpB,WAAW,EAAE,MAAM,EACnB,IAAI,EAAE,UAAU,EAChB,SAAS,CAAC,EAAE,MAAM,GACjB,OAAO,CAAC,CAAC,CAAC,CAAC;CACf;AAED;;;;;GAKG;AACH,MAAM,WAAW,eAAe;IAC9B,QAAQ,EAAE,WAAW,GAAG,QAAQ,GAAG,QAAQ,GAAG,YAAY,GAAG,SAAS,CAAC;IACvE,KAAK,EAAE,MAAM,CAAC;IACd,MAAM,CAAC,EAAE,MAAM,CAAC;CACjB"}
@@ -0,0 +1,38 @@
1
+ /**
2
+ * `LLMCaller` — the decision engine's LLM dispatcher.
3
+ *
4
+ * Narrow abstraction over "call a chat model, optionally with a forced
5
+ * tool-use schema for guaranteed-JSON structured output." Kept public
6
+ * so OSS consumers of `@ggui-ai/negotiator` can bring their own LLM
7
+ * provider (Anthropic direct, OpenAI, Google, a local model, a
8
+ * community LiteLLM wrapper) without touching the decision-engine
9
+ * source.
10
+ *
11
+ * **Why this lives in `@ggui-ai/negotiator`, not
12
+ * `@ggui-ai/mcp-server-core`.** `mcp-server-core` contains the
13
+ * storage + runtime seams an MCP server implementer binds against
14
+ * (`VectorStore`, `EmbeddingProvider`, `KeyValueStore`,
15
+ * `BlueprintProvider`, `Negotiator`). `LLMCaller` is an
16
+ * engine-internal dispatcher — one level below `Negotiator` — so
17
+ * lifting it to `mcp-server-core` would grow the public seam count
18
+ * speculatively. If a second consumer outside the negotiator
19
+ * surfaces later, the "where does `LLMCaller` live?" question can be
20
+ * re-opened at that point.
21
+ *
22
+ * Normative semantics:
23
+ * - `call(systemPrompt, userMessage, maxTokens?)` returns the raw
24
+ * model text. Implementations MUST NOT inject tool-use blocks when
25
+ * the caller didn't request them — the text path is used as a
26
+ * regex-JSON fallback.
27
+ * - `callStructured?<T>(...)` is OPTIONAL. When present, it MUST
28
+ * force tool use against the supplied `ToolSchema` and return the
29
+ * tool input, parsed as `T`. Implementations that don't support
30
+ * forced structured output simply omit this method; consumers
31
+ * fall back to `call` + regex JSON extraction. Absence is not an
32
+ * error.
33
+ * - `ToolSchema.input_schema` follows the OpenAI tool-use JSON
34
+ * Schema convention. Implementations that use a different
35
+ * tool-use protocol (e.g., Anthropic's variant) MUST translate at
36
+ * the adapter boundary.
37
+ */
38
+ export {};
@@ -0,0 +1,101 @@
1
+ /**
2
+ * LLM rerank — Tier-2 precision oracle for the blueprint registry.
3
+ *
4
+ * Given a user's UI request (intent + contract structure) and a set
5
+ * of candidate cached blueprints retrieved by RAG, ask a fast LLM
6
+ * (Haiku 4.5) which candidate (if any) matches. Returns a structured
7
+ * decision so the caller can branch deterministically.
8
+ *
9
+ * This module is the precision half of the blueprint-first
10
+ * architecture: RAG retrieval is high-recall but low-precision (bge-
11
+ * small confuses topic-similar but UI-divergent prompts); the LLM
12
+ * judge restores precision. Combined break-even hit rate is ~10%;
13
+ * realistic workloads observe 30-70%.
14
+ */
15
+ import { summarizeContract } from '@ggui-ai/protocol';
16
+ import type { LLMCaller, ToolSchema } from './llm-caller.js';
17
+ /**
18
+ * One candidate blueprint for the LLM judge to consider.
19
+ *
20
+ * Keep this struct narrow — the prompt sees only what's necessary
21
+ * to decide match-vs-no-match. componentCode is intentionally
22
+ * absent (huge, distracting, doesn't change the decision).
23
+ */
24
+ export interface RerankCandidate {
25
+ /** Stable blueprint id — echoed back as `matchId` on a hit. */
26
+ readonly id: string;
27
+ /** The intent prose that originally produced this blueprint. */
28
+ readonly cachedIntent: string;
29
+ /**
30
+ * One-line summary of the blueprint's contract surface. Format
31
+ * matches `summarizeContract()` below — `slots=...; actions=...;
32
+ * streams=...; props=...` so the judge sees the structural shape
33
+ * without the JSON noise.
34
+ */
35
+ readonly cachedContractSummary: string;
36
+ /**
37
+ * Optional retrieval signal. Higher cosine biases the prior, but
38
+ * the judge's decision is the source of truth. Pass-through so
39
+ * the prompt can include "candidate retrieved at cosine X" if the
40
+ * judge would benefit (default: omit).
41
+ */
42
+ readonly cosine?: number;
43
+ }
44
+ /** Decision returned by the LLM judge. */
45
+ export interface RerankDecision {
46
+ /**
47
+ * The matched candidate's id, or `null` if no candidate matches.
48
+ * `null` means "all candidates rejected — generate fresh."
49
+ */
50
+ readonly matchId: string | null;
51
+ /**
52
+ * Confidence on `[0, 1]`. Caller compares against a threshold (e.g.
53
+ * 0.6) before treating the decision as a hit. Returned even when
54
+ * matchId is null so callers can log "judge declined with
55
+ * confidence X."
56
+ */
57
+ readonly confidence: number;
58
+ /**
59
+ * Free-text reason from the judge. Surface in trace logs so
60
+ * operators can debug "why didn't this hit." Truncate at the
61
+ * persistence boundary if cardinality is a concern.
62
+ */
63
+ readonly reason: string;
64
+ /** Wall-clock latency of the LLM call. */
65
+ readonly latencyMs: number;
66
+ /**
67
+ * Token cost of the call — for the cache-trace sink and cost
68
+ * accounting. Implementations that can't surface token counts may
69
+ * report `{input: 0, output: 0}` and the cost-per-call gate will
70
+ * have to be measured externally.
71
+ */
72
+ readonly tokenCost: {
73
+ readonly input: number;
74
+ readonly output: number;
75
+ };
76
+ }
77
+ /** Query the user's request the judge is matching against. */
78
+ export interface RerankQuery {
79
+ readonly intent: string;
80
+ readonly contractSummary: string;
81
+ }
82
+ declare const RERANK_SYSTEM_PROMPT = "You match user UI requests against previously-generated UI blueprints. Each blueprint was produced for a past request and stored. Decide whether any candidate produces the SAME USEFUL UI for the current request.\n\nMATCH means the candidate would correctly satisfy the user's current request \u2014 same UI shape (component types, layout pattern), same wire surface (slot names, action names), same intended user task, and same load-bearing parameters (dates, months, ranges, enum values).\n\nNO-MATCH means the candidate would NOT satisfy the user's current request \u2014 different task (haiku composer vs tweet draft, login vs signup), different UI shape (form vs list vs dashboard), or load-bearing parameters differ (calendar-Jan vs calendar-Mar \u2014 same contract, different value).\n\nVisual style differences alone (minimal vs ornate, dense vs spacious) DO NOT block a match \u2014 the user can refine those after they get a working UI.\n\nOutput exactly ONE tool call with your decision. confidence is a number in [0, 1]. matchId is a string from the candidate ids, or null when no candidate matches. reason is a short sentence the operator can use to debug.";
83
+ declare const RERANK_TOOL: ToolSchema;
84
+ export { summarizeContract };
85
+ /**
86
+ * Run the rerank judge against a query + candidate list.
87
+ *
88
+ * Empty candidate list short-circuits without an LLM call — the
89
+ * caller's RAG step decided there's nothing close enough; we don't
90
+ * burn tokens to confirm it.
91
+ *
92
+ * Operational errors (LLM throws, parse fails) collapse to a `null`
93
+ * match with confidence=0 and a diagnostic reason. The caller treats
94
+ * confidence-below-threshold as no-match anyway, so the failure mode
95
+ * lands on the cold-gen path automatically.
96
+ */
97
+ export declare function rerankCandidates(deps: {
98
+ readonly llm: LLMCaller;
99
+ }, query: RerankQuery, candidates: readonly RerankCandidate[]): Promise<RerankDecision>;
100
+ export { RERANK_SYSTEM_PROMPT, RERANK_TOOL };
101
+ //# sourceMappingURL=llm-rerank.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"llm-rerank.d.ts","sourceRoot":"","sources":["../src/llm-rerank.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;GAaG;AACH,OAAO,EAAE,iBAAiB,EAAE,MAAM,mBAAmB,CAAC;AACtD,OAAO,KAAK,EAAE,SAAS,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAE7D;;;;;;GAMG;AACH,MAAM,WAAW,eAAe;IAC9B,+DAA+D;IAC/D,QAAQ,CAAC,EAAE,EAAE,MAAM,CAAC;IACpB,gEAAgE;IAChE,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;IAC9B;;;;;OAKG;IACH,QAAQ,CAAC,qBAAqB,EAAE,MAAM,CAAC;IACvC;;;;;OAKG;IACH,QAAQ,CAAC,MAAM,CAAC,EAAE,MAAM,CAAC;CAC1B;AAED,0CAA0C;AAC1C,MAAM,WAAW,cAAc;IAC7B;;;OAGG;IACH,QAAQ,CAAC,OAAO,EAAE,MAAM,GAAG,IAAI,CAAC;IAChC;;;;;OAKG;IACH,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B;;;;OAIG;IACH,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,0CAA0C;IAC1C,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B;;;;;OAKG;IACH,QAAQ,CAAC,SAAS,EAAE;QAAE,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;QAAC,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAA;KAAE,CAAC;CACzE;AAED,8DAA8D;AAC9D,MAAM,WAAW,WAAW;IAC1B,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,QAAQ,CAAC,eAAe,EAAE,MAAM,CAAC;CAClC;AAED,QAAA,MAAM,oBAAoB,spCAQkM,CAAC;AAE7N,QAAA,MAAM,WAAW,EAAE,UA4BlB,CAAC;AAsCF,OAAO,EAAE,iBAAiB,EAAE,CAAC;AA8C7B;;;;;;;;;;;GAWG;AACH,wBAAsB,gBAAgB,CACpC,IAAI,EAAE;IAAE,QAAQ,CAAC,GAAG,EAAE,SAAS,CAAA;CAAE,EACjC,KAAK,EAAE,WAAW,EAClB,UAAU,EAAE,SAAS,eAAe,EAAE,GACrC,OAAO,CAAC,cAAc,CAAC,CAwDzB;AAGD,OAAO,EAAE,oBAAoB,EAAE,WAAW,EAAE,CAAC"}
@@ -0,0 +1,178 @@
1
+ /**
2
+ * LLM rerank — Tier-2 precision oracle for the blueprint registry.
3
+ *
4
+ * Given a user's UI request (intent + contract structure) and a set
5
+ * of candidate cached blueprints retrieved by RAG, ask a fast LLM
6
+ * (Haiku 4.5) which candidate (if any) matches. Returns a structured
7
+ * decision so the caller can branch deterministically.
8
+ *
9
+ * This module is the precision half of the blueprint-first
10
+ * architecture: RAG retrieval is high-recall but low-precision (bge-
11
+ * small confuses topic-similar but UI-divergent prompts); the LLM
12
+ * judge restores precision. Combined break-even hit rate is ~10%;
13
+ * realistic workloads observe 30-70%.
14
+ */
15
+ import { summarizeContract } from '@ggui-ai/protocol';
16
+ const RERANK_SYSTEM_PROMPT = `You match user UI requests against previously-generated UI blueprints. Each blueprint was produced for a past request and stored. Decide whether any candidate produces the SAME USEFUL UI for the current request.
17
+
18
+ MATCH means the candidate would correctly satisfy the user's current request — same UI shape (component types, layout pattern), same wire surface (slot names, action names), same intended user task, and same load-bearing parameters (dates, months, ranges, enum values).
19
+
20
+ NO-MATCH means the candidate would NOT satisfy the user's current request — different task (haiku composer vs tweet draft, login vs signup), different UI shape (form vs list vs dashboard), or load-bearing parameters differ (calendar-Jan vs calendar-Mar — same contract, different value).
21
+
22
+ Visual style differences alone (minimal vs ornate, dense vs spacious) DO NOT block a match — the user can refine those after they get a working UI.
23
+
24
+ Output exactly ONE tool call with your decision. confidence is a number in [0, 1]. matchId is a string from the candidate ids, or null when no candidate matches. reason is a short sentence the operator can use to debug.`;
25
+ const RERANK_TOOL = {
26
+ name: 'submit_rerank_decision',
27
+ description: 'Submit your match-vs-no-match decision over the candidates.',
28
+ input_schema: {
29
+ type: 'object',
30
+ additionalProperties: false,
31
+ properties: {
32
+ matchId: {
33
+ type: ['string', 'null'],
34
+ description: "ID of the matching candidate, or null when no candidate matches. MUST be one of the candidate ids supplied in the user message, or null.",
35
+ },
36
+ confidence: {
37
+ type: 'number',
38
+ minimum: 0,
39
+ maximum: 1,
40
+ description: 'Confidence in the decision on [0, 1]. Caller will compare against a threshold before treating it as a hit.',
41
+ },
42
+ reason: {
43
+ type: 'string',
44
+ description: 'Brief explanation — one sentence — that the operator can use to debug match decisions.',
45
+ },
46
+ },
47
+ required: ['matchId', 'confidence', 'reason'],
48
+ },
49
+ };
50
+ function buildUserMessage(query, candidates) {
51
+ const lines = [];
52
+ lines.push('CURRENT REQUEST');
53
+ lines.push(` intent: ${query.intent}`);
54
+ lines.push(` contract: ${query.contractSummary}`);
55
+ lines.push('');
56
+ lines.push(`CANDIDATES (${candidates.length})`);
57
+ for (const c of candidates) {
58
+ lines.push('---');
59
+ lines.push(` id: ${c.id}`);
60
+ lines.push(` intent: ${truncate(c.cachedIntent, 280)}`);
61
+ lines.push(` contract: ${c.cachedContractSummary}`);
62
+ if (typeof c.cosine === 'number') {
63
+ lines.push(` cosine: ${c.cosine.toFixed(3)}`);
64
+ }
65
+ }
66
+ lines.push('');
67
+ lines.push('Decide: does any candidate match the current request? Submit via the tool.');
68
+ return lines.join('\n');
69
+ }
70
+ function truncate(text, max) {
71
+ if (text.length <= max)
72
+ return text;
73
+ return `${text.slice(0, max - 1)}…`;
74
+ }
75
+ // Re-export `summarizeContract` from protocol for backwards-compatible
76
+ // access through `@ggui-ai/negotiator/llm-rerank` consumers (the
77
+ // canonical home is `@ggui-ai/protocol` — both the registry storage
78
+ // and the rerank prompt depend on it). Kept as a re-export so the
79
+ // existing test imports here keep working.
80
+ export { summarizeContract };
81
+ function clampConfidence(value) {
82
+ if (typeof value !== 'number' || !Number.isFinite(value))
83
+ return 0;
84
+ if (value < 0)
85
+ return 0;
86
+ if (value > 1)
87
+ return 1;
88
+ return value;
89
+ }
90
+ function parseToolInput(raw, candidateIds) {
91
+ if (raw === null || typeof raw !== 'object') {
92
+ return { matchId: null, confidence: 0, reason: 'parse-failed: non-object tool input' };
93
+ }
94
+ const obj = raw;
95
+ const reason = typeof obj['reason'] === 'string' ? obj['reason'] : '';
96
+ const confidence = clampConfidence(obj['confidence']);
97
+ const rawMatchId = obj['matchId'];
98
+ if (rawMatchId === null || rawMatchId === undefined) {
99
+ return { matchId: null, confidence, reason };
100
+ }
101
+ if (typeof rawMatchId !== 'string') {
102
+ return {
103
+ matchId: null,
104
+ confidence: 0,
105
+ reason: `parse-failed: matchId was ${typeof rawMatchId}, not string|null`,
106
+ };
107
+ }
108
+ if (!candidateIds.has(rawMatchId)) {
109
+ return {
110
+ matchId: null,
111
+ confidence: 0,
112
+ reason: `judge returned matchId='${rawMatchId}' but that id is not in the candidate set — treating as no-match`,
113
+ };
114
+ }
115
+ return { matchId: rawMatchId, confidence, reason };
116
+ }
117
+ /**
118
+ * Run the rerank judge against a query + candidate list.
119
+ *
120
+ * Empty candidate list short-circuits without an LLM call — the
121
+ * caller's RAG step decided there's nothing close enough; we don't
122
+ * burn tokens to confirm it.
123
+ *
124
+ * Operational errors (LLM throws, parse fails) collapse to a `null`
125
+ * match with confidence=0 and a diagnostic reason. The caller treats
126
+ * confidence-below-threshold as no-match anyway, so the failure mode
127
+ * lands on the cold-gen path automatically.
128
+ */
129
+ export async function rerankCandidates(deps, query, candidates) {
130
+ const startedAt = Date.now();
131
+ if (candidates.length === 0) {
132
+ return {
133
+ matchId: null,
134
+ confidence: 0,
135
+ reason: 'no candidates — short-circuited without LLM call',
136
+ latencyMs: Date.now() - startedAt,
137
+ tokenCost: { input: 0, output: 0 },
138
+ };
139
+ }
140
+ const userMessage = buildUserMessage(query, candidates);
141
+ const candidateIds = new Set(candidates.map((c) => c.id));
142
+ if (typeof deps.llm.callStructured !== 'function') {
143
+ return {
144
+ matchId: null,
145
+ confidence: 0,
146
+ reason: 'llm-rerank: provider does not support callStructured (forced tool-use). Bind a structured-capable LLMCaller (Anthropic adapter via mcp-server) for rerank.',
147
+ latencyMs: Date.now() - startedAt,
148
+ tokenCost: { input: 0, output: 0 },
149
+ };
150
+ }
151
+ let toolInput;
152
+ try {
153
+ toolInput = await deps.llm.callStructured(RERANK_SYSTEM_PROMPT, userMessage, RERANK_TOOL, 512);
154
+ }
155
+ catch (err) {
156
+ const message = err instanceof Error ? err.message : String(err);
157
+ return {
158
+ matchId: null,
159
+ confidence: 0,
160
+ reason: `llm-rerank: callStructured threw — ${message}`,
161
+ latencyMs: Date.now() - startedAt,
162
+ tokenCost: { input: 0, output: 0 },
163
+ };
164
+ }
165
+ const parsed = parseToolInput(toolInput, candidateIds);
166
+ return {
167
+ matchId: parsed.matchId,
168
+ confidence: parsed.confidence,
169
+ reason: parsed.reason,
170
+ latencyMs: Date.now() - startedAt,
171
+ // Token cost surfacing requires LLMCaller-level instrumentation
172
+ // we don't have today. Default to zero; the cost gate is measured
173
+ // out-of-band from billing data during the probe.
174
+ tokenCost: { input: 0, output: 0 },
175
+ };
176
+ }
177
+ // Re-exports for the eval harness — keep public surface explicit.
178
+ export { RERANK_SYSTEM_PROMPT, RERANK_TOOL };
@@ -0,0 +1,141 @@
1
+ /**
2
+ * Negotiator — top-level orchestrator over the public storage +
3
+ * decision-engine seams.
4
+ *
5
+ * Pipeline:
6
+ * 1. RAG search (per-scope + optional shared pool) via
7
+ * {@link ragSearch} — composes `EmbeddingProvider.embed` +
8
+ * `VectorStore.query` from `@ggui-ai/mcp-server-core`.
9
+ * 2. Read session state (optional injectable).
10
+ * 3. Fast-path for exact blueprint hits — skip the decision LLM.
11
+ * 4. Otherwise call {@link makeDecision} with the RAG candidates and
12
+ * session stack; fold the picked blueprint's pool provenance into
13
+ * the return value.
14
+ *
15
+ * Timing logs (stable format — consumed by benchmarks):
16
+ * [negotiate] embedding: Xms | search: Xms | candidates: N
17
+ * [negotiate] candidate: <id8> | <description> | hash=<h16>
18
+ * [negotiate] FAST PATH: blueprint=<id8> hash=<h12> | LLM decision: 0ms | total: Xms
19
+ * [negotiate] LLM PATH: action=X blueprint=<id8> hash=<h12> | LLM decision: Xms | total: Xms
20
+ *
21
+ * ### Public surface + semver weight
22
+ *
23
+ * Exported:
24
+ * - `negotiate(deps, input)` — runtime orchestrator.
25
+ * - `NegotiateDeps` — injection shape (embedding / vectors / llm +
26
+ * optional session-state reader + optional progress callback).
27
+ * - `NegotiateInput` — agent signal + config.
28
+ * - `NegotiateConfig` — minimum fields the orchestrator actually
29
+ * reads. Pool selection is expressed as `includeSharedPool:
30
+ * boolean` rather than a `poolMode` enum, so callers decide
31
+ * shared-pool inclusion explicitly at each call site.
32
+ * - `NegotiateResult` — the decision result returned to callers.
33
+ *
34
+ * ### Why this package, not `mcp-server-core`
35
+ *
36
+ * `mcp-server-core` locks storage/runtime seams MCP server
37
+ * implementers bind against (`EmbeddingProvider`, `VectorStore`,
38
+ * `Negotiator`). `negotiate()` is a *composition* over those seams —
39
+ * decision-engine semantics, not a new seam. Adding it to
40
+ * `mcp-server-core` would drag the LLM prompts + tool-schemas into a
41
+ * package whose job is to stay minimal and runtime-agnostic.
42
+ */
43
+ import type { NegotiatorAlternative, NegotiatorDecision } from '@ggui-ai/protocol';
44
+ import type { EmbeddingProvider, VectorStore } from '@ggui-ai/mcp-server-core';
45
+ import type { LLMCaller } from './llm-caller.js';
46
+ import type { SessionState } from './session.js';
47
+ /**
48
+ * Minimum config the orchestrator reads.
49
+ *
50
+ * `appId` is the primary RAG scope (per-app registered UIs live
51
+ * here). `sessionId` keys the optional `readSessionState` callback.
52
+ * `includeSharedPool` (default `false`) gates whether to also search
53
+ * the global `"shared"` pool in parallel and fold its hits into the
54
+ * candidate set.
55
+ */
56
+ export interface NegotiateConfig {
57
+ appId: string;
58
+ sessionId: string;
59
+ /** Also search the shared (global) pool in parallel. Default `false`. */
60
+ includeSharedPool?: boolean;
61
+ }
62
+ /**
63
+ * Agent signal + config for a single negotiation call. Bundled as one
64
+ * object so future additive fields don't force yet another positional
65
+ * arg on the public API.
66
+ */
67
+ export interface NegotiateInput {
68
+ agent: {
69
+ /** Raw data the agent wants to render. */
70
+ data?: Record<string, unknown>;
71
+ /** Natural-language prompt. Used as RAG query when present. */
72
+ prompt?: string;
73
+ /** Free-form context string or structured map. Forwarded to the decision LLM. */
74
+ context?: string | Record<string, unknown>;
75
+ /**
76
+ * Names of MCP tools the agent invokes (catalog seed). Merged
77
+ * into the decision's `agentCapabilities.tools` deterministically
78
+ * (see {@link makeDecision}).
79
+ */
80
+ agentTools?: string[];
81
+ /**
82
+ * Browser-capability gadget catalog the app exposes (default
83
+ * `STDLIB_GADGETS`, operator-extensible). Forwarded to
84
+ * the decision LLM so it knows which gadget bindings are
85
+ * available; canonical entries enrich partial LLM output
86
+ * downstream (see {@link mergeGadgets}).
87
+ */
88
+ gadgets?: readonly import('@ggui-ai/protocol').GadgetDescriptor[];
89
+ };
90
+ config: NegotiateConfig;
91
+ }
92
+ /**
93
+ * Dependencies injected into {@link negotiate}. Public field names
94
+ * (`embedding` / `vectors` / `llm`) align with {@link ragSearch}'s
95
+ * already-shipped deps shape.
96
+ */
97
+ export interface NegotiateDeps {
98
+ /**
99
+ * Embedding provider for the RAG search step. **Optional** —
100
+ * when omitted (paired with omitted `vectors`), the negotiator
101
+ * skips RAG entirely and runs the decision LLM against an empty
102
+ * candidate list. OSS without vector-store infrastructure binds
103
+ * with `embedding: undefined, vectors: undefined` and still gets
104
+ * useful negotiation via the decision LLM alone.
105
+ *
106
+ * `embedding` and `vectors` are paired — both must be present for
107
+ * RAG to fire, or both must be absent. A half-bound config (one
108
+ * present, one missing) is a configuration bug; the search step
109
+ * skips and logs a warn rather than erroring, but consumers
110
+ * should fix the binding.
111
+ */
112
+ embedding?: EmbeddingProvider;
113
+ vectors?: VectorStore;
114
+ llm: LLMCaller;
115
+ /** Optional: read current session state for stack-aware decisions. */
116
+ readSessionState?: (sessionId: string) => Promise<SessionState | null>;
117
+ /** Optional: surface pipeline progress to consumers. */
118
+ onProgress?: (phase: string, summary: string) => void;
119
+ }
120
+ /**
121
+ * Output shape — mirrors the legacy `DecisionResult` so the
122
+ * back-compat shim can return the object verbatim. See `NegotiateDeps`
123
+ * semver note.
124
+ */
125
+ export interface NegotiateResult {
126
+ decision: NegotiatorDecision;
127
+ alternatives: NegotiatorAlternative[];
128
+ /** Stored contract hash from blueprint match — deterministic pool key. */
129
+ storedContractHash?: string;
130
+ /** Which pool the matched blueprint's code lives in. */
131
+ storedPoolSource?: 'shared' | 'private';
132
+ embeddingLatencyMs: number;
133
+ searchLatencyMs: number;
134
+ decisionLatencyMs: number;
135
+ }
136
+ /**
137
+ * Orchestrate one negotiation call — RAG search, session read, fast
138
+ * path, decision LLM. See module docstring for the pipeline outline.
139
+ */
140
+ export declare function negotiate(deps: NegotiateDeps, input: NegotiateInput): Promise<NegotiateResult>;
141
+ //# sourceMappingURL=negotiate.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"negotiate.d.ts","sourceRoot":"","sources":["../src/negotiate.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAyCG;AAEH,OAAO,KAAK,EAEV,qBAAqB,EACrB,kBAAkB,EACnB,MAAM,mBAAmB,CAAC;AAC3B,OAAO,KAAK,EACV,iBAAiB,EACjB,WAAW,EACZ,MAAM,0BAA0B,CAAC;AAClC,OAAO,KAAK,EAAE,SAAS,EAAE,MAAM,iBAAiB,CAAC;AACjD,OAAO,KAAK,EAAE,YAAY,EAAE,MAAM,cAAc,CAAC;AAWjD;;;;;;;;GAQG;AACH,MAAM,WAAW,eAAe;IAC9B,KAAK,EAAE,MAAM,CAAC;IACd,SAAS,EAAE,MAAM,CAAC;IAClB,yEAAyE;IACzE,iBAAiB,CAAC,EAAE,OAAO,CAAC;CAC7B;AAED;;;;GAIG;AACH,MAAM,WAAW,cAAc;IAC7B,KAAK,EAAE;QACL,0CAA0C;QAC1C,IAAI,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;QAC/B,+DAA+D;QAC/D,MAAM,CAAC,EAAE,MAAM,CAAC;QAChB,iFAAiF;QACjF,OAAO,CAAC,EAAE,MAAM,GAAG,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;QAC3C;;;;WAIG;QACH,UAAU,CAAC,EAAE,MAAM,EAAE,CAAC;QACtB;;;;;;WAMG;QACH,OAAO,CAAC,EAAE,SAAS,OAAO,mBAAmB,EAAE,gBAAgB,EAAE,CAAC;KACnE,CAAC;IACF,MAAM,EAAE,eAAe,CAAC;CACzB;AAED;;;;GAIG;AACH,MAAM,WAAW,aAAa;IAC5B;;;;;;;;;;;;;OAaG;IACH,SAAS,CAAC,EAAE,iBAAiB,CAAC;IAC9B,OAAO,CAAC,EAAE,WAAW,CAAC;IACtB,GAAG,EAAE,SAAS,CAAC;IACf,sEAAsE;IACtE,gBAAgB,CAAC,EAAE,CAAC,SAAS,EAAE,MAAM,KAAK,OAAO,CAAC,YAAY,GAAG,IAAI,CAAC,CAAC;IACvE,wDAAwD;IACxD,UAAU,CAAC,EAAE,CAAC,KAAK,EAAE,MAAM,EAAE,OAAO,EAAE,MAAM,KAAK,IAAI,CAAC;CACvD;AAED;;;;GAIG;AACH,MAAM,WAAW,eAAe;IAC9B,QAAQ,EAAE,kBAAkB,CAAC;IAC7B,YAAY,EAAE,qBAAqB,EAAE,CAAC;IACtC,0EAA0E;IAC1E,kBAAkB,CAAC,EAAE,MAAM,CAAC;IAC5B,wDAAwD;IACxD,gBAAgB,CAAC,EAAE,QAAQ,GAAG,SAAS,CAAC;IACxC,kBAAkB,EAAE,MAAM,CAAC;IAC3B,eAAe,EAAE,MAAM,CAAC;IACxB,iBAAiB,EAAE,MAAM,CAAC;CAC3B;AAED;;;GAGG;AACH,wBAAsB,SAAS,CAC7B,IAAI,EAAE,aAAa,EACnB,KAAK,EAAE,cAAc,GACpB,OAAO,CAAC,eAAe,CAAC,CAmJ1B"}