@ggui-ai/negotiator 0.23.0 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +2 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -1
- package/dist/llm-rerank.d.ts +18 -16
- package/dist/llm-rerank.d.ts.map +1 -1
- package/dist/llm-rerank.js +28 -2
- package/dist/rerank-eval/run-probe-cli.js +4 -1
- package/package.json +3 -3
- package/src/index.ts +2 -1
- package/src/llm-rerank.ts +31 -3
- package/src/rerank-eval/run-probe-cli.ts +3 -1
package/dist/index.d.ts
CHANGED
|
@@ -26,8 +26,8 @@
|
|
|
26
26
|
*/
|
|
27
27
|
export { hashContract, buildVariant } from './contract-hash.js';
|
|
28
28
|
export type { LLMCaller, LLMCallerConfig, ToolSchema } from './llm-caller.js';
|
|
29
|
-
export { rerankCandidates } from './llm-rerank.js';
|
|
30
|
-
export type { RerankCandidate, RerankDecision, RerankQuery, } from './llm-rerank.js';
|
|
29
|
+
export { llmRerankJudge, rerankCandidates } from './llm-rerank.js';
|
|
30
|
+
export type { RerankCandidate, RerankDecision, RerankJudge, RerankQuery, } from './llm-rerank.js';
|
|
31
31
|
export { synthesizeContract } from './synthesize-contract.js';
|
|
32
32
|
export type { SynthesizeContractResult } from './synthesize-contract.js';
|
|
33
33
|
export { ensureConformingContract } from './ensure-conforming-contract.js';
|
package/dist/index.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AAEH,OAAO,EAAE,YAAY,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAChE,YAAY,EAAE,SAAS,EAAE,eAAe,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAC9E,OAAO,EAAE,gBAAgB,EAAE,MAAM,iBAAiB,CAAC;
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AAEH,OAAO,EAAE,YAAY,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAChE,YAAY,EAAE,SAAS,EAAE,eAAe,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAC9E,OAAO,EAAE,cAAc,EAAE,gBAAgB,EAAE,MAAM,iBAAiB,CAAC;AACnE,YAAY,EACV,eAAe,EACf,cAAc,EACd,WAAW,EACX,WAAW,GACZ,MAAM,iBAAiB,CAAC;AACzB,OAAO,EAAE,kBAAkB,EAAE,MAAM,0BAA0B,CAAC;AAC9D,YAAY,EAAE,wBAAwB,EAAE,MAAM,0BAA0B,CAAC;AACzE,OAAO,EAAE,wBAAwB,EAAE,MAAM,iCAAiC,CAAC;AAC3E,YAAY,EACV,wBAAwB,EACxB,wBAAwB,EACxB,sBAAsB,EACtB,sBAAsB,GACvB,MAAM,iCAAiC,CAAC;AACzC,OAAO,EAAE,cAAc,EAAE,MAAM,sBAAsB,CAAC;AAItD,OAAO,EACL,uBAAuB,EACvB,kBAAkB,EAClB,KAAK,aAAa,GACnB,MAAM,oBAAoB,CAAC;AAC5B,OAAO,EACL,0BAA0B,EAC1B,uBAAuB,EACvB,wBAAwB,GACzB,MAAM,0BAA0B,CAAC;AAClC,YAAY,EACV,yBAAyB,EACzB,6BAA6B,EAC7B,wBAAwB,EACxB,6BAA6B,EAC7B,gCAAgC,GACjC,MAAM,0BAA0B,CAAC"}
|
package/dist/index.js
CHANGED
|
@@ -25,7 +25,7 @@
|
|
|
25
25
|
* consumers need. Each additive export carries semver weight.
|
|
26
26
|
*/
|
|
27
27
|
export { hashContract, buildVariant } from './contract-hash.js';
|
|
28
|
-
export { rerankCandidates } from './llm-rerank.js';
|
|
28
|
+
export { llmRerankJudge, rerankCandidates } from './llm-rerank.js';
|
|
29
29
|
export { synthesizeContract } from './synthesize-contract.js';
|
|
30
30
|
export { ensureConformingContract } from './ensure-conforming-contract.js';
|
|
31
31
|
export { normalizeDraft } from './normalize-draft.js';
|
package/dist/llm-rerank.d.ts
CHANGED
|
@@ -1,17 +1,3 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* LLM rerank — Tier-2 precision oracle for the blueprint registry.
|
|
3
|
-
*
|
|
4
|
-
* Given a user's UI request (intent + contract structure) and a set
|
|
5
|
-
* of candidate cached blueprints retrieved by RAG, ask a fast LLM
|
|
6
|
-
* (Haiku 4.5) which candidate (if any) matches. Returns a structured
|
|
7
|
-
* decision so the caller can branch deterministically.
|
|
8
|
-
*
|
|
9
|
-
* This module is the precision half of the blueprint-first
|
|
10
|
-
* architecture: RAG retrieval is high-recall but low-precision (bge-
|
|
11
|
-
* small confuses topic-similar but UI-divergent prompts); the LLM
|
|
12
|
-
* judge restores precision. Combined break-even hit rate is ~10%;
|
|
13
|
-
* realistic workloads observe 30-70%.
|
|
14
|
-
*/
|
|
15
1
|
import type { LLMCaller, ToolSchema } from './llm-caller.js';
|
|
16
2
|
/**
|
|
17
3
|
* One candidate blueprint for the LLM judge to consider.
|
|
@@ -58,11 +44,12 @@ export interface RerankDecision {
|
|
|
58
44
|
*/
|
|
59
45
|
readonly confidence: number;
|
|
60
46
|
/**
|
|
61
|
-
* Free-text reason from the judge
|
|
47
|
+
* Free-text reason from the judge, when it gives one — a judge that
|
|
48
|
+
* decides without prose sends none (ggui#1235). Surface in trace logs so
|
|
62
49
|
* operators can debug "why didn't this hit." Truncate at the
|
|
63
50
|
* persistence boundary if cardinality is a concern.
|
|
64
51
|
*/
|
|
65
|
-
readonly reason
|
|
52
|
+
readonly reason?: string;
|
|
66
53
|
/** Wall-clock latency of the LLM call. */
|
|
67
54
|
readonly latencyMs: number;
|
|
68
55
|
/**
|
|
@@ -98,5 +85,20 @@ declare const RERANK_TOOL: ToolSchema;
|
|
|
98
85
|
export declare function rerankCandidates(deps: {
|
|
99
86
|
readonly llm: LLMCaller;
|
|
100
87
|
}, query: RerankQuery, candidates: readonly RerankCandidate[]): Promise<RerankDecision>;
|
|
88
|
+
/**
|
|
89
|
+
* The judge seam (ggui#1235): one function from a query and its candidates
|
|
90
|
+
* to a {@link RerankDecision}. The matcher takes a judge together with the
|
|
91
|
+
* confidence threshold it was measured on (the pair), so a judge on
|
|
92
|
+
* another scale never meets a cut calibrated for a different one.
|
|
93
|
+
* `matchId: null` is a judge's only decline; `confidence` is the judge's
|
|
94
|
+
* own, compared by the caller against the pair's threshold.
|
|
95
|
+
*/
|
|
96
|
+
export type RerankJudge = (query: RerankQuery, candidates: readonly RerankCandidate[]) => Promise<RerankDecision>;
|
|
97
|
+
/**
|
|
98
|
+
* Today's judge as a {@link RerankJudge}: {@link rerankCandidates} bound to
|
|
99
|
+
* an {@link LLMCaller} — the same prompt, the same tool, the same decision,
|
|
100
|
+
* so a caller that injects nothing else behaves exactly as before.
|
|
101
|
+
*/
|
|
102
|
+
export declare function llmRerankJudge(llm: LLMCaller): RerankJudge;
|
|
101
103
|
export { RERANK_SYSTEM_PROMPT, RERANK_TOOL };
|
|
102
104
|
//# sourceMappingURL=llm-rerank.d.ts.map
|
package/dist/llm-rerank.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"llm-rerank.d.ts","sourceRoot":"","sources":["../src/llm-rerank.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"llm-rerank.d.ts","sourceRoot":"","sources":["../src/llm-rerank.ts"],"names":[],"mappings":"AAeA,OAAO,KAAK,EAAE,SAAS,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAE7D;;;;;;GAMG;AACH,MAAM,WAAW,eAAe;IAC9B,+DAA+D;IAC/D,QAAQ,CAAC,EAAE,EAAE,MAAM,CAAC;IACpB,gEAAgE;IAChE,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;IAC9B;;;;;OAKG;IACH,QAAQ,CAAC,qBAAqB,EAAE,MAAM,CAAC;IACvC;;;;;OAKG;IACH,QAAQ,CAAC,MAAM,CAAC,EAAE,MAAM,CAAC;CAC1B;AAED,0CAA0C;AAC1C,MAAM,WAAW,cAAc;IAC7B;;;OAGG;IACH,QAAQ,CAAC,OAAO,EAAE,MAAM,GAAG,IAAI,CAAC;IAChC;;;;;;;;OAQG;IACH,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B;;;;;OAKG;IACH,QAAQ,CAAC,MAAM,CAAC,EAAE,MAAM,CAAC;IACzB,0CAA0C;IAC1C,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B;;;;;OAKG;IACH,QAAQ,CAAC,SAAS,EAAE;QAAE,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;QAAC,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAA;KAAE,CAAC;CACzE;AAED,8DAA8D;AAC9D,MAAM,WAAW,WAAW;IAC1B,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,QAAQ,CAAC,eAAe,EAAE,MAAM,CAAC;CAClC;AAED,QAAA,MAAM,oBAAoB,03DAQ6U,CAAC;AAExW,QAAA,MAAM,WAAW,EAAE,UA4BlB,CAAC;AAkFF;;;;;;;;;;;GAWG;AACH,wBAAsB,gBAAgB,CACpC,IAAI,EAAE;IAAE,QAAQ,CAAC,GAAG,EAAE,SAAS,CAAA;CAAE,EACjC,KAAK,EAAE,WAAW,EAClB,UAAU,EAAE,SAAS,eAAe,EAAE,GACrC,OAAO,CAAC,cAAc,CAAC,CAwDzB;AAGD;;;;;;;GAOG;AACH,MAAM,MAAM,WAAW,GAAG,CACxB,KAAK,EAAE,WAAW,EAClB,UAAU,EAAE,SAAS,eAAe,EAAE,KACnC,OAAO,CAAC,cAAc,CAAC,CAAC;AAE7B;;;;GAIG;AACH,wBAAgB,cAAc,CAAC,GAAG,EAAE,SAAS,GAAG,WAAW,CAE1D;AAED,OAAO,EAAE,oBAAoB,EAAE,WAAW,EAAE,CAAC"}
|
package/dist/llm-rerank.js
CHANGED
|
@@ -1,3 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* LLM rerank — Tier-2 precision oracle for the blueprint registry.
|
|
3
|
+
*
|
|
4
|
+
* Given a user's UI request (intent + contract structure) and a set
|
|
5
|
+
* of candidate cached blueprints retrieved by RAG, ask a fast LLM
|
|
6
|
+
* (Haiku 4.5) which candidate (if any) matches. Returns a structured
|
|
7
|
+
* decision so the caller can branch deterministically.
|
|
8
|
+
*
|
|
9
|
+
* This module is the precision half of the blueprint-first
|
|
10
|
+
* architecture: RAG retrieval is high-recall but low-precision (bge-
|
|
11
|
+
* small confuses topic-similar but UI-divergent prompts); the LLM
|
|
12
|
+
* judge restores precision. Combined break-even hit rate is ~10%;
|
|
13
|
+
* realistic workloads observe 30-70%.
|
|
14
|
+
*/
|
|
15
|
+
import { MATCHED_INTENT_MAX_CHARS } from '@ggui-ai/protocol';
|
|
1
16
|
const RERANK_SYSTEM_PROMPT = `You match user UI requests against previously-generated UI blueprints. Each blueprint was produced for a past request and stored. Decide whether any candidate belongs to the SAME FAMILY as the current request — i.e. it would serve as a reasonable starting point that the requester can refine, not a pixel-exact replica.
|
|
2
17
|
|
|
3
18
|
MATCH means the candidate is the same intended user task AND the same broad UI shape (component types, layout pattern). A candidate still MATCHES when the current request adds or omits fields, slots, or actions relative to the cached blueprint — a superset, a subset, or an overlapping wire surface all still match. Added or omitted fields/slots/actions DO NOT block a match and are NOT yours to judge: those wire-surface deltas are reconciled and reported to the agent separately, after you decide. Judge similarity of task and shape, never coverage of fields.
|
|
@@ -42,7 +57,11 @@ function buildUserMessage(query, candidates) {
|
|
|
42
57
|
for (const c of candidates) {
|
|
43
58
|
lines.push('---');
|
|
44
59
|
lines.push(` id: ${c.id}`);
|
|
45
|
-
|
|
60
|
+
// The cap is the protocol's `MATCHED_INTENT_MAX_CHARS`: the same
|
|
61
|
+
// string the judge reads here is what a judged hit hands back to the
|
|
62
|
+
// agent as `blueprintMeta.matchedIntent` (ggui#1336), so the two cuts
|
|
63
|
+
// are one constant, not two literals that happen to agree.
|
|
64
|
+
lines.push(` intent: ${truncate(c.cachedIntent, MATCHED_INTENT_MAX_CHARS)}`);
|
|
46
65
|
lines.push(` contract: ${c.cachedContractSummary}`);
|
|
47
66
|
if (typeof c.cosine === 'number') {
|
|
48
67
|
lines.push(` cosine: ${c.cosine.toFixed(3)}`);
|
|
@@ -153,5 +172,12 @@ export async function rerankCandidates(deps, query, candidates) {
|
|
|
153
172
|
tokenCost: { input: 0, output: 0 },
|
|
154
173
|
};
|
|
155
174
|
}
|
|
156
|
-
|
|
175
|
+
/**
|
|
176
|
+
* Today's judge as a {@link RerankJudge}: {@link rerankCandidates} bound to
|
|
177
|
+
* an {@link LLMCaller} — the same prompt, the same tool, the same decision,
|
|
178
|
+
* so a caller that injects nothing else behaves exactly as before.
|
|
179
|
+
*/
|
|
180
|
+
export function llmRerankJudge(llm) {
|
|
181
|
+
return (query, candidates) => rerankCandidates({ llm }, query, candidates);
|
|
182
|
+
}
|
|
157
183
|
export { RERANK_SYSTEM_PROMPT, RERANK_TOOL };
|
|
@@ -65,7 +65,10 @@ async function main() {
|
|
|
65
65
|
const usage = getTokenUsage();
|
|
66
66
|
const totalCost = usage.input * HAIKU_4_5_PRICE_INPUT_PER_TOKEN +
|
|
67
67
|
usage.output * HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN;
|
|
68
|
-
const callsMade = report.outcomes.filter(
|
|
68
|
+
const callsMade = report.outcomes.filter(
|
|
69
|
+
// ggui#1235 — `reason` is optional on the seam; a judge with no prose
|
|
70
|
+
// still made a call.
|
|
71
|
+
(o) => !/short-circuited/.test(o.decision.reason ?? '')).length;
|
|
69
72
|
const costPerCall = callsMade === 0 ? 0 : totalCost / callsMade;
|
|
70
73
|
process.stdout.write(`Tokens: input=${usage.input} · output=${usage.output}\n`);
|
|
71
74
|
process.stdout.write(`Cost: total=$${totalCost.toFixed(4)} · per-call=$${costPerCall.toFixed(4)}\n`);
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ggui-ai/negotiator",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.24.0",
|
|
4
4
|
"description": "Contract-synthesis + match-judge engine for ggui's handshake. Synthesizes or repairs a conforming DataContract from an agent's draft, judges blueprint-match candidates for reuse, and validates contract structure + novelty — the primitives composed by decideHandshake in @ggui-ai/mcp-server-handlers. Deployment-agnostic: concrete embedding and vector-store bindings plug in via the storage interfaces from @ggui-ai/mcp-server-core.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"keywords": [
|
|
@@ -47,8 +47,8 @@
|
|
|
47
47
|
}
|
|
48
48
|
},
|
|
49
49
|
"dependencies": {
|
|
50
|
-
"@ggui-ai/mcp-server-core": "0.
|
|
51
|
-
"@ggui-ai/protocol": "0.
|
|
50
|
+
"@ggui-ai/mcp-server-core": "0.24.0",
|
|
51
|
+
"@ggui-ai/protocol": "0.24.0"
|
|
52
52
|
},
|
|
53
53
|
"devDependencies": {
|
|
54
54
|
"@types/node": "^24.0.0",
|
package/src/index.ts
CHANGED
|
@@ -27,10 +27,11 @@
|
|
|
27
27
|
|
|
28
28
|
export { hashContract, buildVariant } from './contract-hash.js';
|
|
29
29
|
export type { LLMCaller, LLMCallerConfig, ToolSchema } from './llm-caller.js';
|
|
30
|
-
export { rerankCandidates } from './llm-rerank.js';
|
|
30
|
+
export { llmRerankJudge, rerankCandidates } from './llm-rerank.js';
|
|
31
31
|
export type {
|
|
32
32
|
RerankCandidate,
|
|
33
33
|
RerankDecision,
|
|
34
|
+
RerankJudge,
|
|
34
35
|
RerankQuery,
|
|
35
36
|
} from './llm-rerank.js';
|
|
36
37
|
export { synthesizeContract } from './synthesize-contract.js';
|
package/src/llm-rerank.ts
CHANGED
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
* judge restores precision. Combined break-even hit rate is ~10%;
|
|
13
13
|
* realistic workloads observe 30-70%.
|
|
14
14
|
*/
|
|
15
|
+
import { MATCHED_INTENT_MAX_CHARS } from '@ggui-ai/protocol';
|
|
15
16
|
import type { LLMCaller, ToolSchema } from './llm-caller.js';
|
|
16
17
|
|
|
17
18
|
/**
|
|
@@ -60,11 +61,12 @@ export interface RerankDecision {
|
|
|
60
61
|
*/
|
|
61
62
|
readonly confidence: number;
|
|
62
63
|
/**
|
|
63
|
-
* Free-text reason from the judge
|
|
64
|
+
* Free-text reason from the judge, when it gives one — a judge that
|
|
65
|
+
* decides without prose sends none (ggui#1235). Surface in trace logs so
|
|
64
66
|
* operators can debug "why didn't this hit." Truncate at the
|
|
65
67
|
* persistence boundary if cardinality is a concern.
|
|
66
68
|
*/
|
|
67
|
-
readonly reason
|
|
69
|
+
readonly reason?: string;
|
|
68
70
|
/** Wall-clock latency of the LLM call. */
|
|
69
71
|
readonly latencyMs: number;
|
|
70
72
|
/**
|
|
@@ -135,7 +137,11 @@ function buildUserMessage(
|
|
|
135
137
|
for (const c of candidates) {
|
|
136
138
|
lines.push('---');
|
|
137
139
|
lines.push(` id: ${c.id}`);
|
|
138
|
-
|
|
140
|
+
// The cap is the protocol's `MATCHED_INTENT_MAX_CHARS`: the same
|
|
141
|
+
// string the judge reads here is what a judged hit hands back to the
|
|
142
|
+
// agent as `blueprintMeta.matchedIntent` (ggui#1336), so the two cuts
|
|
143
|
+
// are one constant, not two literals that happen to agree.
|
|
144
|
+
lines.push(` intent: ${truncate(c.cachedIntent, MATCHED_INTENT_MAX_CHARS)}`);
|
|
139
145
|
lines.push(` contract: ${c.cachedContractSummary}`);
|
|
140
146
|
if (typeof c.cosine === 'number') {
|
|
141
147
|
lines.push(` cosine: ${c.cosine.toFixed(3)}`);
|
|
@@ -273,4 +279,26 @@ export async function rerankCandidates(
|
|
|
273
279
|
}
|
|
274
280
|
|
|
275
281
|
// Re-exports for the eval harness — keep public surface explicit.
|
|
282
|
+
/**
|
|
283
|
+
* The judge seam (ggui#1235): one function from a query and its candidates
|
|
284
|
+
* to a {@link RerankDecision}. The matcher takes a judge together with the
|
|
285
|
+
* confidence threshold it was measured on (the pair), so a judge on
|
|
286
|
+
* another scale never meets a cut calibrated for a different one.
|
|
287
|
+
* `matchId: null` is a judge's only decline; `confidence` is the judge's
|
|
288
|
+
* own, compared by the caller against the pair's threshold.
|
|
289
|
+
*/
|
|
290
|
+
export type RerankJudge = (
|
|
291
|
+
query: RerankQuery,
|
|
292
|
+
candidates: readonly RerankCandidate[],
|
|
293
|
+
) => Promise<RerankDecision>;
|
|
294
|
+
|
|
295
|
+
/**
|
|
296
|
+
* Today's judge as a {@link RerankJudge}: {@link rerankCandidates} bound to
|
|
297
|
+
* an {@link LLMCaller} — the same prompt, the same tool, the same decision,
|
|
298
|
+
* so a caller that injects nothing else behaves exactly as before.
|
|
299
|
+
*/
|
|
300
|
+
export function llmRerankJudge(llm: LLMCaller): RerankJudge {
|
|
301
|
+
return (query, candidates) => rerankCandidates({ llm }, query, candidates);
|
|
302
|
+
}
|
|
303
|
+
|
|
276
304
|
export { RERANK_SYSTEM_PROMPT, RERANK_TOOL };
|
|
@@ -79,7 +79,9 @@ async function main(): Promise<void> {
|
|
|
79
79
|
usage.input * HAIKU_4_5_PRICE_INPUT_PER_TOKEN +
|
|
80
80
|
usage.output * HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN;
|
|
81
81
|
const callsMade = report.outcomes.filter(
|
|
82
|
-
|
|
82
|
+
// ggui#1235 — `reason` is optional on the seam; a judge with no prose
|
|
83
|
+
// still made a call.
|
|
84
|
+
(o) => !/short-circuited/.test(o.decision.reason ?? ''),
|
|
83
85
|
).length;
|
|
84
86
|
const costPerCall = callsMade === 0 ? 0 : totalCost / callsMade;
|
|
85
87
|
process.stdout.write(
|