@ggui-ai/negotiator 0.22.0 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +2 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -1
- package/dist/llm-caller.d.ts +5 -4
- package/dist/llm-caller.d.ts.map +1 -1
- package/dist/llm-caller.js +4 -3
- package/dist/llm-rerank.d.ts +18 -16
- package/dist/llm-rerank.d.ts.map +1 -1
- package/dist/llm-rerank.js +28 -2
- package/dist/rerank-eval/run-probe-cli.js +4 -1
- package/package.json +3 -3
- package/src/index.ts +2 -1
- package/src/llm-caller.ts +6 -5
- package/src/llm-rerank.ts +34 -5
- package/src/rerank-eval/run-probe-cli.ts +3 -1
- package/src/synth-bench/cli-llm.ts +3 -3
- package/src/synthesize-contract.ts +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -26,8 +26,8 @@
|
|
|
26
26
|
*/
|
|
27
27
|
export { hashContract, buildVariant } from './contract-hash.js';
|
|
28
28
|
export type { LLMCaller, LLMCallerConfig, ToolSchema } from './llm-caller.js';
|
|
29
|
-
export { rerankCandidates } from './llm-rerank.js';
|
|
30
|
-
export type { RerankCandidate, RerankDecision, RerankQuery, } from './llm-rerank.js';
|
|
29
|
+
export { llmRerankJudge, rerankCandidates } from './llm-rerank.js';
|
|
30
|
+
export type { RerankCandidate, RerankDecision, RerankJudge, RerankQuery, } from './llm-rerank.js';
|
|
31
31
|
export { synthesizeContract } from './synthesize-contract.js';
|
|
32
32
|
export type { SynthesizeContractResult } from './synthesize-contract.js';
|
|
33
33
|
export { ensureConformingContract } from './ensure-conforming-contract.js';
|
package/dist/index.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AAEH,OAAO,EAAE,YAAY,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAChE,YAAY,EAAE,SAAS,EAAE,eAAe,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAC9E,OAAO,EAAE,gBAAgB,EAAE,MAAM,iBAAiB,CAAC;
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AAEH,OAAO,EAAE,YAAY,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAChE,YAAY,EAAE,SAAS,EAAE,eAAe,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAC9E,OAAO,EAAE,cAAc,EAAE,gBAAgB,EAAE,MAAM,iBAAiB,CAAC;AACnE,YAAY,EACV,eAAe,EACf,cAAc,EACd,WAAW,EACX,WAAW,GACZ,MAAM,iBAAiB,CAAC;AACzB,OAAO,EAAE,kBAAkB,EAAE,MAAM,0BAA0B,CAAC;AAC9D,YAAY,EAAE,wBAAwB,EAAE,MAAM,0BAA0B,CAAC;AACzE,OAAO,EAAE,wBAAwB,EAAE,MAAM,iCAAiC,CAAC;AAC3E,YAAY,EACV,wBAAwB,EACxB,wBAAwB,EACxB,sBAAsB,EACtB,sBAAsB,GACvB,MAAM,iCAAiC,CAAC;AACzC,OAAO,EAAE,cAAc,EAAE,MAAM,sBAAsB,CAAC;AAItD,OAAO,EACL,uBAAuB,EACvB,kBAAkB,EAClB,KAAK,aAAa,GACnB,MAAM,oBAAoB,CAAC;AAC5B,OAAO,EACL,0BAA0B,EAC1B,uBAAuB,EACvB,wBAAwB,GACzB,MAAM,0BAA0B,CAAC;AAClC,YAAY,EACV,yBAAyB,EACzB,6BAA6B,EAC7B,wBAAwB,EACxB,6BAA6B,EAC7B,gCAAgC,GACjC,MAAM,0BAA0B,CAAC"}
|
package/dist/index.js
CHANGED
|
@@ -25,7 +25,7 @@
|
|
|
25
25
|
* consumers need. Each additive export carries semver weight.
|
|
26
26
|
*/
|
|
27
27
|
export { hashContract, buildVariant } from './contract-hash.js';
|
|
28
|
-
export { rerankCandidates } from './llm-rerank.js';
|
|
28
|
+
export { llmRerankJudge, rerankCandidates } from './llm-rerank.js';
|
|
29
29
|
export { synthesizeContract } from './synthesize-contract.js';
|
|
30
30
|
export { ensureConformingContract } from './ensure-conforming-contract.js';
|
|
31
31
|
export { normalizeDraft } from './normalize-draft.js';
|
package/dist/llm-caller.d.ts
CHANGED
|
@@ -24,9 +24,10 @@
|
|
|
24
24
|
* model text. Implementations MUST NOT inject tool-use blocks when
|
|
25
25
|
* the caller didn't request them — the text path is used as a
|
|
26
26
|
* regex-JSON fallback.
|
|
27
|
-
* - `callStructured
|
|
28
|
-
* return the input of the supplied `ToolSchema`'s tool,
|
|
29
|
-
*
|
|
27
|
+
* - `callStructured?(...)` is OPTIONAL. When present, it MUST
|
|
28
|
+
* return the input of the supplied `ToolSchema`'s tool, as the model
|
|
29
|
+
* produced it (`unknown`: the caller parses it; ggui#1317), or THROW —
|
|
30
|
+
* never return anything else. It forces the tool
|
|
30
31
|
* where the model allows that; a model that refuses a forced tool
|
|
31
32
|
* (the always-thinking family) is asked for it without forcing, so
|
|
32
33
|
* the call can end with no tool input, and that ending is a throw.
|
|
@@ -60,7 +61,7 @@ export interface LLMCaller {
|
|
|
60
61
|
* this method — consumers detect absence and fall back to regex JSON
|
|
61
62
|
* extraction on the text path.
|
|
62
63
|
*/
|
|
63
|
-
callStructured
|
|
64
|
+
callStructured?(systemPrompt: string, userMessage: string, tool: ToolSchema, maxTokens?: number): Promise<unknown>;
|
|
64
65
|
}
|
|
65
66
|
/**
|
|
66
67
|
* Provider + model selector for factory-style LLM caller
|
package/dist/llm-caller.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"llm-caller.d.ts","sourceRoot":"","sources":["../src/llm-caller.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"llm-caller.d.ts","sourceRoot":"","sources":["../src/llm-caller.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAyCG;AAEH,6DAA6D;AAC7D,MAAM,WAAW,UAAU;IACzB,IAAI,EAAE,MAAM,CAAC;IACb,WAAW,EAAE,MAAM,CAAC;IACpB,YAAY,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACvC;AAED,uFAAuF;AACvF,MAAM,WAAW,SAAS;IACxB;;;OAGG;IACH,IAAI,CACF,YAAY,EAAE,MAAM,EACpB,WAAW,EAAE,MAAM,EACnB,SAAS,CAAC,EAAE,MAAM,GACjB,OAAO,CAAC,MAAM,CAAC,CAAC;IAEnB;;;;;;;OAOG;IACH,cAAc,CAAC,CACb,YAAY,EAAE,MAAM,EACpB,WAAW,EAAE,MAAM,EACnB,IAAI,EAAE,UAAU,EAChB,SAAS,CAAC,EAAE,MAAM,GACjB,OAAO,CAAC,OAAO,CAAC,CAAC;CACrB;AAED;;;;;GAKG;AACH,MAAM,WAAW,eAAe;IAC9B,QAAQ,EAAE,WAAW,GAAG,QAAQ,GAAG,QAAQ,GAAG,YAAY,GAAG,SAAS,CAAC;IACvE,KAAK,EAAE,MAAM,CAAC;IACd,MAAM,CAAC,EAAE,MAAM,CAAC;CACjB"}
|
package/dist/llm-caller.js
CHANGED
|
@@ -24,9 +24,10 @@
|
|
|
24
24
|
* model text. Implementations MUST NOT inject tool-use blocks when
|
|
25
25
|
* the caller didn't request them — the text path is used as a
|
|
26
26
|
* regex-JSON fallback.
|
|
27
|
-
* - `callStructured
|
|
28
|
-
* return the input of the supplied `ToolSchema`'s tool,
|
|
29
|
-
*
|
|
27
|
+
* - `callStructured?(...)` is OPTIONAL. When present, it MUST
|
|
28
|
+
* return the input of the supplied `ToolSchema`'s tool, as the model
|
|
29
|
+
* produced it (`unknown`: the caller parses it; ggui#1317), or THROW —
|
|
30
|
+
* never return anything else. It forces the tool
|
|
30
31
|
* where the model allows that; a model that refuses a forced tool
|
|
31
32
|
* (the always-thinking family) is asked for it without forcing, so
|
|
32
33
|
* the call can end with no tool input, and that ending is a throw.
|
package/dist/llm-rerank.d.ts
CHANGED
|
@@ -1,17 +1,3 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* LLM rerank — Tier-2 precision oracle for the blueprint registry.
|
|
3
|
-
*
|
|
4
|
-
* Given a user's UI request (intent + contract structure) and a set
|
|
5
|
-
* of candidate cached blueprints retrieved by RAG, ask a fast LLM
|
|
6
|
-
* (Haiku 4.5) which candidate (if any) matches. Returns a structured
|
|
7
|
-
* decision so the caller can branch deterministically.
|
|
8
|
-
*
|
|
9
|
-
* This module is the precision half of the blueprint-first
|
|
10
|
-
* architecture: RAG retrieval is high-recall but low-precision (bge-
|
|
11
|
-
* small confuses topic-similar but UI-divergent prompts); the LLM
|
|
12
|
-
* judge restores precision. Combined break-even hit rate is ~10%;
|
|
13
|
-
* realistic workloads observe 30-70%.
|
|
14
|
-
*/
|
|
15
1
|
import type { LLMCaller, ToolSchema } from './llm-caller.js';
|
|
16
2
|
/**
|
|
17
3
|
* One candidate blueprint for the LLM judge to consider.
|
|
@@ -58,11 +44,12 @@ export interface RerankDecision {
|
|
|
58
44
|
*/
|
|
59
45
|
readonly confidence: number;
|
|
60
46
|
/**
|
|
61
|
-
* Free-text reason from the judge
|
|
47
|
+
* Free-text reason from the judge, when it gives one — a judge that
|
|
48
|
+
* decides without prose sends none (ggui#1235). Surface in trace logs so
|
|
62
49
|
* operators can debug "why didn't this hit." Truncate at the
|
|
63
50
|
* persistence boundary if cardinality is a concern.
|
|
64
51
|
*/
|
|
65
|
-
readonly reason
|
|
52
|
+
readonly reason?: string;
|
|
66
53
|
/** Wall-clock latency of the LLM call. */
|
|
67
54
|
readonly latencyMs: number;
|
|
68
55
|
/**
|
|
@@ -98,5 +85,20 @@ declare const RERANK_TOOL: ToolSchema;
|
|
|
98
85
|
export declare function rerankCandidates(deps: {
|
|
99
86
|
readonly llm: LLMCaller;
|
|
100
87
|
}, query: RerankQuery, candidates: readonly RerankCandidate[]): Promise<RerankDecision>;
|
|
88
|
+
/**
|
|
89
|
+
* The judge seam (ggui#1235): one function from a query and its candidates
|
|
90
|
+
* to a {@link RerankDecision}. The matcher takes a judge together with the
|
|
91
|
+
* confidence threshold it was measured on (the pair), so a judge on
|
|
92
|
+
* another scale never meets a cut calibrated for a different one.
|
|
93
|
+
* `matchId: null` is a judge's only decline; `confidence` is the judge's
|
|
94
|
+
* own, compared by the caller against the pair's threshold.
|
|
95
|
+
*/
|
|
96
|
+
export type RerankJudge = (query: RerankQuery, candidates: readonly RerankCandidate[]) => Promise<RerankDecision>;
|
|
97
|
+
/**
|
|
98
|
+
* Today's judge as a {@link RerankJudge}: {@link rerankCandidates} bound to
|
|
99
|
+
* an {@link LLMCaller} — the same prompt, the same tool, the same decision,
|
|
100
|
+
* so a caller that injects nothing else behaves exactly as before.
|
|
101
|
+
*/
|
|
102
|
+
export declare function llmRerankJudge(llm: LLMCaller): RerankJudge;
|
|
101
103
|
export { RERANK_SYSTEM_PROMPT, RERANK_TOOL };
|
|
102
104
|
//# sourceMappingURL=llm-rerank.d.ts.map
|
package/dist/llm-rerank.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"llm-rerank.d.ts","sourceRoot":"","sources":["../src/llm-rerank.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"llm-rerank.d.ts","sourceRoot":"","sources":["../src/llm-rerank.ts"],"names":[],"mappings":"AAeA,OAAO,KAAK,EAAE,SAAS,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAE7D;;;;;;GAMG;AACH,MAAM,WAAW,eAAe;IAC9B,+DAA+D;IAC/D,QAAQ,CAAC,EAAE,EAAE,MAAM,CAAC;IACpB,gEAAgE;IAChE,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;IAC9B;;;;;OAKG;IACH,QAAQ,CAAC,qBAAqB,EAAE,MAAM,CAAC;IACvC;;;;;OAKG;IACH,QAAQ,CAAC,MAAM,CAAC,EAAE,MAAM,CAAC;CAC1B;AAED,0CAA0C;AAC1C,MAAM,WAAW,cAAc;IAC7B;;;OAGG;IACH,QAAQ,CAAC,OAAO,EAAE,MAAM,GAAG,IAAI,CAAC;IAChC;;;;;;;;OAQG;IACH,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B;;;;;OAKG;IACH,QAAQ,CAAC,MAAM,CAAC,EAAE,MAAM,CAAC;IACzB,0CAA0C;IAC1C,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B;;;;;OAKG;IACH,QAAQ,CAAC,SAAS,EAAE;QAAE,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;QAAC,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAA;KAAE,CAAC;CACzE;AAED,8DAA8D;AAC9D,MAAM,WAAW,WAAW;IAC1B,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,QAAQ,CAAC,eAAe,EAAE,MAAM,CAAC;CAClC;AAED,QAAA,MAAM,oBAAoB,03DAQ6U,CAAC;AAExW,QAAA,MAAM,WAAW,EAAE,UA4BlB,CAAC;AAkFF;;;;;;;;;;;GAWG;AACH,wBAAsB,gBAAgB,CACpC,IAAI,EAAE;IAAE,QAAQ,CAAC,GAAG,EAAE,SAAS,CAAA;CAAE,EACjC,KAAK,EAAE,WAAW,EAClB,UAAU,EAAE,SAAS,eAAe,EAAE,GACrC,OAAO,CAAC,cAAc,CAAC,CAwDzB;AAGD;;;;;;;GAOG;AACH,MAAM,MAAM,WAAW,GAAG,CACxB,KAAK,EAAE,WAAW,EAClB,UAAU,EAAE,SAAS,eAAe,EAAE,KACnC,OAAO,CAAC,cAAc,CAAC,CAAC;AAE7B;;;;GAIG;AACH,wBAAgB,cAAc,CAAC,GAAG,EAAE,SAAS,GAAG,WAAW,CAE1D;AAED,OAAO,EAAE,oBAAoB,EAAE,WAAW,EAAE,CAAC"}
|
package/dist/llm-rerank.js
CHANGED
|
@@ -1,3 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* LLM rerank — Tier-2 precision oracle for the blueprint registry.
|
|
3
|
+
*
|
|
4
|
+
* Given a user's UI request (intent + contract structure) and a set
|
|
5
|
+
* of candidate cached blueprints retrieved by RAG, ask a fast LLM
|
|
6
|
+
* (Haiku 4.5) which candidate (if any) matches. Returns a structured
|
|
7
|
+
* decision so the caller can branch deterministically.
|
|
8
|
+
*
|
|
9
|
+
* This module is the precision half of the blueprint-first
|
|
10
|
+
* architecture: RAG retrieval is high-recall but low-precision (bge-
|
|
11
|
+
* small confuses topic-similar but UI-divergent prompts); the LLM
|
|
12
|
+
* judge restores precision. Combined break-even hit rate is ~10%;
|
|
13
|
+
* realistic workloads observe 30-70%.
|
|
14
|
+
*/
|
|
15
|
+
import { MATCHED_INTENT_MAX_CHARS } from '@ggui-ai/protocol';
|
|
1
16
|
const RERANK_SYSTEM_PROMPT = `You match user UI requests against previously-generated UI blueprints. Each blueprint was produced for a past request and stored. Decide whether any candidate belongs to the SAME FAMILY as the current request — i.e. it would serve as a reasonable starting point that the requester can refine, not a pixel-exact replica.
|
|
2
17
|
|
|
3
18
|
MATCH means the candidate is the same intended user task AND the same broad UI shape (component types, layout pattern). A candidate still MATCHES when the current request adds or omits fields, slots, or actions relative to the cached blueprint — a superset, a subset, or an overlapping wire surface all still match. Added or omitted fields/slots/actions DO NOT block a match and are NOT yours to judge: those wire-surface deltas are reconciled and reported to the agent separately, after you decide. Judge similarity of task and shape, never coverage of fields.
|
|
@@ -42,7 +57,11 @@ function buildUserMessage(query, candidates) {
|
|
|
42
57
|
for (const c of candidates) {
|
|
43
58
|
lines.push('---');
|
|
44
59
|
lines.push(` id: ${c.id}`);
|
|
45
|
-
|
|
60
|
+
// The cap is the protocol's `MATCHED_INTENT_MAX_CHARS`: the same
|
|
61
|
+
// string the judge reads here is what a judged hit hands back to the
|
|
62
|
+
// agent as `blueprintMeta.matchedIntent` (ggui#1336), so the two cuts
|
|
63
|
+
// are one constant, not two literals that happen to agree.
|
|
64
|
+
lines.push(` intent: ${truncate(c.cachedIntent, MATCHED_INTENT_MAX_CHARS)}`);
|
|
46
65
|
lines.push(` contract: ${c.cachedContractSummary}`);
|
|
47
66
|
if (typeof c.cosine === 'number') {
|
|
48
67
|
lines.push(` cosine: ${c.cosine.toFixed(3)}`);
|
|
@@ -153,5 +172,12 @@ export async function rerankCandidates(deps, query, candidates) {
|
|
|
153
172
|
tokenCost: { input: 0, output: 0 },
|
|
154
173
|
};
|
|
155
174
|
}
|
|
156
|
-
|
|
175
|
+
/**
|
|
176
|
+
* Today's judge as a {@link RerankJudge}: {@link rerankCandidates} bound to
|
|
177
|
+
* an {@link LLMCaller} — the same prompt, the same tool, the same decision,
|
|
178
|
+
* so a caller that injects nothing else behaves exactly as before.
|
|
179
|
+
*/
|
|
180
|
+
export function llmRerankJudge(llm) {
|
|
181
|
+
return (query, candidates) => rerankCandidates({ llm }, query, candidates);
|
|
182
|
+
}
|
|
157
183
|
export { RERANK_SYSTEM_PROMPT, RERANK_TOOL };
|
|
@@ -65,7 +65,10 @@ async function main() {
|
|
|
65
65
|
const usage = getTokenUsage();
|
|
66
66
|
const totalCost = usage.input * HAIKU_4_5_PRICE_INPUT_PER_TOKEN +
|
|
67
67
|
usage.output * HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN;
|
|
68
|
-
const callsMade = report.outcomes.filter(
|
|
68
|
+
const callsMade = report.outcomes.filter(
|
|
69
|
+
// ggui#1235 — `reason` is optional on the seam; a judge with no prose
|
|
70
|
+
// still made a call.
|
|
71
|
+
(o) => !/short-circuited/.test(o.decision.reason ?? '')).length;
|
|
69
72
|
const costPerCall = callsMade === 0 ? 0 : totalCost / callsMade;
|
|
70
73
|
process.stdout.write(`Tokens: input=${usage.input} · output=${usage.output}\n`);
|
|
71
74
|
process.stdout.write(`Cost: total=$${totalCost.toFixed(4)} · per-call=$${costPerCall.toFixed(4)}\n`);
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ggui-ai/negotiator",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.24.0",
|
|
4
4
|
"description": "Contract-synthesis + match-judge engine for ggui's handshake. Synthesizes or repairs a conforming DataContract from an agent's draft, judges blueprint-match candidates for reuse, and validates contract structure + novelty — the primitives composed by decideHandshake in @ggui-ai/mcp-server-handlers. Deployment-agnostic: concrete embedding and vector-store bindings plug in via the storage interfaces from @ggui-ai/mcp-server-core.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"keywords": [
|
|
@@ -47,8 +47,8 @@
|
|
|
47
47
|
}
|
|
48
48
|
},
|
|
49
49
|
"dependencies": {
|
|
50
|
-
"@ggui-ai/mcp-server-core": "0.
|
|
51
|
-
"@ggui-ai/protocol": "0.
|
|
50
|
+
"@ggui-ai/mcp-server-core": "0.24.0",
|
|
51
|
+
"@ggui-ai/protocol": "0.24.0"
|
|
52
52
|
},
|
|
53
53
|
"devDependencies": {
|
|
54
54
|
"@types/node": "^24.0.0",
|
package/src/index.ts
CHANGED
|
@@ -27,10 +27,11 @@
|
|
|
27
27
|
|
|
28
28
|
export { hashContract, buildVariant } from './contract-hash.js';
|
|
29
29
|
export type { LLMCaller, LLMCallerConfig, ToolSchema } from './llm-caller.js';
|
|
30
|
-
export { rerankCandidates } from './llm-rerank.js';
|
|
30
|
+
export { llmRerankJudge, rerankCandidates } from './llm-rerank.js';
|
|
31
31
|
export type {
|
|
32
32
|
RerankCandidate,
|
|
33
33
|
RerankDecision,
|
|
34
|
+
RerankJudge,
|
|
34
35
|
RerankQuery,
|
|
35
36
|
} from './llm-rerank.js';
|
|
36
37
|
export { synthesizeContract } from './synthesize-contract.js';
|
package/src/llm-caller.ts
CHANGED
|
@@ -24,9 +24,10 @@
|
|
|
24
24
|
* model text. Implementations MUST NOT inject tool-use blocks when
|
|
25
25
|
* the caller didn't request them — the text path is used as a
|
|
26
26
|
* regex-JSON fallback.
|
|
27
|
-
* - `callStructured
|
|
28
|
-
* return the input of the supplied `ToolSchema`'s tool,
|
|
29
|
-
*
|
|
27
|
+
* - `callStructured?(...)` is OPTIONAL. When present, it MUST
|
|
28
|
+
* return the input of the supplied `ToolSchema`'s tool, as the model
|
|
29
|
+
* produced it (`unknown`: the caller parses it; ggui#1317), or THROW —
|
|
30
|
+
* never return anything else. It forces the tool
|
|
30
31
|
* where the model allows that; a model that refuses a forced tool
|
|
31
32
|
* (the always-thinking family) is asked for it without forcing, so
|
|
32
33
|
* the call can end with no tool input, and that ending is a throw.
|
|
@@ -67,12 +68,12 @@ export interface LLMCaller {
|
|
|
67
68
|
* this method — consumers detect absence and fall back to regex JSON
|
|
68
69
|
* extraction on the text path.
|
|
69
70
|
*/
|
|
70
|
-
callStructured
|
|
71
|
+
callStructured?(
|
|
71
72
|
systemPrompt: string,
|
|
72
73
|
userMessage: string,
|
|
73
74
|
tool: ToolSchema,
|
|
74
75
|
maxTokens?: number,
|
|
75
|
-
): Promise<
|
|
76
|
+
): Promise<unknown>;
|
|
76
77
|
}
|
|
77
78
|
|
|
78
79
|
/**
|
package/src/llm-rerank.ts
CHANGED
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
* judge restores precision. Combined break-even hit rate is ~10%;
|
|
13
13
|
* realistic workloads observe 30-70%.
|
|
14
14
|
*/
|
|
15
|
+
import { MATCHED_INTENT_MAX_CHARS } from '@ggui-ai/protocol';
|
|
15
16
|
import type { LLMCaller, ToolSchema } from './llm-caller.js';
|
|
16
17
|
|
|
17
18
|
/**
|
|
@@ -60,11 +61,12 @@ export interface RerankDecision {
|
|
|
60
61
|
*/
|
|
61
62
|
readonly confidence: number;
|
|
62
63
|
/**
|
|
63
|
-
* Free-text reason from the judge
|
|
64
|
+
* Free-text reason from the judge, when it gives one — a judge that
|
|
65
|
+
* decides without prose sends none (ggui#1235). Surface in trace logs so
|
|
64
66
|
* operators can debug "why didn't this hit." Truncate at the
|
|
65
67
|
* persistence boundary if cardinality is a concern.
|
|
66
68
|
*/
|
|
67
|
-
readonly reason
|
|
69
|
+
readonly reason?: string;
|
|
68
70
|
/** Wall-clock latency of the LLM call. */
|
|
69
71
|
readonly latencyMs: number;
|
|
70
72
|
/**
|
|
@@ -135,7 +137,11 @@ function buildUserMessage(
|
|
|
135
137
|
for (const c of candidates) {
|
|
136
138
|
lines.push('---');
|
|
137
139
|
lines.push(` id: ${c.id}`);
|
|
138
|
-
|
|
140
|
+
// The cap is the protocol's `MATCHED_INTENT_MAX_CHARS`: the same
|
|
141
|
+
// string the judge reads here is what a judged hit hands back to the
|
|
142
|
+
// agent as `blueprintMeta.matchedIntent` (ggui#1336), so the two cuts
|
|
143
|
+
// are one constant, not two literals that happen to agree.
|
|
144
|
+
lines.push(` intent: ${truncate(c.cachedIntent, MATCHED_INTENT_MAX_CHARS)}`);
|
|
139
145
|
lines.push(` contract: ${c.cachedContractSummary}`);
|
|
140
146
|
if (typeof c.cosine === 'number') {
|
|
141
147
|
lines.push(` cosine: ${c.cosine.toFixed(3)}`);
|
|
@@ -153,6 +159,7 @@ function truncate(text: string, max: number): string {
|
|
|
153
159
|
return `${text.slice(0, max - 1)}…`;
|
|
154
160
|
}
|
|
155
161
|
|
|
162
|
+
/** The rerank tool's input once {@link parseToolInput} has validated it. */
|
|
156
163
|
interface RerankToolInput {
|
|
157
164
|
matchId: string | null;
|
|
158
165
|
confidence: number;
|
|
@@ -169,7 +176,7 @@ function clampConfidence(value: unknown): number {
|
|
|
169
176
|
function parseToolInput(
|
|
170
177
|
raw: unknown,
|
|
171
178
|
candidateIds: ReadonlySet<string>,
|
|
172
|
-
):
|
|
179
|
+
): RerankToolInput {
|
|
173
180
|
if (raw === null || typeof raw !== 'object') {
|
|
174
181
|
return { matchId: null, confidence: 0, reason: 'parse-failed: non-object tool input' };
|
|
175
182
|
}
|
|
@@ -241,7 +248,7 @@ export async function rerankCandidates(
|
|
|
241
248
|
|
|
242
249
|
let toolInput: unknown;
|
|
243
250
|
try {
|
|
244
|
-
toolInput = await deps.llm.callStructured
|
|
251
|
+
toolInput = await deps.llm.callStructured(
|
|
245
252
|
RERANK_SYSTEM_PROMPT,
|
|
246
253
|
userMessage,
|
|
247
254
|
RERANK_TOOL,
|
|
@@ -272,4 +279,26 @@ export async function rerankCandidates(
|
|
|
272
279
|
}
|
|
273
280
|
|
|
274
281
|
// Re-exports for the eval harness — keep public surface explicit.
|
|
282
|
+
/**
|
|
283
|
+
* The judge seam (ggui#1235): one function from a query and its candidates
|
|
284
|
+
* to a {@link RerankDecision}. The matcher takes a judge together with the
|
|
285
|
+
* confidence threshold it was measured on (the pair), so a judge on
|
|
286
|
+
* another scale never meets a cut calibrated for a different one.
|
|
287
|
+
* `matchId: null` is a judge's only decline; `confidence` is the judge's
|
|
288
|
+
* own, compared by the caller against the pair's threshold.
|
|
289
|
+
*/
|
|
290
|
+
export type RerankJudge = (
|
|
291
|
+
query: RerankQuery,
|
|
292
|
+
candidates: readonly RerankCandidate[],
|
|
293
|
+
) => Promise<RerankDecision>;
|
|
294
|
+
|
|
295
|
+
/**
|
|
296
|
+
* Today's judge as a {@link RerankJudge}: {@link rerankCandidates} bound to
|
|
297
|
+
* an {@link LLMCaller} — the same prompt, the same tool, the same decision,
|
|
298
|
+
* so a caller that injects nothing else behaves exactly as before.
|
|
299
|
+
*/
|
|
300
|
+
export function llmRerankJudge(llm: LLMCaller): RerankJudge {
|
|
301
|
+
return (query, candidates) => rerankCandidates({ llm }, query, candidates);
|
|
302
|
+
}
|
|
303
|
+
|
|
275
304
|
export { RERANK_SYSTEM_PROMPT, RERANK_TOOL };
|
|
@@ -79,7 +79,9 @@ async function main(): Promise<void> {
|
|
|
79
79
|
usage.input * HAIKU_4_5_PRICE_INPUT_PER_TOKEN +
|
|
80
80
|
usage.output * HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN;
|
|
81
81
|
const callsMade = report.outcomes.filter(
|
|
82
|
-
|
|
82
|
+
// ggui#1235 — `reason` is optional on the seam; a judge with no prose
|
|
83
|
+
// still made a call.
|
|
84
|
+
(o) => !/short-circuited/.test(o.decision.reason ?? ''),
|
|
83
85
|
).length;
|
|
84
86
|
const costPerCall = callsMade === 0 ? 0 : totalCost / callsMade;
|
|
85
87
|
process.stdout.write(
|
|
@@ -99,12 +99,12 @@ export function buildAnthropicLlmCaller(
|
|
|
99
99
|
'negotiator dev caller: text-mode not exercised — use callStructured',
|
|
100
100
|
);
|
|
101
101
|
},
|
|
102
|
-
async callStructured
|
|
102
|
+
async callStructured(
|
|
103
103
|
systemPrompt: string,
|
|
104
104
|
userMessage: string,
|
|
105
105
|
tool: ToolSchema,
|
|
106
106
|
maxTokens?: number,
|
|
107
|
-
): Promise<
|
|
107
|
+
): Promise<unknown> {
|
|
108
108
|
// `temperature` deprecated on Haiku 4.5+ — Anthropic rejects with
|
|
109
109
|
// HTTP 400. Residual stochasticity stays bounded via canonical-key
|
|
110
110
|
// normalization downstream.
|
|
@@ -162,7 +162,7 @@ export function buildAnthropicLlmCaller(
|
|
|
162
162
|
`anthropic: no tool_use block in response (stop_reason=${json.stop_reason ?? 'unknown'})`,
|
|
163
163
|
);
|
|
164
164
|
}
|
|
165
|
-
return toolBlock.input
|
|
165
|
+
return toolBlock.input;
|
|
166
166
|
},
|
|
167
167
|
};
|
|
168
168
|
}
|
|
@@ -634,7 +634,7 @@ async function callSynthesizeTool(
|
|
|
634
634
|
user: string,
|
|
635
635
|
): Promise<unknown> {
|
|
636
636
|
if (typeof llm.callStructured === 'function') {
|
|
637
|
-
return llm.callStructured
|
|
637
|
+
return llm.callStructured(
|
|
638
638
|
system,
|
|
639
639
|
user,
|
|
640
640
|
SYNTHESIZE_TOOL,
|