@ggui-ai/negotiator 0.1.0-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +49 -0
- package/dist/contract-hash.d.ts +54 -0
- package/dist/contract-hash.d.ts.map +1 -0
- package/dist/contract-hash.js +96 -0
- package/dist/contract-validators.d.ts +171 -0
- package/dist/contract-validators.d.ts.map +1 -0
- package/dist/contract-validators.js +478 -0
- package/dist/decision-input.d.ts +48 -0
- package/dist/decision-input.d.ts.map +1 -0
- package/dist/decision-input.js +14 -0
- package/dist/decision.d.ts +54 -0
- package/dist/decision.d.ts.map +1 -0
- package/dist/decision.js +500 -0
- package/dist/index.d.ts +36 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +25 -0
- package/dist/intent.d.ts +22 -0
- package/dist/intent.d.ts.map +1 -0
- package/dist/intent.js +28 -0
- package/dist/llm-caller.d.ts +70 -0
- package/dist/llm-caller.d.ts.map +1 -0
- package/dist/llm-caller.js +38 -0
- package/dist/llm-rerank.d.ts +101 -0
- package/dist/llm-rerank.d.ts.map +1 -0
- package/dist/llm-rerank.js +178 -0
- package/dist/negotiate.d.ts +141 -0
- package/dist/negotiate.d.ts.map +1 -0
- package/dist/negotiate.js +161 -0
- package/dist/normalize-schema.d.ts +22 -0
- package/dist/normalize-schema.d.ts.map +1 -0
- package/dist/normalize-schema.js +191 -0
- package/dist/pure.d.ts +30 -0
- package/dist/pure.d.ts.map +1 -0
- package/dist/pure.js +43 -0
- package/dist/rag-search.d.ts +73 -0
- package/dist/rag-search.d.ts.map +1 -0
- package/dist/rag-search.js +192 -0
- package/dist/rerank-eval/pairs.d.ts +28 -0
- package/dist/rerank-eval/pairs.d.ts.map +1 -0
- package/dist/rerank-eval/pairs.js +531 -0
- package/dist/rerank-eval/run-probe-cli.d.ts +3 -0
- package/dist/rerank-eval/run-probe-cli.d.ts.map +1 -0
- package/dist/rerank-eval/run-probe-cli.js +146 -0
- package/dist/rerank-eval/run-probe.d.ts +68 -0
- package/dist/rerank-eval/run-probe.d.ts.map +1 -0
- package/dist/rerank-eval/run-probe.js +113 -0
- package/dist/session.d.ts +42 -0
- package/dist/session.d.ts.map +1 -0
- package/dist/session.js +21 -0
- package/dist/suggestion.d.ts +38 -0
- package/dist/suggestion.d.ts.map +1 -0
- package/dist/suggestion.js +47 -0
- package/dist/synth-bench/corpus.d.ts +106 -0
- package/dist/synth-bench/corpus.d.ts.map +1 -0
- package/dist/synth-bench/corpus.js +994 -0
- package/dist/synth-bench/run-bench-cli.d.ts +3 -0
- package/dist/synth-bench/run-bench-cli.d.ts.map +1 -0
- package/dist/synth-bench/run-bench-cli.js +181 -0
- package/dist/synth-bench/run-bench.d.ts +101 -0
- package/dist/synth-bench/run-bench.d.ts.map +1 -0
- package/dist/synth-bench/run-bench.js +374 -0
- package/dist/synthesize-contract.d.ts +131 -0
- package/dist/synthesize-contract.d.ts.map +1 -0
- package/dist/synthesize-contract.js +948 -0
- package/dist/types.d.ts +30 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +13 -0
- package/package.json +74 -0
- package/src/contract-hash.ts +102 -0
- package/src/contract-validators.ts +604 -0
- package/src/decision-input.ts +49 -0
- package/src/decision.ts +581 -0
- package/src/index.ts +63 -0
- package/src/intent.ts +37 -0
- package/src/llm-caller.ts +82 -0
- package/src/llm-rerank.ts +280 -0
- package/src/negotiate.ts +312 -0
- package/src/normalize-schema.ts +193 -0
- package/src/pure.ts +46 -0
- package/src/rag-search.ts +274 -0
- package/src/rerank-eval/pairs.ts +624 -0
- package/src/rerank-eval/run-probe-cli.ts +197 -0
- package/src/rerank-eval/run-probe.ts +198 -0
- package/src/session.ts +41 -0
- package/src/suggestion.ts +73 -0
- package/src/synth-bench/corpus.ts +1126 -0
- package/src/synth-bench/run-bench-cli.ts +237 -0
- package/src/synth-bench/run-bench.ts +525 -0
- package/src/synthesize-contract.ts +1161 -0
- package/src/types.ts +31 -0
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Rerank quality probe.
|
|
3
|
+
*
|
|
4
|
+
* Runs the configured `LLMCaller` over the {@link EVAL_PAIRS} eval
|
|
5
|
+
* set and returns precision@1 + adversarial false-positive +
|
|
6
|
+
* latency p95. The caller compares these against the quality gates
|
|
7
|
+
* (precision, adversarial false-positive rate, latency, cost)
|
|
8
|
+
* reported by the probe CLI.
|
|
9
|
+
*
|
|
10
|
+
* Eval-only — not exported from the package index.
|
|
11
|
+
*/
|
|
12
|
+
import { type RerankDecision } from '../llm-rerank.js';
|
|
13
|
+
import type { LLMCaller } from '../llm-caller.js';
|
|
14
|
+
import { type EvalPair } from './pairs.js';
|
|
15
|
+
export interface ProbeOutcome {
|
|
16
|
+
readonly pair: EvalPair;
|
|
17
|
+
readonly decision: RerankDecision;
|
|
18
|
+
/** Confidence threshold applied to compute correctness. */
|
|
19
|
+
readonly threshold: number;
|
|
20
|
+
/** Predicted matchId AFTER threshold gate (null if confidence below). */
|
|
21
|
+
readonly predictedMatchId: string | null;
|
|
22
|
+
/** Was the prediction correct vs gold? */
|
|
23
|
+
readonly correct: boolean;
|
|
24
|
+
}
|
|
25
|
+
export interface ProbeReport {
|
|
26
|
+
readonly outcomes: readonly ProbeOutcome[];
|
|
27
|
+
readonly totals: {
|
|
28
|
+
readonly all: number;
|
|
29
|
+
readonly correct: number;
|
|
30
|
+
readonly precision: number;
|
|
31
|
+
};
|
|
32
|
+
readonly byKind: Readonly<{
|
|
33
|
+
[K in EvalPair['kind']]: {
|
|
34
|
+
readonly all: number;
|
|
35
|
+
readonly correct: number;
|
|
36
|
+
readonly precision: number;
|
|
37
|
+
};
|
|
38
|
+
}>;
|
|
39
|
+
readonly adversarialFalsePositiveRate: number;
|
|
40
|
+
readonly latency: {
|
|
41
|
+
readonly p50Ms: number;
|
|
42
|
+
readonly p95Ms: number;
|
|
43
|
+
};
|
|
44
|
+
readonly threshold: number;
|
|
45
|
+
}
|
|
46
|
+
export interface RunProbeOptions {
|
|
47
|
+
/** Use a subset for smoke; default = all 25 pairs. */
|
|
48
|
+
readonly limit?: number;
|
|
49
|
+
/** Confidence threshold for considering a non-null match a hit. Default 0.6. */
|
|
50
|
+
readonly threshold?: number;
|
|
51
|
+
/**
|
|
52
|
+
* Optional progress callback. Fired AFTER each pair completes — the
|
|
53
|
+
* CLI uses this to print a per-pair line so the operator sees live
|
|
54
|
+
* progress instead of waiting for the full sweep to finish.
|
|
55
|
+
*/
|
|
56
|
+
readonly onProgress?: (outcome: ProbeOutcome, index: number, total: number) => void;
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* Run the probe against `pairs` (defaults to {@link EVAL_PAIRS}).
|
|
60
|
+
* Sequential by design — small N (~25), and concurrent calls would
|
|
61
|
+
* complicate latency measurement without changing the gate decision.
|
|
62
|
+
*/
|
|
63
|
+
export declare function runProbe(deps: {
|
|
64
|
+
readonly llm: LLMCaller;
|
|
65
|
+
}, options?: RunProbeOptions, pairs?: readonly EvalPair[]): Promise<ProbeReport>;
|
|
66
|
+
/** Pretty-print a {@link ProbeReport} for terminal output. */
|
|
67
|
+
export declare function formatReport(report: ProbeReport): string;
|
|
68
|
+
//# sourceMappingURL=run-probe.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"run-probe.d.ts","sourceRoot":"","sources":["../../src/rerank-eval/run-probe.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AACH,OAAO,EAAoB,KAAK,cAAc,EAAE,MAAM,kBAAkB,CAAC;AACzE,OAAO,KAAK,EAAE,SAAS,EAAE,MAAM,kBAAkB,CAAC;AAClD,OAAO,EAAc,KAAK,QAAQ,EAAE,MAAM,YAAY,CAAC;AAEvD,MAAM,WAAW,YAAY;IAC3B,QAAQ,CAAC,IAAI,EAAE,QAAQ,CAAC;IACxB,QAAQ,CAAC,QAAQ,EAAE,cAAc,CAAC;IAClC,2DAA2D;IAC3D,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,yEAAyE;IACzE,QAAQ,CAAC,gBAAgB,EAAE,MAAM,GAAG,IAAI,CAAC;IACzC,0CAA0C;IAC1C,QAAQ,CAAC,OAAO,EAAE,OAAO,CAAC;CAC3B;AAED,MAAM,WAAW,WAAW;IAC1B,QAAQ,CAAC,QAAQ,EAAE,SAAS,YAAY,EAAE,CAAC;IAC3C,QAAQ,CAAC,MAAM,EAAE;QACf,QAAQ,CAAC,GAAG,EAAE,MAAM,CAAC;QACrB,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;QACzB,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;KAC5B,CAAC;IACF,QAAQ,CAAC,MAAM,EAAE,QAAQ,CAAC;SACvB,CAAC,IAAI,QAAQ,CAAC,MAAM,CAAC,GAAG;YACvB,QAAQ,CAAC,GAAG,EAAE,MAAM,CAAC;YACrB,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;YACzB,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;SAC5B;KACF,CAAC,CAAC;IACH,QAAQ,CAAC,4BAA4B,EAAE,MAAM,CAAC;IAC9C,QAAQ,CAAC,OAAO,EAAE;QAAE,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;QAAC,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAA;KAAE,CAAC;IACrE,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;CAC5B;AAED,MAAM,WAAW,eAAe;IAC9B,sDAAsD;IACtD,QAAQ,CAAC,KAAK,CAAC,EAAE,MAAM,CAAC;IACxB,gFAAgF;IAChF,QAAQ,CAAC,SAAS,CAAC,EAAE,MAAM,CAAC;IAC5B;;;;OAIG;IACH,QAAQ,CAAC,UAAU,CAAC,EAAE,CAAC,OAAO,EAAE,YAAY,EAAE,KAAK,EAAE,MAAM,EAAE,KAAK,EAAE,MAAM,KAAK,IAAI,CAAC;CACrF;AAED;;;;GAIG;AACH,wBAAsB,QAAQ,CAC5B,IAAI,EAAE;IAAE,QAAQ,CAAC,GAAG,EAAE,SAAS,CAAA;CAAE,EACjC,OAAO,GAAE,eAAoB,EAC7B,KAAK,GAAE,SAAS,QAAQ,EAAe,GACtC,OAAO,CAAC,WAAW,CAAC,CAyDtB;AAsBD,8DAA8D;AAC9D,wBAAgB,YAAY,CAAC,MAAM,EAAE,WAAW,GAAG,MAAM,CAkDxD"}
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Rerank quality probe.
|
|
3
|
+
*
|
|
4
|
+
* Runs the configured `LLMCaller` over the {@link EVAL_PAIRS} eval
|
|
5
|
+
* set and returns precision@1 + adversarial false-positive +
|
|
6
|
+
* latency p95. The caller compares these against the quality gates
|
|
7
|
+
* (precision, adversarial false-positive rate, latency, cost)
|
|
8
|
+
* reported by the probe CLI.
|
|
9
|
+
*
|
|
10
|
+
* Eval-only — not exported from the package index.
|
|
11
|
+
*/
|
|
12
|
+
import { rerankCandidates } from '../llm-rerank.js';
|
|
13
|
+
import { EVAL_PAIRS } from './pairs.js';
|
|
14
|
+
/**
|
|
15
|
+
* Run the probe against `pairs` (defaults to {@link EVAL_PAIRS}).
|
|
16
|
+
* Sequential by design — small N (~25), and concurrent calls would
|
|
17
|
+
* complicate latency measurement without changing the gate decision.
|
|
18
|
+
*/
|
|
19
|
+
export async function runProbe(deps, options = {}, pairs = EVAL_PAIRS) {
|
|
20
|
+
const threshold = options.threshold ?? 0.6;
|
|
21
|
+
const subset = options.limit ? pairs.slice(0, options.limit) : pairs;
|
|
22
|
+
const outcomes = [];
|
|
23
|
+
for (let i = 0; i < subset.length; i++) {
|
|
24
|
+
const pair = subset[i];
|
|
25
|
+
const decision = await rerankCandidates(deps, pair.query, pair.candidates);
|
|
26
|
+
const predictedMatchId = decision.confidence >= threshold ? decision.matchId : null;
|
|
27
|
+
const correct = predictedMatchId === pair.goldMatchId;
|
|
28
|
+
const outcome = {
|
|
29
|
+
pair,
|
|
30
|
+
decision,
|
|
31
|
+
threshold,
|
|
32
|
+
predictedMatchId,
|
|
33
|
+
correct,
|
|
34
|
+
};
|
|
35
|
+
outcomes.push(outcome);
|
|
36
|
+
options.onProgress?.(outcome, i, subset.length);
|
|
37
|
+
}
|
|
38
|
+
const totals = aggregate(outcomes);
|
|
39
|
+
const byKind = {
|
|
40
|
+
'should-match': aggregate(outcomes.filter((o) => o.pair.kind === 'should-match')),
|
|
41
|
+
'no-match': aggregate(outcomes.filter((o) => o.pair.kind === 'no-match')),
|
|
42
|
+
adversarial: aggregate(outcomes.filter((o) => o.pair.kind === 'adversarial')),
|
|
43
|
+
};
|
|
44
|
+
// Adversarial false-positive: judge accepted a candidate when gold
|
|
45
|
+
// says null. Compute as `(adversarial-pairs-with-prediction-when-
|
|
46
|
+
// gold-was-null) / (adversarial-pairs-with-gold-null)`.
|
|
47
|
+
const adversarial = outcomes.filter((o) => o.pair.kind === 'adversarial');
|
|
48
|
+
const adversarialNullGold = adversarial.filter((o) => o.pair.goldMatchId === null);
|
|
49
|
+
const adversarialFalsePositives = adversarialNullGold.filter((o) => o.predictedMatchId !== null);
|
|
50
|
+
const adversarialFalsePositiveRate = adversarialNullGold.length === 0
|
|
51
|
+
? 0
|
|
52
|
+
: adversarialFalsePositives.length / adversarialNullGold.length;
|
|
53
|
+
const latencies = outcomes
|
|
54
|
+
.map((o) => o.decision.latencyMs)
|
|
55
|
+
.sort((a, b) => a - b);
|
|
56
|
+
const latency = {
|
|
57
|
+
p50Ms: percentile(latencies, 0.5),
|
|
58
|
+
p95Ms: percentile(latencies, 0.95),
|
|
59
|
+
};
|
|
60
|
+
return {
|
|
61
|
+
outcomes,
|
|
62
|
+
totals,
|
|
63
|
+
byKind,
|
|
64
|
+
adversarialFalsePositiveRate,
|
|
65
|
+
latency,
|
|
66
|
+
threshold,
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
function aggregate(outcomes) {
|
|
70
|
+
const all = outcomes.length;
|
|
71
|
+
const correct = outcomes.filter((o) => o.correct).length;
|
|
72
|
+
const precision = all === 0 ? 0 : correct / all;
|
|
73
|
+
return { all, correct, precision };
|
|
74
|
+
}
|
|
75
|
+
function percentile(sorted, p) {
|
|
76
|
+
if (sorted.length === 0)
|
|
77
|
+
return 0;
|
|
78
|
+
const idx = Math.min(sorted.length - 1, Math.floor(p * sorted.length));
|
|
79
|
+
return sorted[idx] ?? 0;
|
|
80
|
+
}
|
|
81
|
+
/** Pretty-print a {@link ProbeReport} for terminal output. */
|
|
82
|
+
export function formatReport(report) {
|
|
83
|
+
const lines = [];
|
|
84
|
+
lines.push('━━━ rerank quality probe ━━━');
|
|
85
|
+
lines.push('');
|
|
86
|
+
lines.push(`Threshold: confidence ≥ ${report.threshold}`);
|
|
87
|
+
lines.push('');
|
|
88
|
+
lines.push('Precision by kind:');
|
|
89
|
+
for (const [kind, stats] of Object.entries(report.byKind)) {
|
|
90
|
+
const pct = (stats.precision * 100).toFixed(1);
|
|
91
|
+
lines.push(` ${kind.padEnd(14)} ${stats.correct}/${stats.all} (${pct}%)`);
|
|
92
|
+
}
|
|
93
|
+
lines.push('');
|
|
94
|
+
lines.push(`Overall: ${report.totals.correct}/${report.totals.all} (${(report.totals.precision * 100).toFixed(1)}%)`);
|
|
95
|
+
lines.push(`Adversarial FP: ${(report.adversarialFalsePositiveRate * 100).toFixed(1)}%`);
|
|
96
|
+
lines.push(`Latency: p50=${report.latency.p50Ms}ms · p95=${report.latency.p95Ms}ms`);
|
|
97
|
+
lines.push('');
|
|
98
|
+
lines.push('Gates:');
|
|
99
|
+
lines.push(` G1 precision@1 ≥ 85% → ${report.totals.precision >= 0.85 ? 'PASS' : 'FAIL'} (${(report.totals.precision * 100).toFixed(1)}%)`);
|
|
100
|
+
lines.push(` G2 adversarial FP ≤ 5% → ${report.adversarialFalsePositiveRate <= 0.05 ? 'PASS' : 'FAIL'} (${(report.adversarialFalsePositiveRate * 100).toFixed(1)}%)`);
|
|
101
|
+
lines.push(` G3 latency p95 ≤ 600ms → ${report.latency.p95Ms <= 600 ? 'PASS' : 'FAIL'} (p95=${report.latency.p95Ms}ms)`);
|
|
102
|
+
// Per-pair detail for failed predictions
|
|
103
|
+
const failed = report.outcomes.filter((o) => !o.correct);
|
|
104
|
+
if (failed.length > 0) {
|
|
105
|
+
lines.push('');
|
|
106
|
+
lines.push('Failed predictions:');
|
|
107
|
+
for (const o of failed) {
|
|
108
|
+
lines.push(` [${o.pair.kind}] ${o.pair.id}: predicted=${o.predictedMatchId ?? 'null'}, gold=${o.pair.goldMatchId ?? 'null'}, conf=${o.decision.confidence.toFixed(2)}`);
|
|
109
|
+
lines.push(` reason: ${o.decision.reason}`);
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
return lines.join('\n');
|
|
113
|
+
}
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `SessionState` — what the decision engine sees about the live
|
|
3
|
+
* session at the moment of a negotiation call.
|
|
4
|
+
*
|
|
5
|
+
* Captures the UI stack (pages previously pushed into this
|
|
6
|
+
* session), recent conversation, and optional interface context
|
|
7
|
+
* (viewport, device class). Consumed by `NegotiatorDecisionInput`
|
|
8
|
+
* to drive `create / update / compose / replace` decisions — the
|
|
9
|
+
* stack tells the engine whether something already on screen can
|
|
10
|
+
* absorb this push as an `update`, and the conversation history
|
|
11
|
+
* feeds the decision LLM.
|
|
12
|
+
*
|
|
13
|
+
* Kept in `@ggui-ai/negotiator` (not `mcp-server-core`): this is
|
|
14
|
+
* the decision engine's input shape, not a storage seam. An MCP
|
|
15
|
+
* server implementer binding against the public `Negotiator`
|
|
16
|
+
* interface sees `NegotiatorInput` / `NegotiatorResult` — never
|
|
17
|
+
* this internal input shape. Community adapters that want to
|
|
18
|
+
* call `makeDecision` directly (bypassing the `Negotiator` wrapper)
|
|
19
|
+
* import this type.
|
|
20
|
+
*/
|
|
21
|
+
import type { DataContract, InterfaceContext } from '@ggui-ai/protocol';
|
|
22
|
+
/**
|
|
23
|
+
* One entry on the session's UI stack — a page the agent previously
|
|
24
|
+
* pushed. Minimal shape: every field the decision engine actually
|
|
25
|
+
* reads to decide reuse vs. create.
|
|
26
|
+
*/
|
|
27
|
+
export interface SessionStackEntry {
|
|
28
|
+
id: string;
|
|
29
|
+
prompt?: string;
|
|
30
|
+
contract?: DataContract;
|
|
31
|
+
componentCode: string;
|
|
32
|
+
}
|
|
33
|
+
/** Current state of a session — stack + conversation history. */
|
|
34
|
+
export interface SessionState {
|
|
35
|
+
stack: SessionStackEntry[];
|
|
36
|
+
conversationHistory: Array<{
|
|
37
|
+
role: 'user' | 'agent';
|
|
38
|
+
content: string;
|
|
39
|
+
}>;
|
|
40
|
+
interfaceContext?: InterfaceContext;
|
|
41
|
+
}
|
|
42
|
+
//# sourceMappingURL=session.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"session.d.ts","sourceRoot":"","sources":["../src/session.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH,OAAO,KAAK,EAAE,YAAY,EAAE,gBAAgB,EAAE,MAAM,mBAAmB,CAAC;AAExE;;;;GAIG;AACH,MAAM,WAAW,iBAAiB;IAChC,EAAE,EAAE,MAAM,CAAC;IACX,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,QAAQ,CAAC,EAAE,YAAY,CAAC;IACxB,aAAa,EAAE,MAAM,CAAC;CACvB;AAED,iEAAiE;AACjE,MAAM,WAAW,YAAY;IAC3B,KAAK,EAAE,iBAAiB,EAAE,CAAC;IAC3B,mBAAmB,EAAE,KAAK,CAAC;QAAE,IAAI,EAAE,MAAM,GAAG,OAAO,CAAC;QAAC,OAAO,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC;IACxE,gBAAgB,CAAC,EAAE,gBAAgB,CAAC;CACrC"}
|
package/dist/session.js
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `SessionState` — what the decision engine sees about the live
|
|
3
|
+
* session at the moment of a negotiation call.
|
|
4
|
+
*
|
|
5
|
+
* Captures the UI stack (pages previously pushed into this
|
|
6
|
+
* session), recent conversation, and optional interface context
|
|
7
|
+
* (viewport, device class). Consumed by `NegotiatorDecisionInput`
|
|
8
|
+
* to drive `create / update / compose / replace` decisions — the
|
|
9
|
+
* stack tells the engine whether something already on screen can
|
|
10
|
+
* absorb this push as an `update`, and the conversation history
|
|
11
|
+
* feeds the decision LLM.
|
|
12
|
+
*
|
|
13
|
+
* Kept in `@ggui-ai/negotiator` (not `mcp-server-core`): this is
|
|
14
|
+
* the decision engine's input shape, not a storage seam. An MCP
|
|
15
|
+
* server implementer binding against the public `Negotiator`
|
|
16
|
+
* interface sees `NegotiatorInput` / `NegotiatorResult` — never
|
|
17
|
+
* this internal input shape. Community adapters that want to
|
|
18
|
+
* call `makeDecision` directly (bypassing the `Negotiator` wrapper)
|
|
19
|
+
* import this type.
|
|
20
|
+
*/
|
|
21
|
+
export {};
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Suggestion engine — detects structured data patterns in free-form
|
|
3
|
+
* agent text and proposes a ggui_push call that would show them
|
|
4
|
+
* interactively.
|
|
5
|
+
*
|
|
6
|
+
* Pure regex heuristics — zero I/O, zero LLM. The agent can call this
|
|
7
|
+
* opportunistically on streamed output to decide whether to surface a
|
|
8
|
+
* UI suggestion. Callers must pass a deduplication set to avoid
|
|
9
|
+
* re-suggesting the same UI repeatedly in a live session.
|
|
10
|
+
*/
|
|
11
|
+
/** Suggestion event sent to the agent. */
|
|
12
|
+
export interface NegotiatorSuggestion {
|
|
13
|
+
type: 'negotiator:suggest';
|
|
14
|
+
trigger: 'stream-data-detected' | 'session-context' | 'user-pattern';
|
|
15
|
+
message: string;
|
|
16
|
+
suggestedAction: {
|
|
17
|
+
tool: 'ggui_push';
|
|
18
|
+
input: {
|
|
19
|
+
data?: Record<string, unknown>;
|
|
20
|
+
prompt?: string;
|
|
21
|
+
};
|
|
22
|
+
};
|
|
23
|
+
confidence: number;
|
|
24
|
+
intentId: string;
|
|
25
|
+
}
|
|
26
|
+
/** Detect structured data patterns in streamed text. */
|
|
27
|
+
export declare function detectDataPatterns(text: string): Array<{
|
|
28
|
+
pattern: string;
|
|
29
|
+
uiType: string;
|
|
30
|
+
confidence: number;
|
|
31
|
+
}>;
|
|
32
|
+
/** Build a suggestion event from detected patterns. */
|
|
33
|
+
export declare function buildSuggestion(sessionId: string, detections: Array<{
|
|
34
|
+
pattern: string;
|
|
35
|
+
uiType: string;
|
|
36
|
+
confidence: number;
|
|
37
|
+
}>, activeIntentIds: Set<string>): NegotiatorSuggestion | null;
|
|
38
|
+
//# sourceMappingURL=suggestion.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"suggestion.d.ts","sourceRoot":"","sources":["../src/suggestion.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAIH,0CAA0C;AAC1C,MAAM,WAAW,oBAAoB;IACnC,IAAI,EAAE,oBAAoB,CAAC;IAC3B,OAAO,EAAE,sBAAsB,GAAG,iBAAiB,GAAG,cAAc,CAAC;IACrE,OAAO,EAAE,MAAM,CAAC;IAChB,eAAe,EAAE;QACf,IAAI,EAAE,WAAW,CAAC;QAClB,KAAK,EAAE;YAAE,IAAI,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;YAAC,MAAM,CAAC,EAAE,MAAM,CAAA;SAAE,CAAC;KAC5D,CAAC;IACF,UAAU,EAAE,MAAM,CAAC;IACnB,QAAQ,EAAE,MAAM,CAAC;CAClB;AAmBD,wDAAwD;AACxD,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,MAAM,GAAG,KAAK,CAAC;IAAE,OAAO,EAAE,MAAM,CAAC;IAAC,MAAM,EAAE,MAAM,CAAC;IAAC,UAAU,EAAE,MAAM,CAAA;CAAE,CAAC,CAQ/G;AAED,uDAAuD;AACvD,wBAAgB,eAAe,CAC7B,SAAS,EAAE,MAAM,EACjB,UAAU,EAAE,KAAK,CAAC;IAAE,OAAO,EAAE,MAAM,CAAC;IAAC,MAAM,EAAE,MAAM,CAAC;IAAC,UAAU,EAAE,MAAM,CAAA;CAAE,CAAC,EAC1E,eAAe,EAAE,GAAG,CAAC,MAAM,CAAC,GAC3B,oBAAoB,GAAG,IAAI,CAa7B"}
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Suggestion engine — detects structured data patterns in free-form
|
|
3
|
+
* agent text and proposes a ggui_push call that would show them
|
|
4
|
+
* interactively.
|
|
5
|
+
*
|
|
6
|
+
* Pure regex heuristics — zero I/O, zero LLM. The agent can call this
|
|
7
|
+
* opportunistically on streamed output to decide whether to surface a
|
|
8
|
+
* UI suggestion. Callers must pass a deduplication set to avoid
|
|
9
|
+
* re-suggesting the same UI repeatedly in a live session.
|
|
10
|
+
*/
|
|
11
|
+
import { computeIntentId } from './intent.js';
|
|
12
|
+
const PATTERNS = [
|
|
13
|
+
{ name: 'temperature', regex: /\b\d+\s*°[CF]\b/i, uiType: 'weather display', confidence: 0.8 },
|
|
14
|
+
{ name: 'percentage', regex: /\b\d+(\.\d+)?\s*%/, uiType: 'metrics display', confidence: 0.6 },
|
|
15
|
+
{ name: 'currency', regex: /\$\s*[\d,]+(\.\d{2})?|\b\d+(\.\d{2})?\s*(USD|EUR|GBP|JPY|KRW)\b/i, uiType: 'financial display', confidence: 0.7 },
|
|
16
|
+
{ name: 'speed-or-distance', regex: /\b\d+(\.\d+)?\s*(km\/h|mph|km|mi|m\/s)\b/i, uiType: 'data display', confidence: 0.6 },
|
|
17
|
+
{ name: 'numbered-list', regex: /(?:^|\n)\s*[1-9]\.\s+\S/m, uiType: 'step flow or list', confidence: 0.5 },
|
|
18
|
+
{ name: 'comparison', regex: /\b(vs\.?|versus|compared to|on the other hand|alternatively)\b/i, uiType: 'comparison view', confidence: 0.6 },
|
|
19
|
+
{ name: 'table-like', regex: /\|.*\|.*\|/, uiType: 'table or grid', confidence: 0.8 },
|
|
20
|
+
];
|
|
21
|
+
/** Detect structured data patterns in streamed text. */
|
|
22
|
+
export function detectDataPatterns(text) {
|
|
23
|
+
const matches = [];
|
|
24
|
+
for (const p of PATTERNS) {
|
|
25
|
+
if (p.regex.test(text)) {
|
|
26
|
+
matches.push({ pattern: p.name, uiType: p.uiType, confidence: p.confidence });
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
return matches;
|
|
30
|
+
}
|
|
31
|
+
/** Build a suggestion event from detected patterns. */
|
|
32
|
+
export function buildSuggestion(sessionId, detections, activeIntentIds) {
|
|
33
|
+
if (detections.length === 0)
|
|
34
|
+
return null;
|
|
35
|
+
const best = detections.reduce((a, b) => (a.confidence > b.confidence ? a : b));
|
|
36
|
+
const intentId = computeIntentId(sessionId, { detectedPattern: best.pattern }, 'create');
|
|
37
|
+
if (activeIntentIds.has(intentId))
|
|
38
|
+
return null;
|
|
39
|
+
return {
|
|
40
|
+
type: 'negotiator:suggest',
|
|
41
|
+
trigger: 'stream-data-detected',
|
|
42
|
+
message: `I noticed structured data in your response (${best.pattern}). Consider calling ggui_push for an interactive ${best.uiType} instead.`,
|
|
43
|
+
suggestedAction: { tool: 'ggui_push', input: { prompt: best.uiType } },
|
|
44
|
+
confidence: best.confidence,
|
|
45
|
+
intentId,
|
|
46
|
+
};
|
|
47
|
+
}
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Hand-curated synthesizer bench corpus.
|
|
3
|
+
*
|
|
4
|
+
* Each entry pairs a natural-language intent with the structural
|
|
5
|
+
* contract shape we expect the synthesizer to produce. The shape is
|
|
6
|
+
* intentionally coarse — we assert what specs MUST appear, what specs
|
|
7
|
+
* MUST be absent, and (when applicable) which action / slot names are
|
|
8
|
+
* acceptable. Exact action labels and schema details are out of scope
|
|
9
|
+
* because LLM phrasing varies.
|
|
10
|
+
*
|
|
11
|
+
* Entries carry no hand-labeled archetype — the protocol thinks in
|
|
12
|
+
* contract SHAPE, not interaction-pattern categories. The bench
|
|
13
|
+
* report groups entries by `contractShape(entry.expected)`, derived
|
|
14
|
+
* from which specs the expected contract declares: `props-only`,
|
|
15
|
+
* `context-only`, `context+action`, `stream`, `with-gadgets`.
|
|
16
|
+
*
|
|
17
|
+
* Used by:
|
|
18
|
+
* - run-synth-bench.ts (live LLM probe — opt-in, costs ~$0.001/entry)
|
|
19
|
+
* - structure-bench.test.ts (deterministic validator-only check)
|
|
20
|
+
*/
|
|
21
|
+
import type { GadgetDescriptor } from '@ggui-ai/protocol';
|
|
22
|
+
export interface BenchEntry {
|
|
23
|
+
readonly id: string;
|
|
24
|
+
readonly intent: string;
|
|
25
|
+
readonly expected: BenchExpectation;
|
|
26
|
+
/**
|
|
27
|
+
* Per-entry registered gadget catalog. When set, the runner
|
|
28
|
+
* forwards this to `synthesizeContract`'s `appGadgets` option so
|
|
29
|
+
* the synth LLM sees the same "AVAILABLE GADGETS" prompt block a
|
|
30
|
+
* production server emits for an operator that registered these
|
|
31
|
+
* gadget packages in `App.gadgets`.
|
|
32
|
+
*
|
|
33
|
+
* Use for `capability-plugin` cases that test whether the synth
|
|
34
|
+
* reaches for an OPERATOR-REGISTERED 3rd-party gadget (Leaflet,
|
|
35
|
+
* Mapbox, …) — distinct from the STDLIB `capability` cases which
|
|
36
|
+
* rely on the hardcoded stdlib hint baked into the synth system
|
|
37
|
+
* prompt.
|
|
38
|
+
*
|
|
39
|
+
* Pre-plugin entries leave this absent — synth falls through to
|
|
40
|
+
* the static stdlib hint.
|
|
41
|
+
*/
|
|
42
|
+
readonly appGadgets?: readonly GadgetDescriptor[];
|
|
43
|
+
}
|
|
44
|
+
export interface BenchExpectation {
|
|
45
|
+
readonly hasActionSpec: boolean;
|
|
46
|
+
readonly hasContextSpec: boolean;
|
|
47
|
+
readonly hasStreamSpec: boolean;
|
|
48
|
+
readonly hasProps: boolean;
|
|
49
|
+
/**
|
|
50
|
+
* EE+ wire-shape v2 surfaces (added 2026-05-11). Default to `false`
|
|
51
|
+
* on legacy entries so the bench stays strict on the new shape only
|
|
52
|
+
* where the corpus opts in.
|
|
53
|
+
*/
|
|
54
|
+
readonly hasClientCapabilities?: boolean;
|
|
55
|
+
readonly hasAgentTools?: boolean;
|
|
56
|
+
/** Action names the synthesizer MAY emit. Match: at least one
|
|
57
|
+
* synthesized action name contains, or is contained by, an allowed
|
|
58
|
+
* name (case-insensitive — `nextStep` matches `next`). When
|
|
59
|
+
* `hasActionSpec=false`, this MUST be empty / omitted. */
|
|
60
|
+
readonly actionNames?: readonly string[];
|
|
61
|
+
/** Context slot names the synthesizer MAY emit. Match: at least one
|
|
62
|
+
* synthesized slot name contains, or is contained by, an allowed
|
|
63
|
+
* name (case-insensitive). Extra intent-specific slots are
|
|
64
|
+
* tolerated — the check verifies the synth landed a recognizable
|
|
65
|
+
* slot, not that it produced ONLY enumerated ones. */
|
|
66
|
+
readonly contextSlots?: readonly string[];
|
|
67
|
+
/** Gadget export names (hook or component) the synthesizer MAY
|
|
68
|
+
* declare on `clientCapabilities.gadgets` (the inner export key).
|
|
69
|
+
* Match is case-insensitive set-membership over the union of
|
|
70
|
+
* accepted names. Only checked when `hasClientCapabilities=true`. */
|
|
71
|
+
readonly capabilityHooks?: readonly string[];
|
|
72
|
+
/**
|
|
73
|
+
* Gadget export names the synthesizer MUST NOT declare on
|
|
74
|
+
* `clientCapabilities.gadgets`. Stricter dual of
|
|
75
|
+
* `capabilityHooks` — catches the failure mode "intent doesn't
|
|
76
|
+
* call for the gadget but the LLM over-eagerly attached it
|
|
77
|
+
* anyway." Case-insensitive set-membership; checked
|
|
78
|
+
* unconditionally (a contract with no clientCapabilities trivially
|
|
79
|
+
* passes). Used by distractor cases where a gadget is REGISTERED
|
|
80
|
+
* but the intent doesn't justify using it.
|
|
81
|
+
*/
|
|
82
|
+
readonly forbiddenCapabilityHooks?: readonly string[];
|
|
83
|
+
/** AgentTool catalog keys the synthesizer MAY emit. Used for
|
|
84
|
+
* source-fed-stream + action-nextStep cases. Match: at least one
|
|
85
|
+
* synthesized tool name contains, or is contained by, an allowed
|
|
86
|
+
* name (case-insensitive — `fetch_pending_jobs` matches `jobs`);
|
|
87
|
+
* the synth guesses these names, so paraphrase is tolerated, as
|
|
88
|
+
* with `actionNames`. Only checked when `hasAgentTools=true`. */
|
|
89
|
+
readonly agentToolNames?: readonly string[];
|
|
90
|
+
/** True when the corpus tolerates either shape (hard cases). When
|
|
91
|
+
* set, the bench reports the entry as `tolerated` regardless of
|
|
92
|
+
* hasActionSpec / hasContextSpec mismatches. */
|
|
93
|
+
readonly tolerateEitherShape?: boolean;
|
|
94
|
+
readonly notes?: string;
|
|
95
|
+
}
|
|
96
|
+
/**
|
|
97
|
+
* Contract-shape bucket of an expected contract — the grouping the
|
|
98
|
+
* bench report rolls precision up by, replacing the retired
|
|
99
|
+
* interaction-archetype categories. Priority-ordered so each entry
|
|
100
|
+
* lands in exactly one bucket: the most distinctive declared spec
|
|
101
|
+
* wins (a gadget-bearing contract is `with-gadgets` regardless of
|
|
102
|
+
* what else it declares; a stream-bearing one is `stream`; …).
|
|
103
|
+
*/
|
|
104
|
+
export declare function contractShape(expected: BenchExpectation): string;
|
|
105
|
+
export declare const BENCH_CORPUS: readonly BenchEntry[];
|
|
106
|
+
//# sourceMappingURL=corpus.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"corpus.d.ts","sourceRoot":"","sources":["../../src/synth-bench/corpus.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,mBAAmB,CAAC;AAE1D,MAAM,WAAW,UAAU;IACzB,QAAQ,CAAC,EAAE,EAAE,MAAM,CAAC;IACpB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,QAAQ,CAAC,QAAQ,EAAE,gBAAgB,CAAC;IACpC;;;;;;;;;;;;;;;OAeG;IACH,QAAQ,CAAC,UAAU,CAAC,EAAE,SAAS,gBAAgB,EAAE,CAAC;CACnD;AAED,MAAM,WAAW,gBAAgB;IAC/B,QAAQ,CAAC,aAAa,EAAE,OAAO,CAAC;IAChC,QAAQ,CAAC,cAAc,EAAE,OAAO,CAAC;IACjC,QAAQ,CAAC,aAAa,EAAE,OAAO,CAAC;IAChC,QAAQ,CAAC,QAAQ,EAAE,OAAO,CAAC;IAC3B;;;;OAIG;IACH,QAAQ,CAAC,qBAAqB,CAAC,EAAE,OAAO,CAAC;IACzC,QAAQ,CAAC,aAAa,CAAC,EAAE,OAAO,CAAC;IACjC;;;+DAG2D;IAC3D,QAAQ,CAAC,WAAW,CAAC,EAAE,SAAS,MAAM,EAAE,CAAC;IACzC;;;;2DAIuD;IACvD,QAAQ,CAAC,YAAY,CAAC,EAAE,SAAS,MAAM,EAAE,CAAC;IAC1C;;;0EAGsE;IACtE,QAAQ,CAAC,eAAe,CAAC,EAAE,SAAS,MAAM,EAAE,CAAC;IAC7C;;;;;;;;;OASG;IACH,QAAQ,CAAC,wBAAwB,CAAC,EAAE,SAAS,MAAM,EAAE,CAAC;IACtD;;;;;sEAKkE;IAClE,QAAQ,CAAC,cAAc,CAAC,EAAE,SAAS,MAAM,EAAE,CAAC;IAC5C;;qDAEiD;IACjD,QAAQ,CAAC,mBAAmB,CAAC,EAAE,OAAO,CAAC;IACvC,QAAQ,CAAC,KAAK,CAAC,EAAE,MAAM,CAAC;CACzB;AAED;;;;;;;GAOG;AACH,wBAAgB,aAAa,CAAC,QAAQ,EAAE,gBAAgB,GAAG,MAAM,CAOhE;AAoDD,eAAO,MAAM,YAAY,EAAE,SAAS,UAAU,EA+7B7C,CAAC"}
|