@ggui-ai/negotiator 0.1.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +49 -0
  3. package/dist/contract-hash.d.ts +54 -0
  4. package/dist/contract-hash.d.ts.map +1 -0
  5. package/dist/contract-hash.js +96 -0
  6. package/dist/contract-validators.d.ts +171 -0
  7. package/dist/contract-validators.d.ts.map +1 -0
  8. package/dist/contract-validators.js +478 -0
  9. package/dist/decision-input.d.ts +48 -0
  10. package/dist/decision-input.d.ts.map +1 -0
  11. package/dist/decision-input.js +14 -0
  12. package/dist/decision.d.ts +54 -0
  13. package/dist/decision.d.ts.map +1 -0
  14. package/dist/decision.js +500 -0
  15. package/dist/index.d.ts +36 -0
  16. package/dist/index.d.ts.map +1 -0
  17. package/dist/index.js +25 -0
  18. package/dist/intent.d.ts +22 -0
  19. package/dist/intent.d.ts.map +1 -0
  20. package/dist/intent.js +28 -0
  21. package/dist/llm-caller.d.ts +70 -0
  22. package/dist/llm-caller.d.ts.map +1 -0
  23. package/dist/llm-caller.js +38 -0
  24. package/dist/llm-rerank.d.ts +101 -0
  25. package/dist/llm-rerank.d.ts.map +1 -0
  26. package/dist/llm-rerank.js +178 -0
  27. package/dist/negotiate.d.ts +141 -0
  28. package/dist/negotiate.d.ts.map +1 -0
  29. package/dist/negotiate.js +161 -0
  30. package/dist/normalize-schema.d.ts +22 -0
  31. package/dist/normalize-schema.d.ts.map +1 -0
  32. package/dist/normalize-schema.js +191 -0
  33. package/dist/pure.d.ts +30 -0
  34. package/dist/pure.d.ts.map +1 -0
  35. package/dist/pure.js +43 -0
  36. package/dist/rag-search.d.ts +73 -0
  37. package/dist/rag-search.d.ts.map +1 -0
  38. package/dist/rag-search.js +192 -0
  39. package/dist/rerank-eval/pairs.d.ts +28 -0
  40. package/dist/rerank-eval/pairs.d.ts.map +1 -0
  41. package/dist/rerank-eval/pairs.js +531 -0
  42. package/dist/rerank-eval/run-probe-cli.d.ts +3 -0
  43. package/dist/rerank-eval/run-probe-cli.d.ts.map +1 -0
  44. package/dist/rerank-eval/run-probe-cli.js +146 -0
  45. package/dist/rerank-eval/run-probe.d.ts +68 -0
  46. package/dist/rerank-eval/run-probe.d.ts.map +1 -0
  47. package/dist/rerank-eval/run-probe.js +113 -0
  48. package/dist/session.d.ts +42 -0
  49. package/dist/session.d.ts.map +1 -0
  50. package/dist/session.js +21 -0
  51. package/dist/suggestion.d.ts +38 -0
  52. package/dist/suggestion.d.ts.map +1 -0
  53. package/dist/suggestion.js +47 -0
  54. package/dist/synth-bench/corpus.d.ts +106 -0
  55. package/dist/synth-bench/corpus.d.ts.map +1 -0
  56. package/dist/synth-bench/corpus.js +994 -0
  57. package/dist/synth-bench/run-bench-cli.d.ts +3 -0
  58. package/dist/synth-bench/run-bench-cli.d.ts.map +1 -0
  59. package/dist/synth-bench/run-bench-cli.js +181 -0
  60. package/dist/synth-bench/run-bench.d.ts +101 -0
  61. package/dist/synth-bench/run-bench.d.ts.map +1 -0
  62. package/dist/synth-bench/run-bench.js +374 -0
  63. package/dist/synthesize-contract.d.ts +131 -0
  64. package/dist/synthesize-contract.d.ts.map +1 -0
  65. package/dist/synthesize-contract.js +948 -0
  66. package/dist/types.d.ts +30 -0
  67. package/dist/types.d.ts.map +1 -0
  68. package/dist/types.js +13 -0
  69. package/package.json +74 -0
  70. package/src/contract-hash.ts +102 -0
  71. package/src/contract-validators.ts +604 -0
  72. package/src/decision-input.ts +49 -0
  73. package/src/decision.ts +581 -0
  74. package/src/index.ts +63 -0
  75. package/src/intent.ts +37 -0
  76. package/src/llm-caller.ts +82 -0
  77. package/src/llm-rerank.ts +280 -0
  78. package/src/negotiate.ts +312 -0
  79. package/src/normalize-schema.ts +193 -0
  80. package/src/pure.ts +46 -0
  81. package/src/rag-search.ts +274 -0
  82. package/src/rerank-eval/pairs.ts +624 -0
  83. package/src/rerank-eval/run-probe-cli.ts +197 -0
  84. package/src/rerank-eval/run-probe.ts +198 -0
  85. package/src/session.ts +41 -0
  86. package/src/suggestion.ts +73 -0
  87. package/src/synth-bench/corpus.ts +1126 -0
  88. package/src/synth-bench/run-bench-cli.ts +237 -0
  89. package/src/synth-bench/run-bench.ts +525 -0
  90. package/src/synthesize-contract.ts +1161 -0
  91. package/src/types.ts +31 -0
@@ -0,0 +1,478 @@
1
+ /**
2
+ * Programmatic safety validators for `DataContract` shapes.
3
+ *
4
+ * Two detectors live here:
5
+ *
6
+ * - {@link validateContractStructure} — pure structural heuristics that
7
+ * flag over-specified contracts without any runtime dependency. The
8
+ * load-bearing finding is `redundant-action`: an empty-payload
9
+ * `actionSpec` entry whose name parses as a mutator of an existing
10
+ * `contextSpec` slot. Real example that motivated this module: a
11
+ * synthesizer emitted both `actionSpec.increment` (empty payload)
12
+ * and `contextSpec.count` for a counter widget. The generator wired
13
+ * the increment button to `useAction("increment")`, dispatching to
14
+ * the agent instead of locally bumping the count slot — a button
15
+ * that looks right but does nothing. The structural fix is to
16
+ * declare `actionSpec[X]` IFF X is a discrete event the agent must
17
+ * witness; mutators of context slots use the slot setter.
18
+ *
19
+ * - {@link validateContractNovelty} — embeds the contract via the
20
+ * shared `summarizeContract` helper and computes cosine distance to
21
+ * the nearest registered blueprint. Distance above the threshold
22
+ * yields a `novel-shape` finding so operators can review before the
23
+ * contract pollutes the registry's neighborhood.
24
+ *
25
+ * Both validators return a `ContractValidationResult` carrying readonly
26
+ * findings — callers (synthesizer, registerBlueprint) decide how to act
27
+ * on warnings vs errors. Findings are deterministic; no LLM judgment.
28
+ */
29
+ import { summarizeContract } from '@ggui-ai/protocol';
30
+ // =============================================================================
31
+ // Mutator verb dictionary
32
+ // =============================================================================
33
+ /**
34
+ * Verbs whose presence at the START of an action name signals "this
35
+ * action mutates state the agent observes" — i.e., a candidate slot
36
+ * setter masquerading as an action.
37
+ *
38
+ * Scope decisions (in vs out):
39
+ *
40
+ * IN: verbs that almost always mutate observable state in
41
+ * counter/list/form/UI patterns we've seen in benchmark traces.
42
+ *
43
+ * OUT: verbs that often LOOK like mutators but legitimately want
44
+ * the agent in the loop:
45
+ * - `submit` — submit IS the discrete event; the agent must
46
+ * witness "user pressed submit" beyond the form's
47
+ * draft slot. Form pattern is `actionSpec.submit` +
48
+ * `contextSpec.formData` (legitimate pairing).
49
+ * - `save` — save typically triggers a wired tool (write to
50
+ * storage). Same pattern as submit.
51
+ * - `cancel` — discrete user gesture, not a slot mutation.
52
+ * - `confirm` — discrete user gesture.
53
+ * - `select` — selection often IS the slot value (e.g., select
54
+ * reduces to setting `selectedId`), but it's also
55
+ * used as a discrete event ("user picked X, fetch
56
+ * detail"). Out by default; false-negatives here
57
+ * are cheaper than false-positives.
58
+ * - `open`/`close` — often dialog gestures, not slot mutations.
59
+ *
60
+ * Authors who want a tighter or looser policy override the list at
61
+ * call site (option for future expansion — not exposed yet to keep
62
+ * the API minimal).
63
+ */
64
+ const MUTATOR_VERBS = [
65
+ 'increment',
66
+ 'decrement',
67
+ 'reset',
68
+ 'set',
69
+ 'add',
70
+ 'remove',
71
+ 'delete',
72
+ 'update',
73
+ 'change',
74
+ 'toggle',
75
+ 'flip',
76
+ 'clear',
77
+ 'append',
78
+ 'prepend',
79
+ 'insert',
80
+ ];
81
+ /**
82
+ * Returns the mutator verb that prefixes `actionName`, or `null` when
83
+ * none does. Matches case-insensitively and only at the START — `set`
84
+ * matches `setCount` but not `unsetCount`; `reset` is its own verb (not
85
+ * `set` + `Count` with a leading `re`).
86
+ *
87
+ * Strict longest-match: when multiple verbs are valid prefixes (e.g.
88
+ * `reset` is also matched by `set`-with-`re`-prefix only via different
89
+ * boundary, but we don't allow that), we walk verbs longest-first so
90
+ * `reset*` is parsed as `reset|*` not `re|set*`.
91
+ */
92
+ function stripMutatorVerb(actionName) {
93
+ if (actionName.length === 0)
94
+ return null;
95
+ const lower = actionName.toLowerCase();
96
+ // Longest-prefix-first to disambiguate `reset` vs (hypothetical) `re`.
97
+ const sorted = [...MUTATOR_VERBS].sort((a, b) => b.length - a.length);
98
+ for (const verb of sorted) {
99
+ if (!lower.startsWith(verb))
100
+ continue;
101
+ const tail = actionName.slice(verb.length);
102
+ // Boundary: either the verb consumed the whole name, or the
103
+ // character after the verb is a word boundary (uppercase letter,
104
+ // digit, or underscore — typical camelCase / snake_case break).
105
+ if (tail.length === 0) {
106
+ return { verb, remainder: '' };
107
+ }
108
+ const next = tail.charAt(0);
109
+ const isBoundary = (next >= 'A' && next <= 'Z') ||
110
+ (next >= '0' && next <= '9') ||
111
+ next === '_';
112
+ if (!isBoundary)
113
+ continue;
114
+ return { verb, remainder: tail };
115
+ }
116
+ return null;
117
+ }
118
+ /**
119
+ * Decide whether an action name with a mutator-verb prefix targets the
120
+ * given context slot.
121
+ *
122
+ * Algorithm:
123
+ *
124
+ * 1. Strip the verb prefix → remainder.
125
+ * 2. Empty remainder (action is just the verb, e.g., "increment")
126
+ * AND the contract has exactly one context slot ⇒ flag.
127
+ * 3. Non-empty remainder ⇒ flag iff remainder contains slotName OR
128
+ * slotName contains remainder, case-insensitively. This catches
129
+ * `incrementCount` ↔ `count`, `setUserName` ↔ `userName`,
130
+ * `addItem` ↔ `items`, etc., without false-positiving on
131
+ * `setTheme` ↔ `count`.
132
+ *
133
+ * The single-slot case is the load-bearing one for the counter bug:
134
+ * `increment` (no remainder) + the only slot being `count` is the
135
+ * exact pattern.
136
+ */
137
+ function nameImpliesMutation(args) {
138
+ const { actionName, slotName, totalSlots } = args;
139
+ const stripped = stripMutatorVerb(actionName);
140
+ if (!stripped)
141
+ return false;
142
+ if (stripped.remainder.length === 0) {
143
+ return totalSlots === 1;
144
+ }
145
+ const a = stripped.remainder.toLowerCase();
146
+ const b = slotName.toLowerCase();
147
+ return a.includes(b) || b.includes(a);
148
+ }
149
+ // =============================================================================
150
+ // Empty-payload detection
151
+ // =============================================================================
152
+ /**
153
+ * True when an action's schema is "empty payload" — either omitted
154
+ * entirely or the canonical `{type:'object', properties:{},
155
+ * additionalProperties:false}` shape the synthesizer emits when the LLM
156
+ * declines to declare fields. Actions with declared payload fields
157
+ * (e.g., `{chipText: string}`) are NOT empty — those legitimately
158
+ * carry data the agent needs.
159
+ */
160
+ function isEmptyPayloadSchema(schema) {
161
+ if (schema === undefined)
162
+ return true;
163
+ if (schema.type !== 'object')
164
+ return false;
165
+ if (schema.properties === undefined)
166
+ return true;
167
+ return Object.keys(schema.properties).length === 0;
168
+ }
169
+ // =============================================================================
170
+ // Structural validator (synchronous, dependency-free)
171
+ // =============================================================================
172
+ /**
173
+ * Run the synchronous structural detectors against `contract`.
174
+ *
175
+ * Currently:
176
+ * - `redundant-action`: empty-payload action whose name parses as a
177
+ * mutator of an existing context slot.
178
+ *
179
+ * Returns an empty findings array for contracts that don't trip any
180
+ * heuristic.
181
+ */
182
+ export function validateContractStructure(contract) {
183
+ const findings = [];
184
+ const actionSpec = contract.actionSpec;
185
+ const contextSpec = contract.contextSpec;
186
+ if (!actionSpec || !contextSpec) {
187
+ return { findings };
188
+ }
189
+ const slotNames = Object.keys(contextSpec);
190
+ if (slotNames.length === 0) {
191
+ return { findings };
192
+ }
193
+ for (const [actionName, entry] of Object.entries(actionSpec)) {
194
+ if (!isEmptyPayloadSchema(entry.schema))
195
+ continue;
196
+ for (const slotName of slotNames) {
197
+ if (!nameImpliesMutation({
198
+ actionName,
199
+ slotName,
200
+ totalSlots: slotNames.length,
201
+ })) {
202
+ continue;
203
+ }
204
+ findings.push({
205
+ kind: 'redundant-action',
206
+ severity: 'warn',
207
+ actionName,
208
+ slotName,
209
+ hint: `Action "${actionName}" has empty payload and looks like a mutator of context slot "${slotName}". Prefer setting the slot directly from the component instead of declaring an action — the agent observes slot changes without a discrete event.`,
210
+ });
211
+ break;
212
+ }
213
+ }
214
+ return { findings };
215
+ }
216
+ // =============================================================================
217
+ // Actions-vs-context placement validator (synchronous, loose / advisory)
218
+ // =============================================================================
219
+ //
220
+ // Loose self-check: every finding is `severity: 'warn'`. Synth's
221
+ // pipeline can log warnings to telemetry without blocking output. The
222
+ // rule being checked: actions drive agent turns, context observes
223
+ // state, and there is no third category.
224
+ //
225
+ // Three checks, each scoped narrowly to avoid false positives:
226
+ //
227
+ // 1. Name collision across specs — same key in both actionSpec and
228
+ // contextSpec. Structurally a category error: synth couldn't
229
+ // decide which bucket. High-confidence rule.
230
+ //
231
+ // 2. State-y name in actionSpec — names matching common state-mirror
232
+ // patterns (autosave, draft, typing, change, focus, blur, …).
233
+ // These usually represent continuous state, not turn-driving
234
+ // events. Warning, not error — false positives possible (a real
235
+ // "submitDraft" action is legitimate).
236
+ //
237
+ // 3. Action-y name in contextSpec — names matching common terminal
238
+ // verbs (submit, send, confirm, cancel, next, done, apply, …).
239
+ // These usually represent discrete events, not state. Warning,
240
+ // not error — false positives possible (a "submitTime" timestamp
241
+ // slot is legitimate state).
242
+ //
243
+ // All three checks are intentionally LOOSE. They surface candidates;
244
+ // consumers (synth, operator dashboards) decide to act on them.
245
+ const STATE_LIKE_NAME_PATTERN = /^(auto|on)?(save|draft|change|update|sync|saved|typing|scroll|hover|focus|blur|input)/i;
246
+ const ACTION_LIKE_NAME_PATTERN = /^(submit|send|confirm|cancel|next|back|done|apply|delete|create|approve|reject)$/i;
247
+ /**
248
+ * Validate the actions-vs-context placement rule: actions drive agent
249
+ * turns, context observes state. Findings 1-3 are advisory `warn`;
250
+ * finding 4 (`context-props-name-collision`) is `error` — it is a
251
+ * SPEC §2.10 MUST. Synth pipelines use it as a self-check over their
252
+ * own output.
253
+ *
254
+ * Four findings:
255
+ *
256
+ * - `actions-vs-context-name-collision` — same key appears in both
257
+ * `actionSpec` AND `contextSpec`. Synth couldn't decide which
258
+ * bucket. Pick one.
259
+ *
260
+ * - `action-name-looks-state-y` — `actionSpec` entry name matches a
261
+ * state-mirror pattern (autosave, draft, typing, change, …). The
262
+ * thing might be observable state (continuous) rather than a
263
+ * turn-driving event. Consider `contextSpec`.
264
+ *
265
+ * - `context-name-looks-action-y` — `contextSpec` entry name matches
266
+ * a discrete-event pattern (submit, send, confirm, …). The thing
267
+ * might be a one-shot event the agent reacts to. Consider
268
+ * `actionSpec`.
269
+ *
270
+ * - `context-props-name-collision` (ERROR) — a `contextSpec` slot key
271
+ * equals a `propsSpec` property name. SPEC §2.10 MUST: the
272
+ * generated boilerplate would shadow the prop binding.
273
+ *
274
+ * Pure / synchronous / no dependencies. Safe to call on every synth
275
+ * output without measurable cost.
276
+ */
277
+ export function validateActionsVsContext(contract) {
278
+ const findings = [];
279
+ const actionKeys = Object.keys(contract.actionSpec ?? {});
280
+ const contextKeys = Object.keys(contract.contextSpec ?? {});
281
+ // 1. Same key in both specs — category collision.
282
+ for (const k of actionKeys) {
283
+ if (contextKeys.includes(k)) {
284
+ findings.push({
285
+ kind: 'actions-vs-context-name-collision',
286
+ severity: 'warn',
287
+ actionName: k,
288
+ slotName: k,
289
+ hint: `"${k}" appears in both actionSpec and contextSpec. Pick one — actions drive agent turns; contextSpec is observed state.`,
290
+ });
291
+ }
292
+ }
293
+ // 2. State-y name in actionSpec.
294
+ for (const k of actionKeys) {
295
+ if (STATE_LIKE_NAME_PATTERN.test(k)) {
296
+ findings.push({
297
+ kind: 'action-name-looks-state-y',
298
+ severity: 'warn',
299
+ actionName: k,
300
+ hint: `actionSpec entry "${k}" sounds like state (autosave/draft/change/etc.). If it doesn't drive an agent turn, consider contextSpec instead.`,
301
+ });
302
+ }
303
+ }
304
+ // 3. Action-y name in contextSpec.
305
+ for (const k of contextKeys) {
306
+ if (ACTION_LIKE_NAME_PATTERN.test(k)) {
307
+ findings.push({
308
+ kind: 'context-name-looks-action-y',
309
+ severity: 'warn',
310
+ slotName: k,
311
+ hint: `contextSpec entry "${k}" sounds like a discrete event (submit/send/confirm/etc.). If it drives an agent turn, consider actionSpec instead.`,
312
+ });
313
+ }
314
+ }
315
+ // 4. contextSpec slot key collides with a propsSpec property name.
316
+ // SPEC §2.10 MUST — the generated boilerplate binds both into the
317
+ // same scope, so a collision would shadow the prop. The protocol's
318
+ // `CTR_DUP_NAME` only covers action/stream/context collisions, NOT
319
+ // props↔context — so this is the one cross-spec collision the synth
320
+ // gate would otherwise miss. `error` severity → the synth's repair
321
+ // loop fixes it.
322
+ const propKeys = Object.keys(contract.propsSpec?.properties ?? {});
323
+ for (const k of contextKeys) {
324
+ if (propKeys.includes(k)) {
325
+ findings.push({
326
+ kind: 'context-props-name-collision',
327
+ severity: 'error',
328
+ slotName: k,
329
+ hint: `contextSpec slot "${k}" collides with propsSpec.properties.${k} — the generated boilerplate would shadow the prop binding. Rename one of them.`,
330
+ });
331
+ }
332
+ }
333
+ return { findings };
334
+ }
335
+ // =============================================================================
336
+ // Coherence validator (synchronous, intent-aware — the one validator
337
+ // that reads the intent, not just the contract)
338
+ // =============================================================================
339
+ //
340
+ // Catches the "degenerate contract" flake: the synthesizer occasionally
341
+ // emits a contract that is JUST an actionSpec with no data surface at
342
+ // all — no contextSpec, no propsSpec, no streamSpec, no
343
+ // clientCapabilities. The agent then receives a contentless event it
344
+ // cannot act on (a `finish` for a checkout it has no data for; a
345
+ // `share` for an article it was never given).
346
+ //
347
+ // `{actionSpec: {confirm, cancel}}` with no data surface is the ONE
348
+ // legitimate no-surface shape (a pure-decision modal) — so the rule is
349
+ // gated on the intent: it fires only when the intent describes a UI
350
+ // that demonstrably displays or collects data (a flow / form / article
351
+ // / profile / …). A pure-decision modal intent matches none of those,
352
+ // so it is never flagged.
353
+ /**
354
+ * Intent signal for "this UI displays or collects data, so the
355
+ * contract MUST declare a data surface." Deliberately narrow — every
356
+ * keyword is an intent that genuinely cannot work as actionSpec-only.
357
+ */
358
+ // `card` and `flow` are deliberately EXCLUDED — too common in English
359
+ // ("a card to confirm this flow" is a legit pure-decision modal). The
360
+ // flow-* corpus intents still match via `wizard` / `checkout` /
361
+ // `onboard` / `multi-step`, so coverage is unchanged.
362
+ const DATA_SURFACE_INTENT_PATTERN = /\bwizard\b|multi-?step|\d+-step|\bcheckout\b|\bonboard|\barticle\b|\bdocument\b|\bprofile\b|\bdashboard\b|\bform\b|\breport\b|\beditor\b/i;
363
+ /**
364
+ * Validate that a contract is coherent with its intent. The only
365
+ * intent-aware validator: it reads the natural-language intent
366
+ * alongside the contract.
367
+ *
368
+ * One rule — `incoherent-no-data-surface`: the intent describes a
369
+ * data-bearing UI but the contract declares only EMPTY-payload
370
+ * actions, with no contextSpec / propsSpec / streamSpec /
371
+ * clientCapabilities. That contract is structurally degenerate — the
372
+ * agent gets a contentless event with nothing behind it. Emitted at
373
+ * `severity: 'error'` so the synth's repair loop retries (the
374
+ * contract is protocol-valid, so nothing else flags it).
375
+ *
376
+ * The empty-payload condition matters: an action that DOES carry a
377
+ * payload (e.g. an `autosave` action whose schema includes the draft
378
+ * text) gives the agent data through the payload — that is not
379
+ * degenerate and is NOT flagged.
380
+ *
381
+ * Pure / synchronous. The rule fires ONLY on the degenerate output —
382
+ * a correct contract for any data-bearing intent always has a
383
+ * surface, so it never false-positives on good output.
384
+ */
385
+ export function validateContractCoherence(contract, intent) {
386
+ const findings = [];
387
+ const actions = contract.actionSpec ?? {};
388
+ const hasAction = Object.keys(actions).length > 0;
389
+ const allActionsEmptyPayload = Object.values(actions).every((a) => isEmptyPayloadSchema(a.schema));
390
+ const hasContext = contract.contextSpec !== undefined &&
391
+ Object.keys(contract.contextSpec).length > 0;
392
+ const hasProps = contract.propsSpec?.properties !== undefined &&
393
+ Object.keys(contract.propsSpec.properties).length > 0;
394
+ const hasStream = contract.streamSpec !== undefined &&
395
+ Object.keys(contract.streamSpec).length > 0;
396
+ const hasGadgets = contract.clientCapabilities?.gadgets !== undefined &&
397
+ Object.keys(contract.clientCapabilities.gadgets).length > 0;
398
+ if (hasAction &&
399
+ allActionsEmptyPayload &&
400
+ !hasContext &&
401
+ !hasProps &&
402
+ !hasStream &&
403
+ !hasGadgets &&
404
+ DATA_SURFACE_INTENT_PATTERN.test(intent)) {
405
+ findings.push({
406
+ kind: 'incoherent-no-data-surface',
407
+ severity: 'error',
408
+ hint: 'The intent describes a UI that displays or collects data, but the contract declares only an actionSpec — no contextSpec, propsSpec, or streamSpec. The agent would receive an event with nothing behind it. Declare the data surface: contextSpec for state the user enters / the UI tracks (a wizard\'s step + form fields), or propsSpec for content the agent supplies at render (an article, a profile).',
409
+ });
410
+ }
411
+ return { findings };
412
+ }
413
+ // =============================================================================
414
+ // Novelty validator (async, depends on embedding + vector store)
415
+ // =============================================================================
416
+ const DEFAULT_NOVELTY_THRESHOLD_COSINE = 0.8;
417
+ /**
418
+ * Run the cosine-distance novelty detector. Embeds the contract via
419
+ * `summarizeContract` and queries the vector store for the nearest
420
+ * neighbor in `scope`. Distance above `thresholdCosine` yields a
421
+ * `novel-shape` finding so operators see "this contract is far from
422
+ * anything we've registered — review encouraged" before it ships.
423
+ *
424
+ * Defaults to `severity: 'warn'`. Distance is computed as
425
+ * `1 - cosineSimilarity`; `VectorStore.query` returns
426
+ * cosine-similarity scores in `[0, 1]` per the seam contract.
427
+ *
428
+ * Empty index (no nearest neighbor in scope) ⇒ flag with
429
+ * `cosine: undefined` because a fresh registry will register
430
+ * everything as novel; operators learn that the registry is empty.
431
+ */
432
+ export async function validateContractNovelty(contract, deps, options = {}) {
433
+ const threshold = options.thresholdCosine ?? DEFAULT_NOVELTY_THRESHOLD_COSINE;
434
+ const summary = summarizeContract(contract);
435
+ const queryEmbedding = await deps.embedding.embed(summary);
436
+ const results = await deps.vectorStore.query(deps.scope, queryEmbedding, 1);
437
+ if (results.length === 0) {
438
+ return {
439
+ findings: [
440
+ {
441
+ kind: 'novel-shape',
442
+ severity: 'warn',
443
+ hint: `No registered blueprints in scope "${deps.scope}" — this contract has no neighbors. Review encouraged before it becomes the seed for the registry.`,
444
+ },
445
+ ],
446
+ };
447
+ }
448
+ const nearest = results[0];
449
+ if (!nearest) {
450
+ return { findings: [] };
451
+ }
452
+ const distance = 1 - nearest.score;
453
+ if (distance < threshold) {
454
+ return { findings: [] };
455
+ }
456
+ return {
457
+ findings: [
458
+ {
459
+ kind: 'novel-shape',
460
+ severity: 'warn',
461
+ cosine: nearest.score,
462
+ hint: `Cosine distance ${distance.toFixed(3)} from nearest registered blueprint exceeds threshold ${threshold.toFixed(3)} — review encouraged before the contract pollutes the registry's neighborhood.`,
463
+ },
464
+ ],
465
+ };
466
+ }
467
+ /**
468
+ * Render a findings array as a single human-readable line for use in
469
+ * synthesizer `reason` strings, cache trace events, and operator logs.
470
+ * Empty findings → empty string so callers can append unconditionally.
471
+ */
472
+ export function formatValidationFindings(result) {
473
+ if (result.findings.length === 0)
474
+ return '';
475
+ return result.findings
476
+ .map((f) => `[${f.severity}:${f.kind}] ${f.hint}`)
477
+ .join(' | ');
478
+ }
@@ -0,0 +1,48 @@
1
+ /**
2
+ * `NegotiatorDecisionInput` — the full input to `makeDecision`.
3
+ *
4
+ * Kept in its own file so a typed re-export shim in
5
+ * `core/negotiation/src/types.ts` can keep the legacy import path
6
+ * alive without pulling the (bigger, commit-5) decision runtime
7
+ * into a types-only import.
8
+ *
9
+ * The `blueprintCandidates` entry shape is inlined intentionally —
10
+ * those five fields are the only projection `makeDecision` reads
11
+ * from a `NegotiatorOption`; extracting a named type here would
12
+ * grow public surface for no consumer.
13
+ */
14
+ import type { GadgetDescriptor, DataContract } from '@ggui-ai/protocol';
15
+ import type { SessionState } from './session.js';
16
+ /** Input to the decision engine. */
17
+ export interface NegotiatorDecisionInput {
18
+ agentData?: Record<string, unknown>;
19
+ agentPrompt?: string;
20
+ agentContext?: string | Record<string, unknown>;
21
+ /**
22
+ * MCP tools the AGENT invokes (catalog seed). The decision engine
23
+ * merges these into the resulting contract's
24
+ * `agentCapabilities.tools` catalog. Cross-references are authored
25
+ * by the LLM: the catalog is referenced from
26
+ * `actionSpec[*].nextStep` (post-action hint for the agent's next
27
+ * turn) and `streamSpec[*].source.tool` (channel data source). The
28
+ * component never calls these.
29
+ */
30
+ agentTools?: string[];
31
+ /**
32
+ * Browser-capability gadget catalog declared for the app. The
33
+ * handshake handler reads this from `app.gadgets` and
34
+ * threads it here so the decision LLM knows which gadget bindings
35
+ * the produced UI may reference (and so the merge step can enrich
36
+ * partial LLM output with canonical entries from the catalog).
37
+ */
38
+ gadgets?: readonly GadgetDescriptor[];
39
+ sessionState: SessionState;
40
+ blueprintCandidates: Array<{
41
+ blueprintId: string;
42
+ description: string;
43
+ contract?: DataContract;
44
+ similarity: number;
45
+ verdict: 'exact' | 'partial';
46
+ }>;
47
+ }
48
+ //# sourceMappingURL=decision-input.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"decision-input.d.ts","sourceRoot":"","sources":["../src/decision-input.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AAEH,OAAO,KAAK,EAAE,gBAAgB,EAAE,YAAY,EAAE,MAAM,mBAAmB,CAAC;AACxE,OAAO,KAAK,EAAE,YAAY,EAAE,MAAM,cAAc,CAAC;AAEjD,oCAAoC;AACpC,MAAM,WAAW,uBAAuB;IACtC,SAAS,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;IACpC,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,YAAY,CAAC,EAAE,MAAM,GAAG,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;IAChD;;;;;;;;OAQG;IACH,UAAU,CAAC,EAAE,MAAM,EAAE,CAAC;IACtB;;;;;;OAMG;IACH,OAAO,CAAC,EAAE,SAAS,gBAAgB,EAAE,CAAC;IACtC,YAAY,EAAE,YAAY,CAAC;IAC3B,mBAAmB,EAAE,KAAK,CAAC;QACzB,WAAW,EAAE,MAAM,CAAC;QACpB,WAAW,EAAE,MAAM,CAAC;QACpB,QAAQ,CAAC,EAAE,YAAY,CAAC;QACxB,UAAU,EAAE,MAAM,CAAC;QACnB,OAAO,EAAE,OAAO,GAAG,SAAS,CAAC;KAC9B,CAAC,CAAC;CACJ"}
@@ -0,0 +1,14 @@
1
+ /**
2
+ * `NegotiatorDecisionInput` — the full input to `makeDecision`.
3
+ *
4
+ * Kept in its own file so a typed re-export shim in
5
+ * `core/negotiation/src/types.ts` can keep the legacy import path
6
+ * alive without pulling the (bigger, commit-5) decision runtime
7
+ * into a types-only import.
8
+ *
9
+ * The `blueprintCandidates` entry shape is inlined intentionally —
10
+ * those five fields are the only projection `makeDecision` reads
11
+ * from a `NegotiatorOption`; extracting a named type here would
12
+ * grow public surface for no consumer.
13
+ */
14
+ export {};
@@ -0,0 +1,54 @@
1
+ /**
2
+ * Decision Engine — one LLM call, one UI decision.
3
+ *
4
+ * Replaces the V2 brainstorm/option-picker pattern with a single
5
+ * opinionated decision: `create` / `update` / `compose` / `replace`.
6
+ * The LLM sees the agent's data, the current session stack, any
7
+ * `blueprintCandidates` from `ragSearch`, and returns a
8
+ * {@link NegotiatorDecision} with a full {@link DataContract} payload.
9
+ *
10
+ * Contract shape — the returned `contract` always includes an
11
+ * `intent` (semantic identity — same intent = cached component).
12
+ * Other fields are populated opportunistically; `agentCapabilities` is
13
+ * always populated deterministically from `input.agentTools` after
14
+ * the LLM returns (see {@link mergeAgentCapabilities}). The
15
+ * `clientCapabilities.gadgets` catalog is similarly enriched from
16
+ * the per-app gadget list (see {@link mergeGadgets}).
17
+ *
18
+ * Structured output — prefers `llmCaller.callStructured?` with a
19
+ * forced-tool-use schema (guaranteed JSON). Falls back to text + regex
20
+ * JSON extraction if the caller doesn't support structured output, or
21
+ * if the structured call throws. On total parse failure, emits a
22
+ * fallback decision with `action: 'create'` built from the agent's
23
+ * data shape.
24
+ *
25
+ * ## Public surface + semver weight
26
+ *
27
+ * Exported:
28
+ * - `DECISION_SYSTEM_PROMPT` — the system prompt constant. Its
29
+ * content is part of the behavioral contract: changing the prompt
30
+ * changes what the model emits and therefore changes the cache
31
+ * identity of downstream generated blueprints. Treat content
32
+ * changes like `CRITERIA` ordering: CHANGELOG-worthy.
33
+ * - `buildDecisionUserMessage(input)` — pure string builder. Stable
34
+ * output for stable input.
35
+ * - `makeDecision(input, llmCaller)` — runtime orchestrator.
36
+ *
37
+ * Intentionally **not** exported:
38
+ * - `DECISION_TOOL` — the OpenAI-style tool schema handed to
39
+ * `callStructured`. Pinning its shape in the public API would
40
+ * freeze the tool-schema format consumers never need to see.
41
+ * - `mergeAgentCapabilities`, `mergeGadgets`, `buildFallbackDecision`,
42
+ * `inferType` — engine-internal helpers.
43
+ */
44
+ import type { NegotiatorAlternative, NegotiatorDecision } from '@ggui-ai/protocol';
45
+ import type { NegotiatorDecisionInput } from './decision-input.js';
46
+ import type { LLMCaller } from './llm-caller.js';
47
+ export declare const DECISION_SYSTEM_PROMPT = "You are a UI strategist for ggui, a generative UI platform. Given the agent's data, current session state, and blueprint candidates, decide the best way to show this information.\n\nRespond with a JSON object:\n{\n \"action\": \"create\" | \"update\" | \"compose\" | \"replace\",\n \"reasoning\": \"1-2 sentences explaining why\",\n \"blueprintId\": \"matched blueprint ID or null\",\n \"targetStackItemId\": \"existing page to update/compose/replace, or null\",\n \"contract\": {\n \"intent\": \"Concise purpose \u2014 e.g. 'Display current weather conditions for a quick daily check'\",\n \"propsSpec\": {\n \"properties\": {\n \"fieldName\": {\n \"description\": \"what this field is\",\n \"schema\": { \"type\": \"string\" },\n \"required\": true,\n \"example\": \"sample value\"\n }\n }\n }\n },\n \"adaptations\": {\n \"fontSize\": \"compact\" | \"default\" | \"large\",\n \"density\": \"dense\" | \"default\" | \"spacious\",\n \"complexity\": \"simplified\" | \"default\" | \"detailed\"\n }\n}\n\nINTENT RULES (most important):\n- The \"intent\" field captures WHY this UI exists in one sentence.\n- Include: the user's goal (why), what data is shown (what), and how they interact (how).\n- Be abstract enough to match reusable patterns \u2014 \"Display current weather conditions\" not \"Display Tokyo weather at 3pm\".\n- Same intent = same component can be reused with different data.\n- Examples:\n - \"Display current weather conditions for a quick daily check\"\n - \"Collect user feedback via a multi-field survey form\"\n - \"Show real-time stock prices with live updates\"\n - \"Compare two products side by side for purchase decision\"\n\nDECISION RULES:\n- \"create\": No existing UI or blueprint matches this intent. Show something new.\n- \"update\": An existing UI on the stack has the same intent. Update its props.\n- \"compose\": An existing UI could incorporate this data as a section.\n- \"replace\": Two or more related UIs would be better as a single unified view.\n\nREUSE BIAS (critical for performance):\n- Reusing a blueprint = INSTANT render (cached code, <1 second).\n- Creating new = 20+ seconds of generation. The user waits.\n- Default to reuse. Only \"create\" when NO candidate can reasonably serve the request.\n- A candidate that shows the SAME KIND of data (e.g., weather, stock prices, user profiles) is a match \u2014 even if the specific data differs (Tokyo vs Seoul, AAPL vs GOOG).\n- Ask yourself: \"Can this candidate display the agent's data with different prop values?\" If yes \u2192 reuse it.\n- When reusing a blueprint, copy its contract EXACTLY as-is (including its intent). Do NOT rephrase the intent.\n\nCONTRACT RULES:\n- Always include an \"intent\" field \u2014 it's required.\n- If reusing a blueprint: use that blueprint's contract verbatim. Do not modify intent or propsSpec.\n- If no blueprint match: infer propsSpec.properties from the agent's data shape. Each key becomes a prop.\n- Use the data values as examples.\n\nACTION SPEC \u2014 declare interactive affordances WHENEVER you see them in the data:\n- The discrimination is local-state vs persistent-state, NOT \"did the agent declare agentTools\".\n- LOCAL STATE (counter value, theme toggle, slider position, picker selection, search-as-you-type, form draft fields): contextSpec only. NO actionSpec. The slot mirror IS the wire \u2014 the agent observes context via its next ggui_consume.\n- PERSISTENT STATE (items with identity / IDs + mutable fields, draft submissions awaiting save, deletions of agent-owned rows): actionSpec required. Each gesture is a discrete event the agent must witness.\n- Inference signals for persistent state:\n \u2022 Items with `id`/`itemId`/`uuid` fields + boolean toggle fields like `done`/`completed`/`checked`/`pinned`/`enabled` \u2192 toggle action (e.g. `toggleTodo`, `togglePin`).\n \u2022 Lists where the data shape implies the agent maintains identity \u2192 add/delete actions.\n \u2022 Form data with mutable fields + a submit gesture \u2192 submit action.\n \u2022 A click that would cause a server-side side-effect (publish, archive, send, delete-from-database) \u2192 action.\n- nextStep is OPTIONAL on actionSpec entries:\n \u2022 Bind `nextStep: \"tool_name\"` ONLY when input.agentTools contains a matching tool (exact or close name match \u2014 `todo_toggle` matches the `toggleTodo` action).\n \u2022 If no agentTools match, OMIT nextStep. Events drain via ggui_consume on the agent's next turn \u2014 the agent's reasoning loop sees the event and decides what to do.\n- Examples:\n \u2022 agentTools=[\"todo_toggle\",\"todo_delete\"] + todo data \u2192 actionSpec: { \"toggleTodo\": { label: \"Toggle\", schema: {type:\"object\",properties:{id:{type:\"string\"}},required:[\"id\"]}, nextStep: \"todo_toggle\" }, \"deleteTodo\": {...nextStep: \"todo_delete\"} }\n \u2022 agentTools=[] + todo data \u2192 SAME actionSpec entries WITHOUT nextStep. The agent's next turn reads the event and reacts.\n \u2022 agentTools=[] + counter prompt \u2192 contextSpec.count only. NO actionSpec.\n\nAGENT CAPABILITIES (catalog):\n- You do NOT need to emit contract.agentCapabilities \u2014 it is populated deterministically from input.agentTools after you return.\n- Focus on actionSpec entries + their optional nextStep bindings; the catalog auto-populates.\n\nANTI-PATTERNS \u2014 DO NOT EMIT (cross-ref linter rejects at push):\n- \"props\" / \"props.properties\" as a CONTRACT field (retired contract-side spelling \u2014 the contract field is propsSpec; the wire field on push/update is still \"props\" but carries VALUES, not the spec)\n- \"wiredTools\" / \"agentTools\" / \"clientTools\" catalog names (retired; use agentCapabilities.tools / clientCapabilities.gadgets)\n- clientCapabilities.capabilities (retired inner key; use clientCapabilities.gadgets)\n- ActionEntry.tool / ActionEntry.dispatch.kind (retired discriminated union; use the flat nextStep field)\n- mode: 'host-routed' / mode: 'agent-routed' (retired; all actions are agent-routed)\n- broadcast: { ... } as a top-level field (retired; use streamSpec[X].source instead)";
48
+ export declare function buildDecisionUserMessage(input: NegotiatorDecisionInput): string;
49
+ /** Make the UI decision via one LLM call with all context. */
50
+ export declare function makeDecision(input: NegotiatorDecisionInput, llmCaller: LLMCaller): Promise<{
51
+ decision: NegotiatorDecision;
52
+ alternatives: NegotiatorAlternative[];
53
+ }>;
54
+ //# sourceMappingURL=decision.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"decision.d.ts","sourceRoot":"","sources":["../src/decision.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA0CG;AAEH,OAAO,KAAK,EAQV,qBAAqB,EACrB,kBAAkB,EACnB,MAAM,mBAAmB,CAAC;AAE3B,OAAO,KAAK,EAAE,uBAAuB,EAAE,MAAM,qBAAqB,CAAC;AACnE,OAAO,KAAK,EAAE,SAAS,EAAE,MAAM,iBAAiB,CAAC;AAGjD,eAAO,MAAM,sBAAsB,8nMAsFiE,CAAC;AAErG,wBAAgB,wBAAwB,CAAC,KAAK,EAAE,uBAAuB,GAAG,MAAM,CAwG/E;AAuGD,8DAA8D;AAC9D,wBAAsB,YAAY,CAChC,KAAK,EAAE,uBAAuB,EAC9B,SAAS,EAAE,SAAS,GACnB,OAAO,CAAC;IAAE,QAAQ,EAAE,kBAAkB,CAAC;IAAC,YAAY,EAAE,qBAAqB,EAAE,CAAA;CAAE,CAAC,CAkElF"}