@ggui-ai/negotiator 0.2.0-alpha.3 → 0.2.0-alpha.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/dist/ensure-conforming-contract.d.ts +70 -0
  2. package/dist/ensure-conforming-contract.d.ts.map +1 -0
  3. package/dist/ensure-conforming-contract.js +115 -0
  4. package/dist/index.d.ts +2 -0
  5. package/dist/index.d.ts.map +1 -1
  6. package/dist/index.js +1 -0
  7. package/dist/normalize-draft.d.ts +33 -0
  8. package/dist/normalize-draft.d.ts.map +1 -0
  9. package/dist/normalize-draft.js +143 -0
  10. package/dist/preserve-seed-surfaces.d.ts +40 -0
  11. package/dist/preserve-seed-surfaces.d.ts.map +1 -0
  12. package/dist/preserve-seed-surfaces.js +57 -0
  13. package/dist/synth-bench/cli-llm.d.ts +20 -0
  14. package/dist/synth-bench/cli-llm.d.ts.map +1 -0
  15. package/dist/synth-bench/cli-llm.js +97 -0
  16. package/dist/synth-bench/corpus.d.ts +52 -0
  17. package/dist/synth-bench/corpus.d.ts.map +1 -1
  18. package/dist/synth-bench/corpus.js +306 -5
  19. package/dist/synth-bench/round-trip-score.d.ts +87 -0
  20. package/dist/synth-bench/round-trip-score.d.ts.map +1 -0
  21. package/dist/synth-bench/round-trip-score.js +105 -0
  22. package/dist/synth-bench/run-bench-cli.js +6 -82
  23. package/dist/synth-bench/run-repair-bench-cli.d.ts +3 -0
  24. package/dist/synth-bench/run-repair-bench-cli.d.ts.map +1 -0
  25. package/dist/synth-bench/run-repair-bench-cli.js +86 -0
  26. package/dist/synth-bench/run-repair-bench.d.ts +94 -0
  27. package/dist/synth-bench/run-repair-bench.d.ts.map +1 -0
  28. package/dist/synth-bench/run-repair-bench.js +172 -0
  29. package/dist/synthesize-contract.d.ts +38 -6
  30. package/dist/synthesize-contract.d.ts.map +1 -1
  31. package/dist/synthesize-contract.js +246 -32
  32. package/package.json +5 -4
  33. package/src/ensure-conforming-contract.ts +175 -0
  34. package/src/index.ts +2 -0
  35. package/src/normalize-draft.ts +156 -0
  36. package/src/preserve-seed-surfaces.ts +61 -0
  37. package/src/synth-bench/cli-llm.ts +140 -0
  38. package/src/synth-bench/corpus.ts +335 -5
  39. package/src/synth-bench/round-trip-score.ts +169 -0
  40. package/src/synth-bench/run-bench-cli.ts +13 -115
  41. package/src/synth-bench/run-repair-bench-cli.ts +119 -0
  42. package/src/synth-bench/run-repair-bench.ts +266 -0
  43. package/src/synthesize-contract.ts +299 -37
@@ -24,10 +24,12 @@
24
24
  * a richer surface should author the contract themselves on the
25
25
  * handshake input; synthesis is a fallback, not a replacement.
26
26
  *
27
- * **Failure modes collapse to null.** LLM throws, parse fails,
28
- * provider doesn't support `callStructured` → return `null`. Caller
29
- * falls back to an empty stub; behavior regresses to pre-synth but
30
- * doesn't crash.
27
+ * **Failure modes collapse to null.** LLM throws on every attempt or
28
+ * the bounded repair budget is exhausted → return `null`. Caller falls
29
+ * back to an empty stub; behavior regresses to pre-synth but doesn't
30
+ * crash. Providers without `callStructured` (gemini / openai /
31
+ * openrouter) use a text-JSON fallback rather than skipping synthesis,
32
+ * so repair works on every provider.
31
33
  *
32
34
  * **Cost.** ~$0.0005-0.001 per call (Haiku 4.5, ~500 input + ~300
33
35
  * output tokens). Latency ~1.5s. Fires only on cold-path Tier 3
@@ -37,6 +39,7 @@
37
39
  import { dataContractSchema, gadgetExportName, } from '@ggui-ai/protocol';
38
40
  import { lintContract } from '@ggui-ai/protocol';
39
41
  import { normalizeSchema } from './normalize-schema.js';
42
+ import { draftSeedPropKeys, findDroppedSeedSurfaces, } from './preserve-seed-surfaces.js';
40
43
  import { formatValidationFindings, validateActionsVsContext, validateContractCoherence, validateContractStructure, } from './contract-validators.js';
41
44
  /**
42
45
  * Map a protocol-linter {@link ContractIssue} into the negotiator's
@@ -73,14 +76,14 @@ A contract has FOUR specs that describe distinct directions on the wire between
73
76
 
74
77
  THE FOUR-SPEC MODEL
75
78
 
76
- propsSpec (agent → UI, render-time)
77
- Static initial-render data the agent passes once when the UI is mounted. The UI reads it; it never changes after mount. Use ONLY when the intent names data the component cannot render without (a weather card needs city + temp; a profile needs name + avatar). Omit when the UI generates its own state (a counter starting at zero, a blank notepad).
79
+ propsSpec (agent → UI, render-time + agent-pushed refreshes)
80
+ Data the AGENT owns and supplies — the initial values at mount, AND every later refresh via ggui_update. NOT "static / never changes": propsSpec is the ONLY channel for agent-owned data, mutable or not. A weather card's city+temp (fixed) AND the items of the todo list the agent fetched and keeps in sync (mutable) BOTH live here — the agent seeds them at render and pushes each change with ggui_update. Use whenever the agent is the SOURCE of what the UI shows: the intent names data the agent provides / fetches / owns ("my todos", "the cart", "this user's profile", "the directory contents") OR data the component cannot render without (city, temp). Omit only when the UI originates its own state with no agent-supplied contents (a counter starting at zero, a blank notepad, a list the USER builds locally).
78
81
 
79
82
  streamSpec (agent → UI, live, append-only)
80
83
  Channels where the agent pushes live data the UI displays as it arrives. Use ONLY when the intent describes ongoing agent-originated updates (a chat with messages, a live dashboard, a clock, a stock ticker, a notifications feed). Wrong instinct: do NOT use streamSpec for user-driven state, nor for a multi-step wizard / tutorial — its steps are a local stepper plus component-authored copy, not an agent-pushed feed.
81
84
 
82
85
  contextSpec (UI → agent, live, debounced mirror)
83
- Client state the agent OBSERVES continuously. The UI mutates each slot via a setter; the runtime mirrors the value back to the agent. Use for any client-side state whose CURRENT VALUE is what the agent cares about — a counter's count, a form's draft fields, a slider's position, a selected tab, a search query as the user types. The mirror is the wire path: the agent already sees every change.
86
+ Client state the agent OBSERVES continuously. The UI mutates each slot via a setter; the runtime mirrors the value back to the agent. Use for any CLIENT-ORIGINATED state whose CURRENT VALUE is what the agent cares about — a counter's count, a form's draft fields, a slider's position, a selected tab, a search query as the user types. The mirror is the wire path: the agent already sees every change. Ownership boundary: contextSpec is client→agent ONLY — the agent can READ the mirror but CANNOT push values into it (there is no agent→contextSpec channel). Data the AGENT owns or seeds belongs on propsSpec, never here — a collection the agent fetched and must keep in sync placed on contextSpec can never be seeded or updated, and the UI renders empty.
84
87
 
85
88
  actionSpec (UI → agent, one-shot event)
86
89
  Discrete events the agent must WITNESS — a single point in time the agent receives a payload describing what happened. Use for events with semantic meaning beyond the current state of any slot: submit, save, send, finalize, navigate, confirm, cancel, search, delete-by-id. The payload carries the data the agent needs to act on the event.
@@ -146,10 +149,18 @@ CONCRETE PATTERNS
146
149
  contextSpec: { query: {schema: {type: "string"}, default: ""} }
147
150
  Whether to ALSO declare a submit-search action depends on whether the agent acts on every keystroke (no action — the mirror IS the wire) or only on enter/click (declare a search action). Default to no action unless the intent names "search button" / "submit on enter". When declaring, pass the query as payload: schema: {type: "object", properties: {query: {type: "string"}}, required: ["query"]}.
148
151
 
149
- Todo list — "a todo list", "an agent-backed todo list that persists across sessions"
150
- The todos array is client state that mutates (add / delete / toggle) — it is ALWAYS a contextSpec slot, NEVER propsSpec. propsSpec is for data that never changes after mount; a todo list's items change constantly.
151
- contextSpec: { todos: {schema: {type: "array"}, default: []} }
152
- actionSpec: OMIT by default — a local-only list just lets the agent observe the items slot. Declare addTodo / deleteTodo ONLY when the intent says the agent must act on each add/delete ("agent-backed", "synced", "persists across sessions") — those words mean each mutation IS a discrete event the agent witnesses, layered ON TOP of the contextSpec slot, not instead of it.
152
+ Todo list / collection — split on OWNERSHIP, not on mutability. Both kinds mutate; what differs is WHO supplies the items.
153
+
154
+ (a) Agent-owned — "show my todos", "an agent-backed todo list that persists across sessions", "render my cart", "the messages in this thread", "the directory contents"
155
+ The AGENT owns the items: it fetched / persists / keeps them in sync. The collection is the agent's data → it goes on PROPSSPEC, seeded at render and refreshed via ggui_update after each change. This is the ONLY shape that round-trips — contextSpec has no agent-push channel, so an agent-owned list placed there can never be seeded or updated (the UI renders empty). add / delete / toggle are discrete events the agent must witness to persist → declare them on actionSpec (with a matching agentCapabilities tool for each nextStep). Mutability is fine: ggui_update is exactly how the agent pushes the change.
156
+ propsSpec: { properties: { todos: {schema: {type: "array", items: {type: "object", properties: {id: {type: "string"}, text: {type: "string"}, done: {type: "boolean"}}, required: ["id", "text", "done"]}}, required: true} } }
157
+ actionSpec: { toggleTodo: {label: "Toggle todo", schema: {type: "object", properties: {id: {type: "string"}}, required: ["id"]}, nextStep: "todo_toggle"}, addTodo: {label: "Add todo", schema: {type: "object", properties: {text: {type: "string"}}, required: ["text"]}, nextStep: "todo_add"} }
158
+
159
+ (b) User-built local — "a todo list where I can add and remove items", "a shopping list", "a checklist I tick off"
160
+ No agent-owned source: the USER assembles the list in the UI and the agent merely observes it. The items are client-originated state → a CONTEXTSPEC slot; OMIT actionSpec (the slot mirror IS the wire — the agent already sees every change). Only when the intent says the agent must persist / sync each change does it become the agent-owned case (a) above.
161
+ contextSpec: { todos: {schema: {type: "array"}, default: []} }
162
+
163
+ Tell them apart by the SOURCE of the initial items: "my / the / show / render / persisted / agent-backed / synced" → the agent has data to seed → (a) propsSpec. "a / let me build / I add" with no agent source → the user builds it → (b) contextSpec.
153
164
 
154
165
  Confirmation modal — "a delete-confirmation modal with confirm and cancel actions"
155
166
  Confirm and cancel are the discrete events the agent must witness; the modal holds no client state and no live feed. Declare propsSpec ONLY when the intent NAMES the item / data the modal shows ("confirm deleting <the file name>"); a generic confirmation modal that names no data field has NO propsSpec.
@@ -350,10 +361,10 @@ export const SYNTHESIZE_TOOL = {
350
361
  usage: { type: 'string' },
351
362
  },
352
363
  },
353
- description: 'Per-tool map: name → {inputSchema?, outputSchema?, usage?}. Catalog only — the agent owns invocation. Declare entries that are referenced via streamSpec[X].source.tool.',
364
+ description: 'Per-tool map: name → {inputSchema?, outputSchema?, usage?}. Catalog only — the agent owns invocation. Declare an entry for EVERY tool a nextStep / source points at: each actionSpec[X].nextStep AND each streamSpec[X].source.tool — referencing an undeclared tool fails the cross-reference check.',
354
365
  },
355
366
  },
356
- description: 'Catalog of agent-invoked tools the contract references. The component code does NOT call these directly. Required when any streamSpec.source.tool refers to a tool name.',
367
+ description: 'Catalog of agent-invoked tools the contract references. The component code does NOT call these directly. Required whenever any actionSpec.nextStep or streamSpec.source.tool names a tool.',
357
368
  },
358
369
  clientCapabilities: {
359
370
  type: 'object',
@@ -398,7 +409,7 @@ export const SYNTHESIZE_TOOL = {
398
409
  description: 'Per-prop map: name → {schema, required?}. Declares the initial render data the agent passes at push time.',
399
410
  },
400
411
  },
401
- description: 'Static-display data the agent passes at push time. Use ONLY when the intent names initial data fields (weather card → city/temp; profile → name/avatar). Omit when the UI generates its own state.',
412
+ description: 'Agent-OWNED data the UI displays — the initial values seeded at render, refreshed any time after via ggui_update. NOT static-only: mutable collections the agent owns / fetched / keeps in sync (my todos, the cart, this thread\'s messages, a directory listing) go here too — propsSpec is the ONLY agent→client data channel. Use whenever the agent is the SOURCE of the displayed data (weather card → city/temp; profile → name/avatar; "my todos" → todos). Omit only when the UI originates its own state with no agent-supplied contents (a counter, a blank notepad, a list the user builds locally).',
402
413
  },
403
414
  reason: {
404
415
  type: 'string',
@@ -435,6 +446,167 @@ function buildRepairNote(rejected, failure) {
435
446
  'Emit a corrected contract that fixes exactly this problem. Keep every other spec unchanged.',
436
447
  ].join('\n');
437
448
  }
449
+ /**
450
+ * PATCH-MODE preamble (L2) — prepended to the system prompt ONLY on the
451
+ * repair path (a draft is present). The cold-path system prompt is an
452
+ * "infer a contract from intent" authoring brief; reusing it verbatim
453
+ * for repair invites the model to RE-AUTHOR and reshape a near-correct
454
+ * draft (e.g. move a propsSpec collection to contextSpec). This preamble
455
+ * reframes the task as a minimal patch, which — together with the
456
+ * deterministic preservation gate — keeps the repair faithful.
457
+ */
458
+ const REPAIR_PREAMBLE = `PATCH MODE — you are REPAIRING a contract the agent already authored, NOT writing a new one from scratch.
459
+ - Change as LITTLE as possible. Fix ONLY the specific findings listed in the user message.
460
+ - PRESERVE every spec the agent declared — ESPECIALLY every propsSpec property (agent-owned render-time seed data the UI needs). Never drop it.
461
+ - Do NOT move data between specs (e.g. propsSpec → contextSpec) unless a finding explicitly requires it. A collection the agent supplies on propsSpec STAYS on propsSpec.
462
+ - Keep the agent's names, shapes, and structure intact wherever the findings do not force a change.
463
+ The four-spec model and placement rules below still hold — but in PATCH MODE they are guardrails for the fix, not a license to re-author.`;
464
+ /**
465
+ * Preservation-failure repair note (L1) — drives a corrective retry when
466
+ * a VALID candidate dropped an agent-owned propsSpec seed surface. Names
467
+ * the exact missing keys so the model restores them, rather than the
468
+ * generic "fix this validation error" note (the candidate IS valid; the
469
+ * problem is round-trip fidelity the gate can't express).
470
+ */
471
+ function buildPreservationRepairNote(rejected, dropped) {
472
+ const json = JSON.stringify(rejected);
473
+ const capped = json.length > 3000 ? `${json.slice(0, 3000)}…` : json;
474
+ return [
475
+ 'YOUR PREVIOUS ATTEMPT dropped agent-owned render-time data. You returned this contract:',
476
+ capped,
477
+ '',
478
+ `The agent's draft declared these on propsSpec (agent-owned seed data the UI renders): ${dropped.join(', ')}. Your contract no longer carries them as propsSpec properties — so the agent can no longer seed them at render. contextSpec has NO agent seed channel, so moving them there leaves the UI empty.`,
479
+ '',
480
+ `Re-emit the contract with ${dropped.join(', ')} restored as propsSpec properties (agent-owned, seeded at render and refreshed via ggui_update). Keep every other spec unchanged.`,
481
+ ].join('\n');
482
+ }
483
+ /**
484
+ * Compose the FIRST-attempt repair note for the forgiving-handshake
485
+ * path: the agent PROPOSED a contract that failed deterministic
486
+ * validation. Unlike {@link buildRepairNote} (which frames the input as
487
+ * "your previous attempt"), this frames it as the AGENT'S draft to be
488
+ * repaired in place — preserve intent + structure, fix exactly the
489
+ * listed findings. The draft JSON is capped to bound the prompt.
490
+ */
491
+ function buildDraftSeedNote(draft, findings) {
492
+ const json = JSON.stringify(draft);
493
+ const capped = json.length > 3000 ? `${json.slice(0, 3000)}…` : json;
494
+ const findingsText = findings.length > 0
495
+ ? findings.map((f) => `[${f.code}] ${f.path}: ${f.message}`).join('; ')
496
+ : '(unspecified — re-derive a valid contract for the intent)';
497
+ return [
498
+ 'The agent PROPOSED this contract, but it failed deterministic validation:',
499
+ capped,
500
+ '',
501
+ `Validation findings — ${findingsText}`,
502
+ '',
503
+ "Repair it: keep the agent's intent and as much of their structure (spec names, schemas, labels) as possible, fix EXACTLY these findings, and return a valid contract. Do not add unrelated specs.",
504
+ ].join('\n');
505
+ }
506
+ /**
507
+ * Invoke the synthesize tool — structured output when the provider
508
+ * supports it, else a text-JSON fallback. Forced tool use is
509
+ * Anthropic / Bedrock-only today; gemini / openai / openrouter callers
510
+ * reach this via the text path (the model emits one JSON object, which
511
+ * we regex-extract + parse). Lower reliability than forced tool use,
512
+ * but the bounded validate-and-repair loop catches malformed output and
513
+ * retries — so repair works on every provider instead of no-op'ing
514
+ * off-Anthropic.
515
+ */
516
+ async function callSynthesizeTool(llm, system, user) {
517
+ if (typeof llm.callStructured === 'function') {
518
+ return llm.callStructured(system, user, SYNTHESIZE_TOOL, 1024);
519
+ }
520
+ const text = await llm.call(system, `${user}\n\nRespond with ONE JSON object only (no prose, no code fence) carrying the synthesized contract: {actionSpec?, contextSpec?, streamSpec?, propsSpec?, agentCapabilities?, clientCapabilities?, reason}. Every spec entry wraps its JSON Schema under a "schema" field.`, 1024);
521
+ return extractJsonObject(text);
522
+ }
523
+ /**
524
+ * Thrown by {@link extractJsonObject} when the text-fallback response
525
+ * carries no parseable JSON object — distinct from a transient network
526
+ * throw so the repair loop knows to feed a corrective "emit pure JSON"
527
+ * note (a network retry re-sends the same prompt instead).
528
+ */
529
+ class SynthesizeTextParseError extends Error {
530
+ constructor(message) {
531
+ super(message);
532
+ this.name = 'SynthesizeTextParseError';
533
+ }
534
+ }
535
+ /**
536
+ * Robustly extract one JSON object from a possibly-prose-wrapped model
537
+ * response. A naive greedy `/\{[\s\S]*\}/` over-captures (first `{` to
538
+ * LAST `}`), throwing on "prose with a brace before the JSON" or "two
539
+ * objects". This tries, in order: (1) parse the trimmed text directly;
540
+ * (2) parse the contents of a ```json fence; (3) scan from the first `{`
541
+ * for the BALANCED closing `}` (string/escape aware). Throws only when
542
+ * nothing parses — the bounded validate-and-repair loop then retries
543
+ * with a corrective note.
544
+ */
545
+ function extractJsonObject(text) {
546
+ const trimmed = text.trim();
547
+ try {
548
+ return JSON.parse(trimmed);
549
+ }
550
+ catch {
551
+ // fall through
552
+ }
553
+ const fence = trimmed.match(/```(?:json)?\s*([\s\S]*?)```/);
554
+ if (fence && fence[1]) {
555
+ try {
556
+ return JSON.parse(fence[1].trim());
557
+ }
558
+ catch {
559
+ // fall through
560
+ }
561
+ }
562
+ // Scan each balanced {...} from each '{' and return the FIRST that
563
+ // parses — robust against prose braces BEFORE the JSON (e.g. "{your
564
+ // widget}: {...}") and a trailing second object.
565
+ let searchFrom = 0;
566
+ for (;;) {
567
+ const start = trimmed.indexOf('{', searchFrom);
568
+ if (start < 0)
569
+ break;
570
+ let depth = 0;
571
+ let inStr = false;
572
+ let esc = false;
573
+ let end = -1;
574
+ for (let i = start; i < trimmed.length; i++) {
575
+ const ch = trimmed[i];
576
+ if (inStr) {
577
+ if (esc)
578
+ esc = false;
579
+ else if (ch === '\\')
580
+ esc = true;
581
+ else if (ch === '"')
582
+ inStr = false;
583
+ }
584
+ else if (ch === '"') {
585
+ inStr = true;
586
+ }
587
+ else if (ch === '{') {
588
+ depth++;
589
+ }
590
+ else if (ch === '}') {
591
+ depth--;
592
+ if (depth === 0) {
593
+ end = i;
594
+ break;
595
+ }
596
+ }
597
+ }
598
+ if (end < 0)
599
+ break; // unbalanced from here on
600
+ try {
601
+ return JSON.parse(trimmed.slice(start, end + 1));
602
+ }
603
+ catch {
604
+ // not valid JSON from this '{' — advance and try the next one
605
+ }
606
+ searchFrom = start + 1;
607
+ }
608
+ throw new SynthesizeTextParseError(`synthesize text-fallback: no parseable JSON object in response (length ${text.length})`);
609
+ }
438
610
  /**
439
611
  * Drop `actionSpec` entries the structural validator flags as
440
612
  * `redundant-action` — an empty-payload action whose name is a
@@ -477,8 +649,14 @@ function pruneRedundantActions(contract) {
477
649
  * Empty / whitespace intent short-circuits to null with a reason —
478
650
  * no contract can be inferred from nothing.
479
651
  *
480
- * Provider lacking `callStructured` (test stubs, providers without
481
- * tool-use) collapses to null.
652
+ * Providers without `callStructured` (gemini / openai / openrouter) use
653
+ * a text-JSON fallback (the validate-and-repair loop catches malformed
654
+ * output and retries) rather than skipping synthesis.
655
+ *
656
+ * When `options.draft` is supplied, the loop REPAIRS that draft in
657
+ * place (seeded with the agent's contract + the deterministic findings)
658
+ * instead of synthesizing from `intent` alone — the forgiving-handshake
659
+ * path.
482
660
  *
483
661
  * Each attempt is self-checked against the validation gate; a failure
484
662
  * feeds the precise error back for up to {@link MAX_SYNTH_ATTEMPTS}
@@ -498,15 +676,6 @@ export async function synthesizeContract(deps, intent, options) {
498
676
  findings: [],
499
677
  };
500
678
  }
501
- if (typeof deps.llm.callStructured !== 'function') {
502
- return {
503
- contract: null,
504
- reason: 'synthesize-skip: provider does not support callStructured. Bind a structured-capable LLMCaller (Anthropic adapter) to enable contract synthesis.',
505
- latencyMs: Date.now() - startedAt,
506
- attempts: 0,
507
- findings: [],
508
- };
509
- }
510
679
  const gadgetsSection = composeAvailableGadgetsSection(options?.appGadgets);
511
680
  const baseUserPrompt = `INTENT: ${trimmed}${gadgetsSection ? `\n\n${gadgetsSection}` : ''}`;
512
681
  // Bounded validate-and-repair loop. Each attempt is self-checked
@@ -519,21 +688,43 @@ export async function synthesizeContract(deps, intent, options) {
519
688
  // small enough that a full re-emit IS the surgical edit. Budget
520
689
  // exhausted → decline exactly as the one-shot path did (caller falls
521
690
  // back to an empty contract stub).
522
- let repairNote;
691
+ // Repair-in-place seed: when the caller passed the agent's rejected
692
+ // draft, the loop's FIRST attempt repairs that draft (not a blank
693
+ // synthesis). Later attempts overwrite this via buildRepairNote.
694
+ let repairNote = options?.draft !== undefined
695
+ ? buildDraftSeedNote(options.draft, options.draftFindings ?? [])
696
+ : undefined;
523
697
  let lastReason = 'synthesize-fail: exhausted repair attempts';
524
698
  let lastFindings = [];
699
+ // L2 — repair path uses the PATCH-MODE preamble (minimal patch, no
700
+ // re-author); cold path uses the bare authoring prompt.
701
+ const systemPrompt = options?.draft !== undefined
702
+ ? `${REPAIR_PREAMBLE}\n\n${SYNTHESIZE_SYSTEM_PROMPT}`
703
+ : SYNTHESIZE_SYSTEM_PROMPT;
704
+ // L1 — the agent-owned seed surfaces the repaired contract MUST keep,
705
+ // and the best valid-but-non-preserving candidate to fall back on so
706
+ // preservation never makes the result WORSE than the validity-only gate.
707
+ const draftSeedKeys = options?.draft !== undefined ? draftSeedPropKeys(options.draft) : [];
708
+ let lastValidContract = null;
525
709
  for (let attempt = 1; attempt <= MAX_SYNTH_ATTEMPTS; attempt++) {
526
710
  const userPrompt = repairNote === undefined
527
711
  ? baseUserPrompt
528
712
  : `${baseUserPrompt}\n\n${repairNote}`;
529
713
  let toolInput;
530
714
  try {
531
- toolInput = await deps.llm.callStructured(SYNTHESIZE_SYSTEM_PROMPT, userPrompt, SYNTHESIZE_TOOL, 1024);
715
+ toolInput = await callSynthesizeTool(deps.llm, systemPrompt, userPrompt);
532
716
  }
533
717
  catch (err) {
534
- // Transient (network) failure — retry the same prompt; the
535
- // attempt produced nothing to correct, so no repair note.
536
- lastReason = `synthesize-fail: callStructured threw — ${err instanceof Error ? err.message : String(err)}`;
718
+ lastReason = `synthesize-fail: callSynthesizeTool threw — ${err instanceof Error ? err.message : String(err)}`;
719
+ // Distinguish causes: an UNPARSEABLE text-path response (model
720
+ // emitted prose/non-JSON — common on gemini/openai via the text
721
+ // fallback) gets a corrective note so the next attempt emits pure
722
+ // JSON. A transient NETWORK failure re-sends the SAME prompt
723
+ // (nothing to correct) — preserving the retry semantics.
724
+ if (err instanceof SynthesizeTextParseError) {
725
+ repairNote =
726
+ 'Your previous response could not be parsed as JSON. Respond with EXACTLY ONE JSON object and nothing else — no prose, no explanation, no markdown code fence.';
727
+ }
537
728
  continue;
538
729
  }
539
730
  const parsed = parseToolInput(toolInput);
@@ -595,6 +786,23 @@ export async function synthesizeContract(deps, intent, options) {
595
786
  repairNote = buildRepairNote(validatedContract, `it failed contract validation: ${errText}`);
596
787
  continue;
597
788
  }
789
+ // L1 — preservation gate (best-effort). The candidate is VALID, but
790
+ // a repair must not silently drop an agent-owned seed surface the
791
+ // draft declared on propsSpec (the valid-but-round-trip-broken
792
+ // reshape lintContract + the placement validators can't see).
793
+ // Deterministic + model-independent: it drives a corrective retry.
794
+ if (draftSeedKeys.length > 0) {
795
+ const dropped = findDroppedSeedSurfaces(options?.draft, validatedContract);
796
+ if (dropped.length > 0) {
797
+ // Remember the best VALID candidate so an exhausted budget never
798
+ // returns WORSE than the validity-only gate did (a valid contract).
799
+ lastValidContract = validatedContract;
800
+ lastReason = `synthesize-preservation: candidate dropped agent-owned propsSpec seed surface(s) [${dropped.join(', ')}]`;
801
+ lastFindings = allFindings;
802
+ repairNote = buildPreservationRepairNote(validatedContract, dropped);
803
+ continue;
804
+ }
805
+ }
598
806
  const findingsSuffix = allFindings.length > 0
599
807
  ? ` — validator: ${formatValidationFindings({ findings: allFindings })}`
600
808
  : '';
@@ -607,9 +815,15 @@ export async function synthesizeContract(deps, intent, options) {
607
815
  findings: allFindings,
608
816
  };
609
817
  }
818
+ // Budget exhausted. If preservation retries never landed a contract
819
+ // that kept every seed surface, fall back to the best VALID candidate
820
+ // we did produce (never worse than the validity-only gate). Only when
821
+ // no valid candidate ever appeared do we decline with `null`.
610
822
  return {
611
- contract: null,
612
- reason: lastReason,
823
+ contract: lastValidContract,
824
+ reason: lastValidContract !== null
825
+ ? `${lastReason}; returned best valid candidate (preservation retries exhausted)`
826
+ : lastReason,
613
827
  latencyMs: Date.now() - startedAt,
614
828
  attempts: MAX_SYNTH_ATTEMPTS,
615
829
  findings: lastFindings,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ggui-ai/negotiator",
3
- "version": "0.2.0-alpha.3",
3
+ "version": "0.2.0-alpha.4",
4
4
  "description": "UI decision engine for ggui. Given an agent's signal and the current render state, decides which UI to render (create / update / replace) and synthesizes the data contract. Deployment-agnostic: concrete embedding and vector-store bindings plug in via the storage interfaces from @ggui-ai/mcp-server-core.",
5
5
  "license": "Apache-2.0",
6
6
  "keywords": [
@@ -47,8 +47,8 @@
47
47
  }
48
48
  },
49
49
  "dependencies": {
50
- "@ggui-ai/mcp-server-core": "0.2.0-alpha.3",
51
- "@ggui-ai/protocol": "0.2.0-alpha.3"
50
+ "@ggui-ai/mcp-server-core": "0.2.0-alpha.4",
51
+ "@ggui-ai/protocol": "0.2.0-alpha.4"
52
52
  },
53
53
  "devDependencies": {
54
54
  "@types/node": "^24.0.0",
@@ -69,6 +69,7 @@
69
69
  "test": "vitest run",
70
70
  "test:watch": "vitest",
71
71
  "probe-rerank": "tsx src/rerank-eval/run-probe-cli.ts",
72
- "bench-synth": "tsx src/synth-bench/run-bench-cli.ts"
72
+ "bench-synth": "tsx src/synth-bench/run-bench-cli.ts",
73
+ "bench-repair": "tsx src/synth-bench/run-repair-bench-cli.ts"
73
74
  }
74
75
  }
@@ -0,0 +1,175 @@
1
+ /**
2
+ * `ensureConformingContract` — the negotiator's create-path guarantee.
3
+ *
4
+ * Given the agent's PROPOSED draft, return a contract that is
5
+ * GUARANTEED to pass the deterministic gate (`lintContract` with zero
6
+ * errors), so the handshake backstop (`validateContract`) never throws
7
+ * on it. This is the "Propose vs Commit" forgiving-handshake core
8
+ * shared by every negotiator implementation (OSS llm-backed + cloud
9
+ * bedrock) so the behavior cannot drift between deployments:
10
+ *
11
+ * - draft already conforms → return it verbatim (origin: 'agent')
12
+ * - draft has errors → repair-in-place via the bounded LLM loop,
13
+ * seeded with the draft + the deterministic
14
+ * findings, looping until the gate is green
15
+ * (origin: 'synth')
16
+ * - repair impossible → minimal conforming contract (`{}`) + loud
17
+ * (LLM down / provider error findings; STILL origin 'synth';
18
+ * can't synth / budget NEVER throws.
19
+ * exhausted)
20
+ *
21
+ * Determinism lives in the GATE (`lintContract`), never in the repair.
22
+ * The repair LLM is non-deterministic, but the loop only exits when the
23
+ * deterministic gate is green — the same shape ui-gen uses to tolerate
24
+ * non-deterministic code generation behind a deterministic self_check.
25
+ *
26
+ * Cache/blueprint matching is NOT this function's job — the caller
27
+ * (negotiator `decide()`) runs its deployment-specific cache match
28
+ * FIRST and only falls through to here on a miss. That preserves the
29
+ * "cache-first, repair-second" ordering the negotiator contract
30
+ * mandates.
31
+ */
32
+ import {
33
+ lintContract,
34
+ dataContractSchema,
35
+ type DataContract,
36
+ type GadgetDescriptor,
37
+ type SuggestionFinding,
38
+ } from '@ggui-ai/protocol';
39
+ import type { LLMCaller } from './llm-caller.js';
40
+ import { synthesizeContract } from './synthesize-contract.js';
41
+ import { normalizeDraft } from './normalize-draft.js';
42
+
43
+ export interface EnsureConformingResult {
44
+ /** A contract guaranteed to pass `lintContract` with zero errors. */
45
+ readonly contract: DataContract;
46
+ /**
47
+ * - `'agent'` — the draft was already conforming; returned verbatim.
48
+ * - `'synth'` — the draft had errors; this is the repaired result
49
+ * (or the minimal-conforming fallback when repair was impossible).
50
+ */
51
+ readonly origin: 'agent' | 'synth';
52
+ /**
53
+ * How the conforming contract was produced — finer-grained than
54
+ * `origin`, for telemetry (the efficiency tiers):
55
+ * - `verbatim` — draft was clean; returned as-is (origin agent).
56
+ * - `normalized` — deterministic fix only, NO LLM (origin synth).
57
+ * - `llm-repair` — the bounded LLM repair loop ran (origin synth).
58
+ * - `fallback-empty`— unrepairable; minimal `{}` contract (origin synth).
59
+ */
60
+ readonly method: 'verbatim' | 'normalized' | 'llm-repair' | 'fallback-empty';
61
+ /**
62
+ * Findings surfaced to the agent. On `origin: 'agent'`, any hygiene
63
+ * warnings on the (valid) draft. On `origin: 'synth'`, the ERROR
64
+ * findings that rejected the agent's draft — so the agent-side model
65
+ * learns what it got wrong, even though we repaired it.
66
+ */
67
+ readonly findings: readonly SuggestionFinding[];
68
+ /** Operator- + LLM-readable explanation. */
69
+ readonly reasoning: string;
70
+ }
71
+
72
+ /** Trivially-valid last-resort contract — all four specs omitted. */
73
+ const EMPTY_CONTRACT: DataContract = {};
74
+
75
+ export async function ensureConformingContract(
76
+ deps: { readonly llm: LLMCaller },
77
+ args: {
78
+ /** Untrusted: the agent's draft may not be a valid DataContract. */
79
+ readonly draft: unknown;
80
+ readonly intent: string;
81
+ readonly appGadgets?: readonly GadgetDescriptor[];
82
+ },
83
+ ): Promise<EnsureConformingResult> {
84
+ const lint = lintContract(args.draft);
85
+ const warnFindings: SuggestionFinding[] = lint.warnings.map(
86
+ (w): SuggestionFinding => ({
87
+ code: w.code,
88
+ severity: 'warn',
89
+ path: w.path,
90
+ message: w.message,
91
+ }),
92
+ );
93
+
94
+ // Fast path — draft already conforms. Deterministic, no LLM call.
95
+ // `lint.errors.length === 0` implies the shape phase passed, so the
96
+ // strict parse cannot throw — it just re-derives the typed DataContract
97
+ // from the untrusted input (validator returns the typed shape; no cast).
98
+ if (lint.errors.length === 0) {
99
+ return {
100
+ contract: dataContractSchema.parse(args.draft),
101
+ origin: 'agent',
102
+ method: 'verbatim',
103
+ findings: warnFindings,
104
+ reasoning:
105
+ 'agent draft passed validateContract; accepted verbatim (origin: agent)',
106
+ };
107
+ }
108
+
109
+ const errorFindings: SuggestionFinding[] = lint.errors.map(
110
+ (e): SuggestionFinding => ({
111
+ code: e.code,
112
+ severity: 'error',
113
+ path: e.path,
114
+ message: e.message,
115
+ }),
116
+ );
117
+
118
+ // L3 — deterministic normalization tier. Most agent malformations are
119
+ // mechanical (stray illegal wrapper keys, non-canonical schema types).
120
+ // Fix them WITHOUT an LLM: strip + canonicalize, re-lint, and if the
121
+ // draft now conforms, return it verbatim-but-cleaned. Faithful (no
122
+ // reshape risk — the agent's specs are preserved exactly) and free (no
123
+ // LLM call). Semantic deficiencies fall through to the repair loop.
124
+ const normalized = normalizeDraft(args.draft);
125
+ const normLint = lintContract(normalized);
126
+ if (normLint.errors.length === 0) {
127
+ return {
128
+ contract: dataContractSchema.parse(normalized),
129
+ origin: 'synth',
130
+ method: 'normalized',
131
+ findings: errorFindings,
132
+ reasoning:
133
+ 'normalized the agent draft deterministically (stripped invalid keys / canonicalized schema types, no LLM) to pass validateContract',
134
+ };
135
+ }
136
+
137
+ // Repair loop on the NORMALIZED draft (mechanical errors already
138
+ // fixed) with only the REMAINING (semantic) findings — so the LLM
139
+ // patches what reasoning is genuinely needed for, from a clean start.
140
+ const synth = await synthesizeContract(deps, args.intent, {
141
+ ...(args.appGadgets ? { appGadgets: args.appGadgets } : {}),
142
+ draft: normalized,
143
+ draftFindings: normLint.errors.map((e) => ({
144
+ code: e.code,
145
+ path: e.path,
146
+ message: e.message,
147
+ })),
148
+ });
149
+
150
+ if (
151
+ synth.contract !== null &&
152
+ lintContract(synth.contract).errors.length === 0
153
+ ) {
154
+ return {
155
+ contract: synth.contract,
156
+ origin: 'synth',
157
+ method: 'llm-repair',
158
+ findings: errorFindings,
159
+ reasoning: `repaired the agent draft to pass validateContract — ${synth.reason}`,
160
+ };
161
+ }
162
+
163
+ // Repair impossible (LLM down, provider can't synthesize, or the
164
+ // repair budget exhausted). We still MUST return a conforming
165
+ // contract — the handshake never hard-fails. Minimal conforming
166
+ // contract + loud findings so the agent can re-issue a corrected
167
+ // contract via ggui_render override if it needs the declared specs.
168
+ return {
169
+ contract: EMPTY_CONTRACT,
170
+ origin: 'synth',
171
+ method: 'fallback-empty',
172
+ findings: errorFindings,
173
+ reasoning: `could not repair the agent draft within budget (${synth.reason}); returning a minimal conforming contract — re-issue a corrected contract via ggui_render override if you need the declared specs`,
174
+ };
175
+ }
package/src/index.ts CHANGED
@@ -49,6 +49,8 @@ export type {
49
49
  } from './llm-rerank.js';
50
50
  export { synthesizeContract } from './synthesize-contract.js';
51
51
  export type { SynthesizeContractResult } from './synthesize-contract.js';
52
+ export { ensureConformingContract } from './ensure-conforming-contract.js';
53
+ export type { EnsureConformingResult } from './ensure-conforming-contract.js';
52
54
  export {
53
55
  validateContractStructure,
54
56
  validateContractNovelty,