@ggui-ai/negotiator 0.2.0-alpha.3 → 0.2.0-alpha.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ensure-conforming-contract.d.ts +70 -0
- package/dist/ensure-conforming-contract.d.ts.map +1 -0
- package/dist/ensure-conforming-contract.js +115 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/normalize-draft.d.ts +33 -0
- package/dist/normalize-draft.d.ts.map +1 -0
- package/dist/normalize-draft.js +143 -0
- package/dist/preserve-seed-surfaces.d.ts +40 -0
- package/dist/preserve-seed-surfaces.d.ts.map +1 -0
- package/dist/preserve-seed-surfaces.js +57 -0
- package/dist/synth-bench/cli-llm.d.ts +20 -0
- package/dist/synth-bench/cli-llm.d.ts.map +1 -0
- package/dist/synth-bench/cli-llm.js +97 -0
- package/dist/synth-bench/corpus.d.ts +52 -0
- package/dist/synth-bench/corpus.d.ts.map +1 -1
- package/dist/synth-bench/corpus.js +306 -5
- package/dist/synth-bench/round-trip-score.d.ts +87 -0
- package/dist/synth-bench/round-trip-score.d.ts.map +1 -0
- package/dist/synth-bench/round-trip-score.js +105 -0
- package/dist/synth-bench/run-bench-cli.js +6 -82
- package/dist/synth-bench/run-repair-bench-cli.d.ts +3 -0
- package/dist/synth-bench/run-repair-bench-cli.d.ts.map +1 -0
- package/dist/synth-bench/run-repair-bench-cli.js +86 -0
- package/dist/synth-bench/run-repair-bench.d.ts +94 -0
- package/dist/synth-bench/run-repair-bench.d.ts.map +1 -0
- package/dist/synth-bench/run-repair-bench.js +172 -0
- package/dist/synthesize-contract.d.ts +38 -6
- package/dist/synthesize-contract.d.ts.map +1 -1
- package/dist/synthesize-contract.js +246 -32
- package/package.json +5 -4
- package/src/ensure-conforming-contract.ts +175 -0
- package/src/index.ts +2 -0
- package/src/normalize-draft.ts +156 -0
- package/src/preserve-seed-surfaces.ts +61 -0
- package/src/synth-bench/cli-llm.ts +140 -0
- package/src/synth-bench/corpus.ts +335 -5
- package/src/synth-bench/round-trip-score.ts +169 -0
- package/src/synth-bench/run-bench-cli.ts +13 -115
- package/src/synth-bench/run-repair-bench-cli.ts +119 -0
- package/src/synth-bench/run-repair-bench.ts +266 -0
- package/src/synthesize-contract.ts +299 -37
|
@@ -24,10 +24,12 @@
|
|
|
24
24
|
* a richer surface should author the contract themselves on the
|
|
25
25
|
* handshake input; synthesis is a fallback, not a replacement.
|
|
26
26
|
*
|
|
27
|
-
* **Failure modes collapse to null.** LLM throws
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
27
|
+
* **Failure modes collapse to null.** LLM throws on every attempt or
|
|
28
|
+
* the bounded repair budget is exhausted → return `null`. Caller falls
|
|
29
|
+
* back to an empty stub; behavior regresses to pre-synth but doesn't
|
|
30
|
+
* crash. Providers without `callStructured` (gemini / openai /
|
|
31
|
+
* openrouter) use a text-JSON fallback rather than skipping synthesis,
|
|
32
|
+
* so repair works on every provider.
|
|
31
33
|
*
|
|
32
34
|
* **Cost.** ~$0.0005-0.001 per call (Haiku 4.5, ~500 input + ~300
|
|
33
35
|
* output tokens). Latency ~1.5s. Fires only on cold-path Tier 3
|
|
@@ -37,6 +39,7 @@
|
|
|
37
39
|
import { dataContractSchema, gadgetExportName, } from '@ggui-ai/protocol';
|
|
38
40
|
import { lintContract } from '@ggui-ai/protocol';
|
|
39
41
|
import { normalizeSchema } from './normalize-schema.js';
|
|
42
|
+
import { draftSeedPropKeys, findDroppedSeedSurfaces, } from './preserve-seed-surfaces.js';
|
|
40
43
|
import { formatValidationFindings, validateActionsVsContext, validateContractCoherence, validateContractStructure, } from './contract-validators.js';
|
|
41
44
|
/**
|
|
42
45
|
* Map a protocol-linter {@link ContractIssue} into the negotiator's
|
|
@@ -73,14 +76,14 @@ A contract has FOUR specs that describe distinct directions on the wire between
|
|
|
73
76
|
|
|
74
77
|
THE FOUR-SPEC MODEL
|
|
75
78
|
|
|
76
|
-
propsSpec (agent → UI, render-time)
|
|
77
|
-
|
|
79
|
+
propsSpec (agent → UI, render-time + agent-pushed refreshes)
|
|
80
|
+
Data the AGENT owns and supplies — the initial values at mount, AND every later refresh via ggui_update. NOT "static / never changes": propsSpec is the ONLY channel for agent-owned data, mutable or not. A weather card's city+temp (fixed) AND the items of the todo list the agent fetched and keeps in sync (mutable) BOTH live here — the agent seeds them at render and pushes each change with ggui_update. Use whenever the agent is the SOURCE of what the UI shows: the intent names data the agent provides / fetches / owns ("my todos", "the cart", "this user's profile", "the directory contents") OR data the component cannot render without (city, temp). Omit only when the UI originates its own state with no agent-supplied contents (a counter starting at zero, a blank notepad, a list the USER builds locally).
|
|
78
81
|
|
|
79
82
|
streamSpec (agent → UI, live, append-only)
|
|
80
83
|
Channels where the agent pushes live data the UI displays as it arrives. Use ONLY when the intent describes ongoing agent-originated updates (a chat with messages, a live dashboard, a clock, a stock ticker, a notifications feed). Wrong instinct: do NOT use streamSpec for user-driven state, nor for a multi-step wizard / tutorial — its steps are a local stepper plus component-authored copy, not an agent-pushed feed.
|
|
81
84
|
|
|
82
85
|
contextSpec (UI → agent, live, debounced mirror)
|
|
83
|
-
Client state the agent OBSERVES continuously. The UI mutates each slot via a setter; the runtime mirrors the value back to the agent. Use for any
|
|
86
|
+
Client state the agent OBSERVES continuously. The UI mutates each slot via a setter; the runtime mirrors the value back to the agent. Use for any CLIENT-ORIGINATED state whose CURRENT VALUE is what the agent cares about — a counter's count, a form's draft fields, a slider's position, a selected tab, a search query as the user types. The mirror is the wire path: the agent already sees every change. Ownership boundary: contextSpec is client→agent ONLY — the agent can READ the mirror but CANNOT push values into it (there is no agent→contextSpec channel). Data the AGENT owns or seeds belongs on propsSpec, never here — a collection the agent fetched and must keep in sync placed on contextSpec can never be seeded or updated, and the UI renders empty.
|
|
84
87
|
|
|
85
88
|
actionSpec (UI → agent, one-shot event)
|
|
86
89
|
Discrete events the agent must WITNESS — a single point in time the agent receives a payload describing what happened. Use for events with semantic meaning beyond the current state of any slot: submit, save, send, finalize, navigate, confirm, cancel, search, delete-by-id. The payload carries the data the agent needs to act on the event.
|
|
@@ -146,10 +149,18 @@ CONCRETE PATTERNS
|
|
|
146
149
|
contextSpec: { query: {schema: {type: "string"}, default: ""} }
|
|
147
150
|
Whether to ALSO declare a submit-search action depends on whether the agent acts on every keystroke (no action — the mirror IS the wire) or only on enter/click (declare a search action). Default to no action unless the intent names "search button" / "submit on enter". When declaring, pass the query as payload: schema: {type: "object", properties: {query: {type: "string"}}, required: ["query"]}.
|
|
148
151
|
|
|
149
|
-
Todo list —
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
152
|
+
Todo list / collection — split on OWNERSHIP, not on mutability. Both kinds mutate; what differs is WHO supplies the items.
|
|
153
|
+
|
|
154
|
+
(a) Agent-owned — "show my todos", "an agent-backed todo list that persists across sessions", "render my cart", "the messages in this thread", "the directory contents"
|
|
155
|
+
The AGENT owns the items: it fetched / persists / keeps them in sync. The collection is the agent's data → it goes on PROPSSPEC, seeded at render and refreshed via ggui_update after each change. This is the ONLY shape that round-trips — contextSpec has no agent-push channel, so an agent-owned list placed there can never be seeded or updated (the UI renders empty). add / delete / toggle are discrete events the agent must witness to persist → declare them on actionSpec (with a matching agentCapabilities tool for each nextStep). Mutability is fine: ggui_update is exactly how the agent pushes the change.
|
|
156
|
+
propsSpec: { properties: { todos: {schema: {type: "array", items: {type: "object", properties: {id: {type: "string"}, text: {type: "string"}, done: {type: "boolean"}}, required: ["id", "text", "done"]}}, required: true} } }
|
|
157
|
+
actionSpec: { toggleTodo: {label: "Toggle todo", schema: {type: "object", properties: {id: {type: "string"}}, required: ["id"]}, nextStep: "todo_toggle"}, addTodo: {label: "Add todo", schema: {type: "object", properties: {text: {type: "string"}}, required: ["text"]}, nextStep: "todo_add"} }
|
|
158
|
+
|
|
159
|
+
(b) User-built local — "a todo list where I can add and remove items", "a shopping list", "a checklist I tick off"
|
|
160
|
+
No agent-owned source: the USER assembles the list in the UI and the agent merely observes it. The items are client-originated state → a CONTEXTSPEC slot; OMIT actionSpec (the slot mirror IS the wire — the agent already sees every change). Only when the intent says the agent must persist / sync each change does it become the agent-owned case (a) above.
|
|
161
|
+
contextSpec: { todos: {schema: {type: "array"}, default: []} }
|
|
162
|
+
|
|
163
|
+
Tell them apart by the SOURCE of the initial items: "my / the / show / render / persisted / agent-backed / synced" → the agent has data to seed → (a) propsSpec. "a / let me build / I add" with no agent source → the user builds it → (b) contextSpec.
|
|
153
164
|
|
|
154
165
|
Confirmation modal — "a delete-confirmation modal with confirm and cancel actions"
|
|
155
166
|
Confirm and cancel are the discrete events the agent must witness; the modal holds no client state and no live feed. Declare propsSpec ONLY when the intent NAMES the item / data the modal shows ("confirm deleting <the file name>"); a generic confirmation modal that names no data field has NO propsSpec.
|
|
@@ -350,10 +361,10 @@ export const SYNTHESIZE_TOOL = {
|
|
|
350
361
|
usage: { type: 'string' },
|
|
351
362
|
},
|
|
352
363
|
},
|
|
353
|
-
description: 'Per-tool map: name → {inputSchema?, outputSchema?, usage?}. Catalog only — the agent owns invocation. Declare
|
|
364
|
+
description: 'Per-tool map: name → {inputSchema?, outputSchema?, usage?}. Catalog only — the agent owns invocation. Declare an entry for EVERY tool a nextStep / source points at: each actionSpec[X].nextStep AND each streamSpec[X].source.tool — referencing an undeclared tool fails the cross-reference check.',
|
|
354
365
|
},
|
|
355
366
|
},
|
|
356
|
-
description: 'Catalog of agent-invoked tools the contract references. The component code does NOT call these directly. Required
|
|
367
|
+
description: 'Catalog of agent-invoked tools the contract references. The component code does NOT call these directly. Required whenever any actionSpec.nextStep or streamSpec.source.tool names a tool.',
|
|
357
368
|
},
|
|
358
369
|
clientCapabilities: {
|
|
359
370
|
type: 'object',
|
|
@@ -398,7 +409,7 @@ export const SYNTHESIZE_TOOL = {
|
|
|
398
409
|
description: 'Per-prop map: name → {schema, required?}. Declares the initial render data the agent passes at push time.',
|
|
399
410
|
},
|
|
400
411
|
},
|
|
401
|
-
description: '
|
|
412
|
+
description: 'Agent-OWNED data the UI displays — the initial values seeded at render, refreshed any time after via ggui_update. NOT static-only: mutable collections the agent owns / fetched / keeps in sync (my todos, the cart, this thread\'s messages, a directory listing) go here too — propsSpec is the ONLY agent→client data channel. Use whenever the agent is the SOURCE of the displayed data (weather card → city/temp; profile → name/avatar; "my todos" → todos). Omit only when the UI originates its own state with no agent-supplied contents (a counter, a blank notepad, a list the user builds locally).',
|
|
402
413
|
},
|
|
403
414
|
reason: {
|
|
404
415
|
type: 'string',
|
|
@@ -435,6 +446,167 @@ function buildRepairNote(rejected, failure) {
|
|
|
435
446
|
'Emit a corrected contract that fixes exactly this problem. Keep every other spec unchanged.',
|
|
436
447
|
].join('\n');
|
|
437
448
|
}
|
|
449
|
+
/**
|
|
450
|
+
* PATCH-MODE preamble (L2) — prepended to the system prompt ONLY on the
|
|
451
|
+
* repair path (a draft is present). The cold-path system prompt is an
|
|
452
|
+
* "infer a contract from intent" authoring brief; reusing it verbatim
|
|
453
|
+
* for repair invites the model to RE-AUTHOR and reshape a near-correct
|
|
454
|
+
* draft (e.g. move a propsSpec collection to contextSpec). This preamble
|
|
455
|
+
* reframes the task as a minimal patch, which — together with the
|
|
456
|
+
* deterministic preservation gate — keeps the repair faithful.
|
|
457
|
+
*/
|
|
458
|
+
const REPAIR_PREAMBLE = `PATCH MODE — you are REPAIRING a contract the agent already authored, NOT writing a new one from scratch.
|
|
459
|
+
- Change as LITTLE as possible. Fix ONLY the specific findings listed in the user message.
|
|
460
|
+
- PRESERVE every spec the agent declared — ESPECIALLY every propsSpec property (agent-owned render-time seed data the UI needs). Never drop it.
|
|
461
|
+
- Do NOT move data between specs (e.g. propsSpec → contextSpec) unless a finding explicitly requires it. A collection the agent supplies on propsSpec STAYS on propsSpec.
|
|
462
|
+
- Keep the agent's names, shapes, and structure intact wherever the findings do not force a change.
|
|
463
|
+
The four-spec model and placement rules below still hold — but in PATCH MODE they are guardrails for the fix, not a license to re-author.`;
|
|
464
|
+
/**
|
|
465
|
+
* Preservation-failure repair note (L1) — drives a corrective retry when
|
|
466
|
+
* a VALID candidate dropped an agent-owned propsSpec seed surface. Names
|
|
467
|
+
* the exact missing keys so the model restores them, rather than the
|
|
468
|
+
* generic "fix this validation error" note (the candidate IS valid; the
|
|
469
|
+
* problem is round-trip fidelity the gate can't express).
|
|
470
|
+
*/
|
|
471
|
+
function buildPreservationRepairNote(rejected, dropped) {
|
|
472
|
+
const json = JSON.stringify(rejected);
|
|
473
|
+
const capped = json.length > 3000 ? `${json.slice(0, 3000)}…` : json;
|
|
474
|
+
return [
|
|
475
|
+
'YOUR PREVIOUS ATTEMPT dropped agent-owned render-time data. You returned this contract:',
|
|
476
|
+
capped,
|
|
477
|
+
'',
|
|
478
|
+
`The agent's draft declared these on propsSpec (agent-owned seed data the UI renders): ${dropped.join(', ')}. Your contract no longer carries them as propsSpec properties — so the agent can no longer seed them at render. contextSpec has NO agent seed channel, so moving them there leaves the UI empty.`,
|
|
479
|
+
'',
|
|
480
|
+
`Re-emit the contract with ${dropped.join(', ')} restored as propsSpec properties (agent-owned, seeded at render and refreshed via ggui_update). Keep every other spec unchanged.`,
|
|
481
|
+
].join('\n');
|
|
482
|
+
}
|
|
483
|
+
/**
|
|
484
|
+
* Compose the FIRST-attempt repair note for the forgiving-handshake
|
|
485
|
+
* path: the agent PROPOSED a contract that failed deterministic
|
|
486
|
+
* validation. Unlike {@link buildRepairNote} (which frames the input as
|
|
487
|
+
* "your previous attempt"), this frames it as the AGENT'S draft to be
|
|
488
|
+
* repaired in place — preserve intent + structure, fix exactly the
|
|
489
|
+
* listed findings. The draft JSON is capped to bound the prompt.
|
|
490
|
+
*/
|
|
491
|
+
function buildDraftSeedNote(draft, findings) {
|
|
492
|
+
const json = JSON.stringify(draft);
|
|
493
|
+
const capped = json.length > 3000 ? `${json.slice(0, 3000)}…` : json;
|
|
494
|
+
const findingsText = findings.length > 0
|
|
495
|
+
? findings.map((f) => `[${f.code}] ${f.path}: ${f.message}`).join('; ')
|
|
496
|
+
: '(unspecified — re-derive a valid contract for the intent)';
|
|
497
|
+
return [
|
|
498
|
+
'The agent PROPOSED this contract, but it failed deterministic validation:',
|
|
499
|
+
capped,
|
|
500
|
+
'',
|
|
501
|
+
`Validation findings — ${findingsText}`,
|
|
502
|
+
'',
|
|
503
|
+
"Repair it: keep the agent's intent and as much of their structure (spec names, schemas, labels) as possible, fix EXACTLY these findings, and return a valid contract. Do not add unrelated specs.",
|
|
504
|
+
].join('\n');
|
|
505
|
+
}
|
|
506
|
+
/**
|
|
507
|
+
* Invoke the synthesize tool — structured output when the provider
|
|
508
|
+
* supports it, else a text-JSON fallback. Forced tool use is
|
|
509
|
+
* Anthropic / Bedrock-only today; gemini / openai / openrouter callers
|
|
510
|
+
* reach this via the text path (the model emits one JSON object, which
|
|
511
|
+
* we regex-extract + parse). Lower reliability than forced tool use,
|
|
512
|
+
* but the bounded validate-and-repair loop catches malformed output and
|
|
513
|
+
* retries — so repair works on every provider instead of no-op'ing
|
|
514
|
+
* off-Anthropic.
|
|
515
|
+
*/
|
|
516
|
+
async function callSynthesizeTool(llm, system, user) {
|
|
517
|
+
if (typeof llm.callStructured === 'function') {
|
|
518
|
+
return llm.callStructured(system, user, SYNTHESIZE_TOOL, 1024);
|
|
519
|
+
}
|
|
520
|
+
const text = await llm.call(system, `${user}\n\nRespond with ONE JSON object only (no prose, no code fence) carrying the synthesized contract: {actionSpec?, contextSpec?, streamSpec?, propsSpec?, agentCapabilities?, clientCapabilities?, reason}. Every spec entry wraps its JSON Schema under a "schema" field.`, 1024);
|
|
521
|
+
return extractJsonObject(text);
|
|
522
|
+
}
|
|
523
|
+
/**
|
|
524
|
+
* Thrown by {@link extractJsonObject} when the text-fallback response
|
|
525
|
+
* carries no parseable JSON object — distinct from a transient network
|
|
526
|
+
* throw so the repair loop knows to feed a corrective "emit pure JSON"
|
|
527
|
+
* note (a network retry re-sends the same prompt instead).
|
|
528
|
+
*/
|
|
529
|
+
class SynthesizeTextParseError extends Error {
|
|
530
|
+
constructor(message) {
|
|
531
|
+
super(message);
|
|
532
|
+
this.name = 'SynthesizeTextParseError';
|
|
533
|
+
}
|
|
534
|
+
}
|
|
535
|
+
/**
|
|
536
|
+
* Robustly extract one JSON object from a possibly-prose-wrapped model
|
|
537
|
+
* response. A naive greedy `/\{[\s\S]*\}/` over-captures (first `{` to
|
|
538
|
+
* LAST `}`), throwing on "prose with a brace before the JSON" or "two
|
|
539
|
+
* objects". This tries, in order: (1) parse the trimmed text directly;
|
|
540
|
+
* (2) parse the contents of a ```json fence; (3) scan from the first `{`
|
|
541
|
+
* for the BALANCED closing `}` (string/escape aware). Throws only when
|
|
542
|
+
* nothing parses — the bounded validate-and-repair loop then retries
|
|
543
|
+
* with a corrective note.
|
|
544
|
+
*/
|
|
545
|
+
function extractJsonObject(text) {
|
|
546
|
+
const trimmed = text.trim();
|
|
547
|
+
try {
|
|
548
|
+
return JSON.parse(trimmed);
|
|
549
|
+
}
|
|
550
|
+
catch {
|
|
551
|
+
// fall through
|
|
552
|
+
}
|
|
553
|
+
const fence = trimmed.match(/```(?:json)?\s*([\s\S]*?)```/);
|
|
554
|
+
if (fence && fence[1]) {
|
|
555
|
+
try {
|
|
556
|
+
return JSON.parse(fence[1].trim());
|
|
557
|
+
}
|
|
558
|
+
catch {
|
|
559
|
+
// fall through
|
|
560
|
+
}
|
|
561
|
+
}
|
|
562
|
+
// Scan each balanced {...} from each '{' and return the FIRST that
|
|
563
|
+
// parses — robust against prose braces BEFORE the JSON (e.g. "{your
|
|
564
|
+
// widget}: {...}") and a trailing second object.
|
|
565
|
+
let searchFrom = 0;
|
|
566
|
+
for (;;) {
|
|
567
|
+
const start = trimmed.indexOf('{', searchFrom);
|
|
568
|
+
if (start < 0)
|
|
569
|
+
break;
|
|
570
|
+
let depth = 0;
|
|
571
|
+
let inStr = false;
|
|
572
|
+
let esc = false;
|
|
573
|
+
let end = -1;
|
|
574
|
+
for (let i = start; i < trimmed.length; i++) {
|
|
575
|
+
const ch = trimmed[i];
|
|
576
|
+
if (inStr) {
|
|
577
|
+
if (esc)
|
|
578
|
+
esc = false;
|
|
579
|
+
else if (ch === '\\')
|
|
580
|
+
esc = true;
|
|
581
|
+
else if (ch === '"')
|
|
582
|
+
inStr = false;
|
|
583
|
+
}
|
|
584
|
+
else if (ch === '"') {
|
|
585
|
+
inStr = true;
|
|
586
|
+
}
|
|
587
|
+
else if (ch === '{') {
|
|
588
|
+
depth++;
|
|
589
|
+
}
|
|
590
|
+
else if (ch === '}') {
|
|
591
|
+
depth--;
|
|
592
|
+
if (depth === 0) {
|
|
593
|
+
end = i;
|
|
594
|
+
break;
|
|
595
|
+
}
|
|
596
|
+
}
|
|
597
|
+
}
|
|
598
|
+
if (end < 0)
|
|
599
|
+
break; // unbalanced from here on
|
|
600
|
+
try {
|
|
601
|
+
return JSON.parse(trimmed.slice(start, end + 1));
|
|
602
|
+
}
|
|
603
|
+
catch {
|
|
604
|
+
// not valid JSON from this '{' — advance and try the next one
|
|
605
|
+
}
|
|
606
|
+
searchFrom = start + 1;
|
|
607
|
+
}
|
|
608
|
+
throw new SynthesizeTextParseError(`synthesize text-fallback: no parseable JSON object in response (length ${text.length})`);
|
|
609
|
+
}
|
|
438
610
|
/**
|
|
439
611
|
* Drop `actionSpec` entries the structural validator flags as
|
|
440
612
|
* `redundant-action` — an empty-payload action whose name is a
|
|
@@ -477,8 +649,14 @@ function pruneRedundantActions(contract) {
|
|
|
477
649
|
* Empty / whitespace intent short-circuits to null with a reason —
|
|
478
650
|
* no contract can be inferred from nothing.
|
|
479
651
|
*
|
|
480
|
-
*
|
|
481
|
-
*
|
|
652
|
+
* Providers without `callStructured` (gemini / openai / openrouter) use
|
|
653
|
+
* a text-JSON fallback (the validate-and-repair loop catches malformed
|
|
654
|
+
* output and retries) rather than skipping synthesis.
|
|
655
|
+
*
|
|
656
|
+
* When `options.draft` is supplied, the loop REPAIRS that draft in
|
|
657
|
+
* place (seeded with the agent's contract + the deterministic findings)
|
|
658
|
+
* instead of synthesizing from `intent` alone — the forgiving-handshake
|
|
659
|
+
* path.
|
|
482
660
|
*
|
|
483
661
|
* Each attempt is self-checked against the validation gate; a failure
|
|
484
662
|
* feeds the precise error back for up to {@link MAX_SYNTH_ATTEMPTS}
|
|
@@ -498,15 +676,6 @@ export async function synthesizeContract(deps, intent, options) {
|
|
|
498
676
|
findings: [],
|
|
499
677
|
};
|
|
500
678
|
}
|
|
501
|
-
if (typeof deps.llm.callStructured !== 'function') {
|
|
502
|
-
return {
|
|
503
|
-
contract: null,
|
|
504
|
-
reason: 'synthesize-skip: provider does not support callStructured. Bind a structured-capable LLMCaller (Anthropic adapter) to enable contract synthesis.',
|
|
505
|
-
latencyMs: Date.now() - startedAt,
|
|
506
|
-
attempts: 0,
|
|
507
|
-
findings: [],
|
|
508
|
-
};
|
|
509
|
-
}
|
|
510
679
|
const gadgetsSection = composeAvailableGadgetsSection(options?.appGadgets);
|
|
511
680
|
const baseUserPrompt = `INTENT: ${trimmed}${gadgetsSection ? `\n\n${gadgetsSection}` : ''}`;
|
|
512
681
|
// Bounded validate-and-repair loop. Each attempt is self-checked
|
|
@@ -519,21 +688,43 @@ export async function synthesizeContract(deps, intent, options) {
|
|
|
519
688
|
// small enough that a full re-emit IS the surgical edit. Budget
|
|
520
689
|
// exhausted → decline exactly as the one-shot path did (caller falls
|
|
521
690
|
// back to an empty contract stub).
|
|
522
|
-
|
|
691
|
+
// Repair-in-place seed: when the caller passed the agent's rejected
|
|
692
|
+
// draft, the loop's FIRST attempt repairs that draft (not a blank
|
|
693
|
+
// synthesis). Later attempts overwrite this via buildRepairNote.
|
|
694
|
+
let repairNote = options?.draft !== undefined
|
|
695
|
+
? buildDraftSeedNote(options.draft, options.draftFindings ?? [])
|
|
696
|
+
: undefined;
|
|
523
697
|
let lastReason = 'synthesize-fail: exhausted repair attempts';
|
|
524
698
|
let lastFindings = [];
|
|
699
|
+
// L2 — repair path uses the PATCH-MODE preamble (minimal patch, no
|
|
700
|
+
// re-author); cold path uses the bare authoring prompt.
|
|
701
|
+
const systemPrompt = options?.draft !== undefined
|
|
702
|
+
? `${REPAIR_PREAMBLE}\n\n${SYNTHESIZE_SYSTEM_PROMPT}`
|
|
703
|
+
: SYNTHESIZE_SYSTEM_PROMPT;
|
|
704
|
+
// L1 — the agent-owned seed surfaces the repaired contract MUST keep,
|
|
705
|
+
// and the best valid-but-non-preserving candidate to fall back on so
|
|
706
|
+
// preservation never makes the result WORSE than the validity-only gate.
|
|
707
|
+
const draftSeedKeys = options?.draft !== undefined ? draftSeedPropKeys(options.draft) : [];
|
|
708
|
+
let lastValidContract = null;
|
|
525
709
|
for (let attempt = 1; attempt <= MAX_SYNTH_ATTEMPTS; attempt++) {
|
|
526
710
|
const userPrompt = repairNote === undefined
|
|
527
711
|
? baseUserPrompt
|
|
528
712
|
: `${baseUserPrompt}\n\n${repairNote}`;
|
|
529
713
|
let toolInput;
|
|
530
714
|
try {
|
|
531
|
-
toolInput = await deps.llm
|
|
715
|
+
toolInput = await callSynthesizeTool(deps.llm, systemPrompt, userPrompt);
|
|
532
716
|
}
|
|
533
717
|
catch (err) {
|
|
534
|
-
|
|
535
|
-
//
|
|
536
|
-
|
|
718
|
+
lastReason = `synthesize-fail: callSynthesizeTool threw — ${err instanceof Error ? err.message : String(err)}`;
|
|
719
|
+
// Distinguish causes: an UNPARSEABLE text-path response (model
|
|
720
|
+
// emitted prose/non-JSON — common on gemini/openai via the text
|
|
721
|
+
// fallback) gets a corrective note so the next attempt emits pure
|
|
722
|
+
// JSON. A transient NETWORK failure re-sends the SAME prompt
|
|
723
|
+
// (nothing to correct) — preserving the retry semantics.
|
|
724
|
+
if (err instanceof SynthesizeTextParseError) {
|
|
725
|
+
repairNote =
|
|
726
|
+
'Your previous response could not be parsed as JSON. Respond with EXACTLY ONE JSON object and nothing else — no prose, no explanation, no markdown code fence.';
|
|
727
|
+
}
|
|
537
728
|
continue;
|
|
538
729
|
}
|
|
539
730
|
const parsed = parseToolInput(toolInput);
|
|
@@ -595,6 +786,23 @@ export async function synthesizeContract(deps, intent, options) {
|
|
|
595
786
|
repairNote = buildRepairNote(validatedContract, `it failed contract validation: ${errText}`);
|
|
596
787
|
continue;
|
|
597
788
|
}
|
|
789
|
+
// L1 — preservation gate (best-effort). The candidate is VALID, but
|
|
790
|
+
// a repair must not silently drop an agent-owned seed surface the
|
|
791
|
+
// draft declared on propsSpec (the valid-but-round-trip-broken
|
|
792
|
+
// reshape lintContract + the placement validators can't see).
|
|
793
|
+
// Deterministic + model-independent: it drives a corrective retry.
|
|
794
|
+
if (draftSeedKeys.length > 0) {
|
|
795
|
+
const dropped = findDroppedSeedSurfaces(options?.draft, validatedContract);
|
|
796
|
+
if (dropped.length > 0) {
|
|
797
|
+
// Remember the best VALID candidate so an exhausted budget never
|
|
798
|
+
// returns WORSE than the validity-only gate did (a valid contract).
|
|
799
|
+
lastValidContract = validatedContract;
|
|
800
|
+
lastReason = `synthesize-preservation: candidate dropped agent-owned propsSpec seed surface(s) [${dropped.join(', ')}]`;
|
|
801
|
+
lastFindings = allFindings;
|
|
802
|
+
repairNote = buildPreservationRepairNote(validatedContract, dropped);
|
|
803
|
+
continue;
|
|
804
|
+
}
|
|
805
|
+
}
|
|
598
806
|
const findingsSuffix = allFindings.length > 0
|
|
599
807
|
? ` — validator: ${formatValidationFindings({ findings: allFindings })}`
|
|
600
808
|
: '';
|
|
@@ -607,9 +815,15 @@ export async function synthesizeContract(deps, intent, options) {
|
|
|
607
815
|
findings: allFindings,
|
|
608
816
|
};
|
|
609
817
|
}
|
|
818
|
+
// Budget exhausted. If preservation retries never landed a contract
|
|
819
|
+
// that kept every seed surface, fall back to the best VALID candidate
|
|
820
|
+
// we did produce (never worse than the validity-only gate). Only when
|
|
821
|
+
// no valid candidate ever appeared do we decline with `null`.
|
|
610
822
|
return {
|
|
611
|
-
contract:
|
|
612
|
-
reason:
|
|
823
|
+
contract: lastValidContract,
|
|
824
|
+
reason: lastValidContract !== null
|
|
825
|
+
? `${lastReason}; returned best valid candidate (preservation retries exhausted)`
|
|
826
|
+
: lastReason,
|
|
613
827
|
latencyMs: Date.now() - startedAt,
|
|
614
828
|
attempts: MAX_SYNTH_ATTEMPTS,
|
|
615
829
|
findings: lastFindings,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ggui-ai/negotiator",
|
|
3
|
-
"version": "0.2.0-alpha.
|
|
3
|
+
"version": "0.2.0-alpha.4",
|
|
4
4
|
"description": "UI decision engine for ggui. Given an agent's signal and the current render state, decides which UI to render (create / update / replace) and synthesizes the data contract. Deployment-agnostic: concrete embedding and vector-store bindings plug in via the storage interfaces from @ggui-ai/mcp-server-core.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"keywords": [
|
|
@@ -47,8 +47,8 @@
|
|
|
47
47
|
}
|
|
48
48
|
},
|
|
49
49
|
"dependencies": {
|
|
50
|
-
"@ggui-ai/mcp-server-core": "0.2.0-alpha.
|
|
51
|
-
"@ggui-ai/protocol": "0.2.0-alpha.
|
|
50
|
+
"@ggui-ai/mcp-server-core": "0.2.0-alpha.4",
|
|
51
|
+
"@ggui-ai/protocol": "0.2.0-alpha.4"
|
|
52
52
|
},
|
|
53
53
|
"devDependencies": {
|
|
54
54
|
"@types/node": "^24.0.0",
|
|
@@ -69,6 +69,7 @@
|
|
|
69
69
|
"test": "vitest run",
|
|
70
70
|
"test:watch": "vitest",
|
|
71
71
|
"probe-rerank": "tsx src/rerank-eval/run-probe-cli.ts",
|
|
72
|
-
"bench-synth": "tsx src/synth-bench/run-bench-cli.ts"
|
|
72
|
+
"bench-synth": "tsx src/synth-bench/run-bench-cli.ts",
|
|
73
|
+
"bench-repair": "tsx src/synth-bench/run-repair-bench-cli.ts"
|
|
73
74
|
}
|
|
74
75
|
}
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `ensureConformingContract` — the negotiator's create-path guarantee.
|
|
3
|
+
*
|
|
4
|
+
* Given the agent's PROPOSED draft, return a contract that is
|
|
5
|
+
* GUARANTEED to pass the deterministic gate (`lintContract` with zero
|
|
6
|
+
* errors), so the handshake backstop (`validateContract`) never throws
|
|
7
|
+
* on it. This is the "Propose vs Commit" forgiving-handshake core
|
|
8
|
+
* shared by every negotiator implementation (OSS llm-backed + cloud
|
|
9
|
+
* bedrock) so the behavior cannot drift between deployments:
|
|
10
|
+
*
|
|
11
|
+
* - draft already conforms → return it verbatim (origin: 'agent')
|
|
12
|
+
* - draft has errors → repair-in-place via the bounded LLM loop,
|
|
13
|
+
* seeded with the draft + the deterministic
|
|
14
|
+
* findings, looping until the gate is green
|
|
15
|
+
* (origin: 'synth')
|
|
16
|
+
* - repair impossible → minimal conforming contract (`{}`) + loud
|
|
17
|
+
* (LLM down / provider error findings; STILL origin 'synth';
|
|
18
|
+
* can't synth / budget NEVER throws.
|
|
19
|
+
* exhausted)
|
|
20
|
+
*
|
|
21
|
+
* Determinism lives in the GATE (`lintContract`), never in the repair.
|
|
22
|
+
* The repair LLM is non-deterministic, but the loop only exits when the
|
|
23
|
+
* deterministic gate is green — the same shape ui-gen uses to tolerate
|
|
24
|
+
* non-deterministic code generation behind a deterministic self_check.
|
|
25
|
+
*
|
|
26
|
+
* Cache/blueprint matching is NOT this function's job — the caller
|
|
27
|
+
* (negotiator `decide()`) runs its deployment-specific cache match
|
|
28
|
+
* FIRST and only falls through to here on a miss. That preserves the
|
|
29
|
+
* "cache-first, repair-second" ordering the negotiator contract
|
|
30
|
+
* mandates.
|
|
31
|
+
*/
|
|
32
|
+
import {
|
|
33
|
+
lintContract,
|
|
34
|
+
dataContractSchema,
|
|
35
|
+
type DataContract,
|
|
36
|
+
type GadgetDescriptor,
|
|
37
|
+
type SuggestionFinding,
|
|
38
|
+
} from '@ggui-ai/protocol';
|
|
39
|
+
import type { LLMCaller } from './llm-caller.js';
|
|
40
|
+
import { synthesizeContract } from './synthesize-contract.js';
|
|
41
|
+
import { normalizeDraft } from './normalize-draft.js';
|
|
42
|
+
|
|
43
|
+
export interface EnsureConformingResult {
|
|
44
|
+
/** A contract guaranteed to pass `lintContract` with zero errors. */
|
|
45
|
+
readonly contract: DataContract;
|
|
46
|
+
/**
|
|
47
|
+
* - `'agent'` — the draft was already conforming; returned verbatim.
|
|
48
|
+
* - `'synth'` — the draft had errors; this is the repaired result
|
|
49
|
+
* (or the minimal-conforming fallback when repair was impossible).
|
|
50
|
+
*/
|
|
51
|
+
readonly origin: 'agent' | 'synth';
|
|
52
|
+
/**
|
|
53
|
+
* How the conforming contract was produced — finer-grained than
|
|
54
|
+
* `origin`, for telemetry (the efficiency tiers):
|
|
55
|
+
* - `verbatim` — draft was clean; returned as-is (origin agent).
|
|
56
|
+
* - `normalized` — deterministic fix only, NO LLM (origin synth).
|
|
57
|
+
* - `llm-repair` — the bounded LLM repair loop ran (origin synth).
|
|
58
|
+
* - `fallback-empty`— unrepairable; minimal `{}` contract (origin synth).
|
|
59
|
+
*/
|
|
60
|
+
readonly method: 'verbatim' | 'normalized' | 'llm-repair' | 'fallback-empty';
|
|
61
|
+
/**
|
|
62
|
+
* Findings surfaced to the agent. On `origin: 'agent'`, any hygiene
|
|
63
|
+
* warnings on the (valid) draft. On `origin: 'synth'`, the ERROR
|
|
64
|
+
* findings that rejected the agent's draft — so the agent-side model
|
|
65
|
+
* learns what it got wrong, even though we repaired it.
|
|
66
|
+
*/
|
|
67
|
+
readonly findings: readonly SuggestionFinding[];
|
|
68
|
+
/** Operator- + LLM-readable explanation. */
|
|
69
|
+
readonly reasoning: string;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** Trivially-valid last-resort contract — all four specs omitted. */
|
|
73
|
+
const EMPTY_CONTRACT: DataContract = {};
|
|
74
|
+
|
|
75
|
+
export async function ensureConformingContract(
|
|
76
|
+
deps: { readonly llm: LLMCaller },
|
|
77
|
+
args: {
|
|
78
|
+
/** Untrusted: the agent's draft may not be a valid DataContract. */
|
|
79
|
+
readonly draft: unknown;
|
|
80
|
+
readonly intent: string;
|
|
81
|
+
readonly appGadgets?: readonly GadgetDescriptor[];
|
|
82
|
+
},
|
|
83
|
+
): Promise<EnsureConformingResult> {
|
|
84
|
+
const lint = lintContract(args.draft);
|
|
85
|
+
const warnFindings: SuggestionFinding[] = lint.warnings.map(
|
|
86
|
+
(w): SuggestionFinding => ({
|
|
87
|
+
code: w.code,
|
|
88
|
+
severity: 'warn',
|
|
89
|
+
path: w.path,
|
|
90
|
+
message: w.message,
|
|
91
|
+
}),
|
|
92
|
+
);
|
|
93
|
+
|
|
94
|
+
// Fast path — draft already conforms. Deterministic, no LLM call.
|
|
95
|
+
// `lint.errors.length === 0` implies the shape phase passed, so the
|
|
96
|
+
// strict parse cannot throw — it just re-derives the typed DataContract
|
|
97
|
+
// from the untrusted input (validator returns the typed shape; no cast).
|
|
98
|
+
if (lint.errors.length === 0) {
|
|
99
|
+
return {
|
|
100
|
+
contract: dataContractSchema.parse(args.draft),
|
|
101
|
+
origin: 'agent',
|
|
102
|
+
method: 'verbatim',
|
|
103
|
+
findings: warnFindings,
|
|
104
|
+
reasoning:
|
|
105
|
+
'agent draft passed validateContract; accepted verbatim (origin: agent)',
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
const errorFindings: SuggestionFinding[] = lint.errors.map(
|
|
110
|
+
(e): SuggestionFinding => ({
|
|
111
|
+
code: e.code,
|
|
112
|
+
severity: 'error',
|
|
113
|
+
path: e.path,
|
|
114
|
+
message: e.message,
|
|
115
|
+
}),
|
|
116
|
+
);
|
|
117
|
+
|
|
118
|
+
// L3 — deterministic normalization tier. Most agent malformations are
|
|
119
|
+
// mechanical (stray illegal wrapper keys, non-canonical schema types).
|
|
120
|
+
// Fix them WITHOUT an LLM: strip + canonicalize, re-lint, and if the
|
|
121
|
+
// draft now conforms, return it verbatim-but-cleaned. Faithful (no
|
|
122
|
+
// reshape risk — the agent's specs are preserved exactly) and free (no
|
|
123
|
+
// LLM call). Semantic deficiencies fall through to the repair loop.
|
|
124
|
+
const normalized = normalizeDraft(args.draft);
|
|
125
|
+
const normLint = lintContract(normalized);
|
|
126
|
+
if (normLint.errors.length === 0) {
|
|
127
|
+
return {
|
|
128
|
+
contract: dataContractSchema.parse(normalized),
|
|
129
|
+
origin: 'synth',
|
|
130
|
+
method: 'normalized',
|
|
131
|
+
findings: errorFindings,
|
|
132
|
+
reasoning:
|
|
133
|
+
'normalized the agent draft deterministically (stripped invalid keys / canonicalized schema types, no LLM) to pass validateContract',
|
|
134
|
+
};
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
// Repair loop on the NORMALIZED draft (mechanical errors already
|
|
138
|
+
// fixed) with only the REMAINING (semantic) findings — so the LLM
|
|
139
|
+
// patches what reasoning is genuinely needed for, from a clean start.
|
|
140
|
+
const synth = await synthesizeContract(deps, args.intent, {
|
|
141
|
+
...(args.appGadgets ? { appGadgets: args.appGadgets } : {}),
|
|
142
|
+
draft: normalized,
|
|
143
|
+
draftFindings: normLint.errors.map((e) => ({
|
|
144
|
+
code: e.code,
|
|
145
|
+
path: e.path,
|
|
146
|
+
message: e.message,
|
|
147
|
+
})),
|
|
148
|
+
});
|
|
149
|
+
|
|
150
|
+
if (
|
|
151
|
+
synth.contract !== null &&
|
|
152
|
+
lintContract(synth.contract).errors.length === 0
|
|
153
|
+
) {
|
|
154
|
+
return {
|
|
155
|
+
contract: synth.contract,
|
|
156
|
+
origin: 'synth',
|
|
157
|
+
method: 'llm-repair',
|
|
158
|
+
findings: errorFindings,
|
|
159
|
+
reasoning: `repaired the agent draft to pass validateContract — ${synth.reason}`,
|
|
160
|
+
};
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
// Repair impossible (LLM down, provider can't synthesize, or the
|
|
164
|
+
// repair budget exhausted). We still MUST return a conforming
|
|
165
|
+
// contract — the handshake never hard-fails. Minimal conforming
|
|
166
|
+
// contract + loud findings so the agent can re-issue a corrected
|
|
167
|
+
// contract via ggui_render override if it needs the declared specs.
|
|
168
|
+
return {
|
|
169
|
+
contract: EMPTY_CONTRACT,
|
|
170
|
+
origin: 'synth',
|
|
171
|
+
method: 'fallback-empty',
|
|
172
|
+
findings: errorFindings,
|
|
173
|
+
reasoning: `could not repair the agent draft within budget (${synth.reason}); returning a minimal conforming contract — re-issue a corrected contract via ggui_render override if you need the declared specs`,
|
|
174
|
+
};
|
|
175
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -49,6 +49,8 @@ export type {
|
|
|
49
49
|
} from './llm-rerank.js';
|
|
50
50
|
export { synthesizeContract } from './synthesize-contract.js';
|
|
51
51
|
export type { SynthesizeContractResult } from './synthesize-contract.js';
|
|
52
|
+
export { ensureConformingContract } from './ensure-conforming-contract.js';
|
|
53
|
+
export type { EnsureConformingResult } from './ensure-conforming-contract.js';
|
|
52
54
|
export {
|
|
53
55
|
validateContractStructure,
|
|
54
56
|
validateContractNovelty,
|