@ggui-ai/negotiator 0.2.0-alpha.3 → 0.2.0-alpha.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ensure-conforming-contract.d.ts +70 -0
- package/dist/ensure-conforming-contract.d.ts.map +1 -0
- package/dist/ensure-conforming-contract.js +115 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/normalize-draft.d.ts +33 -0
- package/dist/normalize-draft.d.ts.map +1 -0
- package/dist/normalize-draft.js +143 -0
- package/dist/preserve-seed-surfaces.d.ts +40 -0
- package/dist/preserve-seed-surfaces.d.ts.map +1 -0
- package/dist/preserve-seed-surfaces.js +57 -0
- package/dist/synth-bench/cli-llm.d.ts +20 -0
- package/dist/synth-bench/cli-llm.d.ts.map +1 -0
- package/dist/synth-bench/cli-llm.js +97 -0
- package/dist/synth-bench/corpus.d.ts +52 -0
- package/dist/synth-bench/corpus.d.ts.map +1 -1
- package/dist/synth-bench/corpus.js +306 -5
- package/dist/synth-bench/round-trip-score.d.ts +87 -0
- package/dist/synth-bench/round-trip-score.d.ts.map +1 -0
- package/dist/synth-bench/round-trip-score.js +105 -0
- package/dist/synth-bench/run-bench-cli.js +6 -82
- package/dist/synth-bench/run-repair-bench-cli.d.ts +3 -0
- package/dist/synth-bench/run-repair-bench-cli.d.ts.map +1 -0
- package/dist/synth-bench/run-repair-bench-cli.js +86 -0
- package/dist/synth-bench/run-repair-bench.d.ts +94 -0
- package/dist/synth-bench/run-repair-bench.d.ts.map +1 -0
- package/dist/synth-bench/run-repair-bench.js +172 -0
- package/dist/synthesize-contract.d.ts +38 -6
- package/dist/synthesize-contract.d.ts.map +1 -1
- package/dist/synthesize-contract.js +246 -32
- package/package.json +5 -4
- package/src/ensure-conforming-contract.ts +175 -0
- package/src/index.ts +2 -0
- package/src/normalize-draft.ts +156 -0
- package/src/preserve-seed-surfaces.ts +61 -0
- package/src/synth-bench/cli-llm.ts +140 -0
- package/src/synth-bench/corpus.ts +335 -5
- package/src/synth-bench/round-trip-score.ts +169 -0
- package/src/synth-bench/run-bench-cli.ts +13 -115
- package/src/synth-bench/run-repair-bench-cli.ts +119 -0
- package/src/synth-bench/run-repair-bench.ts +266 -0
- package/src/synthesize-contract.ts +299 -37
|
@@ -24,10 +24,12 @@
|
|
|
24
24
|
* a richer surface should author the contract themselves on the
|
|
25
25
|
* handshake input; synthesis is a fallback, not a replacement.
|
|
26
26
|
*
|
|
27
|
-
* **Failure modes collapse to null.** LLM throws
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
27
|
+
* **Failure modes collapse to null.** LLM throws on every attempt or
|
|
28
|
+
* the bounded repair budget is exhausted → return `null`. Caller falls
|
|
29
|
+
* back to an empty stub; behavior regresses to pre-synth but doesn't
|
|
30
|
+
* crash. Providers without `callStructured` (gemini / openai /
|
|
31
|
+
* openrouter) use a text-JSON fallback rather than skipping synthesis,
|
|
32
|
+
* so repair works on every provider.
|
|
31
33
|
*
|
|
32
34
|
* **Cost.** ~$0.0005-0.001 per call (Haiku 4.5, ~500 input + ~300
|
|
33
35
|
* output tokens). Latency ~1.5s. Fires only on cold-path Tier 3
|
|
@@ -43,6 +45,10 @@ import {
|
|
|
43
45
|
import { lintContract, type ContractIssue } from '@ggui-ai/protocol';
|
|
44
46
|
import type { LLMCaller, ToolSchema } from './llm-caller.js';
|
|
45
47
|
import { normalizeSchema } from './normalize-schema.js';
|
|
48
|
+
import {
|
|
49
|
+
draftSeedPropKeys,
|
|
50
|
+
findDroppedSeedSurfaces,
|
|
51
|
+
} from './preserve-seed-surfaces.js';
|
|
46
52
|
import {
|
|
47
53
|
formatValidationFindings,
|
|
48
54
|
validateActionsVsContext,
|
|
@@ -114,14 +120,14 @@ A contract has FOUR specs that describe distinct directions on the wire between
|
|
|
114
120
|
|
|
115
121
|
THE FOUR-SPEC MODEL
|
|
116
122
|
|
|
117
|
-
propsSpec (agent → UI, render-time)
|
|
118
|
-
|
|
123
|
+
propsSpec (agent → UI, render-time + agent-pushed refreshes)
|
|
124
|
+
Data the AGENT owns and supplies — the initial values at mount, AND every later refresh via ggui_update. NOT "static / never changes": propsSpec is the ONLY channel for agent-owned data, mutable or not. A weather card's city+temp (fixed) AND the items of the todo list the agent fetched and keeps in sync (mutable) BOTH live here — the agent seeds them at render and pushes each change with ggui_update. Use whenever the agent is the SOURCE of what the UI shows: the intent names data the agent provides / fetches / owns ("my todos", "the cart", "this user's profile", "the directory contents") OR data the component cannot render without (city, temp). Omit only when the UI originates its own state with no agent-supplied contents (a counter starting at zero, a blank notepad, a list the USER builds locally).
|
|
119
125
|
|
|
120
126
|
streamSpec (agent → UI, live, append-only)
|
|
121
127
|
Channels where the agent pushes live data the UI displays as it arrives. Use ONLY when the intent describes ongoing agent-originated updates (a chat with messages, a live dashboard, a clock, a stock ticker, a notifications feed). Wrong instinct: do NOT use streamSpec for user-driven state, nor for a multi-step wizard / tutorial — its steps are a local stepper plus component-authored copy, not an agent-pushed feed.
|
|
122
128
|
|
|
123
129
|
contextSpec (UI → agent, live, debounced mirror)
|
|
124
|
-
Client state the agent OBSERVES continuously. The UI mutates each slot via a setter; the runtime mirrors the value back to the agent. Use for any
|
|
130
|
+
Client state the agent OBSERVES continuously. The UI mutates each slot via a setter; the runtime mirrors the value back to the agent. Use for any CLIENT-ORIGINATED state whose CURRENT VALUE is what the agent cares about — a counter's count, a form's draft fields, a slider's position, a selected tab, a search query as the user types. The mirror is the wire path: the agent already sees every change. Ownership boundary: contextSpec is client→agent ONLY — the agent can READ the mirror but CANNOT push values into it (there is no agent→contextSpec channel). Data the AGENT owns or seeds belongs on propsSpec, never here — a collection the agent fetched and must keep in sync placed on contextSpec can never be seeded or updated, and the UI renders empty.
|
|
125
131
|
|
|
126
132
|
actionSpec (UI → agent, one-shot event)
|
|
127
133
|
Discrete events the agent must WITNESS — a single point in time the agent receives a payload describing what happened. Use for events with semantic meaning beyond the current state of any slot: submit, save, send, finalize, navigate, confirm, cancel, search, delete-by-id. The payload carries the data the agent needs to act on the event.
|
|
@@ -187,10 +193,18 @@ CONCRETE PATTERNS
|
|
|
187
193
|
contextSpec: { query: {schema: {type: "string"}, default: ""} }
|
|
188
194
|
Whether to ALSO declare a submit-search action depends on whether the agent acts on every keystroke (no action — the mirror IS the wire) or only on enter/click (declare a search action). Default to no action unless the intent names "search button" / "submit on enter". When declaring, pass the query as payload: schema: {type: "object", properties: {query: {type: "string"}}, required: ["query"]}.
|
|
189
195
|
|
|
190
|
-
Todo list —
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
196
|
+
Todo list / collection — split on OWNERSHIP, not on mutability. Both kinds mutate; what differs is WHO supplies the items.
|
|
197
|
+
|
|
198
|
+
(a) Agent-owned — "show my todos", "an agent-backed todo list that persists across sessions", "render my cart", "the messages in this thread", "the directory contents"
|
|
199
|
+
The AGENT owns the items: it fetched / persists / keeps them in sync. The collection is the agent's data → it goes on PROPSSPEC, seeded at render and refreshed via ggui_update after each change. This is the ONLY shape that round-trips — contextSpec has no agent-push channel, so an agent-owned list placed there can never be seeded or updated (the UI renders empty). add / delete / toggle are discrete events the agent must witness to persist → declare them on actionSpec (with a matching agentCapabilities tool for each nextStep). Mutability is fine: ggui_update is exactly how the agent pushes the change.
|
|
200
|
+
propsSpec: { properties: { todos: {schema: {type: "array", items: {type: "object", properties: {id: {type: "string"}, text: {type: "string"}, done: {type: "boolean"}}, required: ["id", "text", "done"]}}, required: true} } }
|
|
201
|
+
actionSpec: { toggleTodo: {label: "Toggle todo", schema: {type: "object", properties: {id: {type: "string"}}, required: ["id"]}, nextStep: "todo_toggle"}, addTodo: {label: "Add todo", schema: {type: "object", properties: {text: {type: "string"}}, required: ["text"]}, nextStep: "todo_add"} }
|
|
202
|
+
|
|
203
|
+
(b) User-built local — "a todo list where I can add and remove items", "a shopping list", "a checklist I tick off"
|
|
204
|
+
No agent-owned source: the USER assembles the list in the UI and the agent merely observes it. The items are client-originated state → a CONTEXTSPEC slot; OMIT actionSpec (the slot mirror IS the wire — the agent already sees every change). Only when the intent says the agent must persist / sync each change does it become the agent-owned case (a) above.
|
|
205
|
+
contextSpec: { todos: {schema: {type: "array"}, default: []} }
|
|
206
|
+
|
|
207
|
+
Tell them apart by the SOURCE of the initial items: "my / the / show / render / persisted / agent-backed / synced" → the agent has data to seed → (a) propsSpec. "a / let me build / I add" with no agent source → the user builds it → (b) contextSpec.
|
|
194
208
|
|
|
195
209
|
Confirmation modal — "a delete-confirmation modal with confirm and cancel actions"
|
|
196
210
|
Confirm and cancel are the discrete events the agent must witness; the modal holds no client state and no live feed. Declare propsSpec ONLY when the intent NAMES the item / data the modal shows ("confirm deleting <the file name>"); a generic confirmation modal that names no data field has NO propsSpec.
|
|
@@ -400,11 +414,11 @@ export const SYNTHESIZE_TOOL: ToolSchema = {
|
|
|
400
414
|
},
|
|
401
415
|
},
|
|
402
416
|
description:
|
|
403
|
-
'Per-tool map: name → {inputSchema?, outputSchema?, usage?}. Catalog only — the agent owns invocation. Declare
|
|
417
|
+
'Per-tool map: name → {inputSchema?, outputSchema?, usage?}. Catalog only — the agent owns invocation. Declare an entry for EVERY tool a nextStep / source points at: each actionSpec[X].nextStep AND each streamSpec[X].source.tool — referencing an undeclared tool fails the cross-reference check.',
|
|
404
418
|
},
|
|
405
419
|
},
|
|
406
420
|
description:
|
|
407
|
-
'Catalog of agent-invoked tools the contract references. The component code does NOT call these directly. Required
|
|
421
|
+
'Catalog of agent-invoked tools the contract references. The component code does NOT call these directly. Required whenever any actionSpec.nextStep or streamSpec.source.tool names a tool.',
|
|
408
422
|
},
|
|
409
423
|
clientCapabilities: {
|
|
410
424
|
type: 'object',
|
|
@@ -456,7 +470,7 @@ export const SYNTHESIZE_TOOL: ToolSchema = {
|
|
|
456
470
|
},
|
|
457
471
|
},
|
|
458
472
|
description:
|
|
459
|
-
'
|
|
473
|
+
'Agent-OWNED data the UI displays — the initial values seeded at render, refreshed any time after via ggui_update. NOT static-only: mutable collections the agent owns / fetched / keeps in sync (my todos, the cart, this thread\'s messages, a directory listing) go here too — propsSpec is the ONLY agent→client data channel. Use whenever the agent is the SOURCE of the displayed data (weather card → city/temp; profile → name/avatar; "my todos" → todos). Omit only when the UI originates its own state with no agent-supplied contents (a counter, a blank notepad, a list the user builds locally).',
|
|
460
474
|
},
|
|
461
475
|
reason: {
|
|
462
476
|
type: 'string',
|
|
@@ -535,6 +549,184 @@ function buildRepairNote(rejected: unknown, failure: string): string {
|
|
|
535
549
|
].join('\n');
|
|
536
550
|
}
|
|
537
551
|
|
|
552
|
+
/**
|
|
553
|
+
* PATCH-MODE preamble (L2) — prepended to the system prompt ONLY on the
|
|
554
|
+
* repair path (a draft is present). The cold-path system prompt is an
|
|
555
|
+
* "infer a contract from intent" authoring brief; reusing it verbatim
|
|
556
|
+
* for repair invites the model to RE-AUTHOR and reshape a near-correct
|
|
557
|
+
* draft (e.g. move a propsSpec collection to contextSpec). This preamble
|
|
558
|
+
* reframes the task as a minimal patch, which — together with the
|
|
559
|
+
* deterministic preservation gate — keeps the repair faithful.
|
|
560
|
+
*/
|
|
561
|
+
const REPAIR_PREAMBLE = `PATCH MODE — you are REPAIRING a contract the agent already authored, NOT writing a new one from scratch.
|
|
562
|
+
- Change as LITTLE as possible. Fix ONLY the specific findings listed in the user message.
|
|
563
|
+
- PRESERVE every spec the agent declared — ESPECIALLY every propsSpec property (agent-owned render-time seed data the UI needs). Never drop it.
|
|
564
|
+
- Do NOT move data between specs (e.g. propsSpec → contextSpec) unless a finding explicitly requires it. A collection the agent supplies on propsSpec STAYS on propsSpec.
|
|
565
|
+
- Keep the agent's names, shapes, and structure intact wherever the findings do not force a change.
|
|
566
|
+
The four-spec model and placement rules below still hold — but in PATCH MODE they are guardrails for the fix, not a license to re-author.`;
|
|
567
|
+
|
|
568
|
+
/**
|
|
569
|
+
* Preservation-failure repair note (L1) — drives a corrective retry when
|
|
570
|
+
* a VALID candidate dropped an agent-owned propsSpec seed surface. Names
|
|
571
|
+
* the exact missing keys so the model restores them, rather than the
|
|
572
|
+
* generic "fix this validation error" note (the candidate IS valid; the
|
|
573
|
+
* problem is round-trip fidelity the gate can't express).
|
|
574
|
+
*/
|
|
575
|
+
function buildPreservationRepairNote(
|
|
576
|
+
rejected: unknown,
|
|
577
|
+
dropped: readonly string[],
|
|
578
|
+
): string {
|
|
579
|
+
const json = JSON.stringify(rejected);
|
|
580
|
+
const capped = json.length > 3000 ? `${json.slice(0, 3000)}…` : json;
|
|
581
|
+
return [
|
|
582
|
+
'YOUR PREVIOUS ATTEMPT dropped agent-owned render-time data. You returned this contract:',
|
|
583
|
+
capped,
|
|
584
|
+
'',
|
|
585
|
+
`The agent's draft declared these on propsSpec (agent-owned seed data the UI renders): ${dropped.join(', ')}. Your contract no longer carries them as propsSpec properties — so the agent can no longer seed them at render. contextSpec has NO agent seed channel, so moving them there leaves the UI empty.`,
|
|
586
|
+
'',
|
|
587
|
+
`Re-emit the contract with ${dropped.join(', ')} restored as propsSpec properties (agent-owned, seeded at render and refreshed via ggui_update). Keep every other spec unchanged.`,
|
|
588
|
+
].join('\n');
|
|
589
|
+
}
|
|
590
|
+
|
|
591
|
+
/**
|
|
592
|
+
* Compose the FIRST-attempt repair note for the forgiving-handshake
|
|
593
|
+
* path: the agent PROPOSED a contract that failed deterministic
|
|
594
|
+
* validation. Unlike {@link buildRepairNote} (which frames the input as
|
|
595
|
+
* "your previous attempt"), this frames it as the AGENT'S draft to be
|
|
596
|
+
* repaired in place — preserve intent + structure, fix exactly the
|
|
597
|
+
* listed findings. The draft JSON is capped to bound the prompt.
|
|
598
|
+
*/
|
|
599
|
+
function buildDraftSeedNote(
|
|
600
|
+
draft: unknown,
|
|
601
|
+
findings: readonly { code: string; path: string; message: string }[],
|
|
602
|
+
): string {
|
|
603
|
+
const json = JSON.stringify(draft);
|
|
604
|
+
const capped = json.length > 3000 ? `${json.slice(0, 3000)}…` : json;
|
|
605
|
+
const findingsText =
|
|
606
|
+
findings.length > 0
|
|
607
|
+
? findings.map((f) => `[${f.code}] ${f.path}: ${f.message}`).join('; ')
|
|
608
|
+
: '(unspecified — re-derive a valid contract for the intent)';
|
|
609
|
+
return [
|
|
610
|
+
'The agent PROPOSED this contract, but it failed deterministic validation:',
|
|
611
|
+
capped,
|
|
612
|
+
'',
|
|
613
|
+
`Validation findings — ${findingsText}`,
|
|
614
|
+
'',
|
|
615
|
+
"Repair it: keep the agent's intent and as much of their structure (spec names, schemas, labels) as possible, fix EXACTLY these findings, and return a valid contract. Do not add unrelated specs.",
|
|
616
|
+
].join('\n');
|
|
617
|
+
}
|
|
618
|
+
|
|
619
|
+
/**
|
|
620
|
+
* Invoke the synthesize tool — structured output when the provider
|
|
621
|
+
* supports it, else a text-JSON fallback. Forced tool use is
|
|
622
|
+
* Anthropic / Bedrock-only today; gemini / openai / openrouter callers
|
|
623
|
+
* reach this via the text path (the model emits one JSON object, which
|
|
624
|
+
* we regex-extract + parse). Lower reliability than forced tool use,
|
|
625
|
+
* but the bounded validate-and-repair loop catches malformed output and
|
|
626
|
+
* retries — so repair works on every provider instead of no-op'ing
|
|
627
|
+
* off-Anthropic.
|
|
628
|
+
*/
|
|
629
|
+
async function callSynthesizeTool(
|
|
630
|
+
llm: LLMCaller,
|
|
631
|
+
system: string,
|
|
632
|
+
user: string,
|
|
633
|
+
): Promise<unknown> {
|
|
634
|
+
if (typeof llm.callStructured === 'function') {
|
|
635
|
+
return llm.callStructured<SynthesizeToolInput>(
|
|
636
|
+
system,
|
|
637
|
+
user,
|
|
638
|
+
SYNTHESIZE_TOOL,
|
|
639
|
+
1024,
|
|
640
|
+
);
|
|
641
|
+
}
|
|
642
|
+
const text = await llm.call(
|
|
643
|
+
system,
|
|
644
|
+
`${user}\n\nRespond with ONE JSON object only (no prose, no code fence) carrying the synthesized contract: {actionSpec?, contextSpec?, streamSpec?, propsSpec?, agentCapabilities?, clientCapabilities?, reason}. Every spec entry wraps its JSON Schema under a "schema" field.`,
|
|
645
|
+
1024,
|
|
646
|
+
);
|
|
647
|
+
return extractJsonObject(text);
|
|
648
|
+
}
|
|
649
|
+
|
|
650
|
+
/**
|
|
651
|
+
* Thrown by {@link extractJsonObject} when the text-fallback response
|
|
652
|
+
* carries no parseable JSON object — distinct from a transient network
|
|
653
|
+
* throw so the repair loop knows to feed a corrective "emit pure JSON"
|
|
654
|
+
* note (a network retry re-sends the same prompt instead).
|
|
655
|
+
*/
|
|
656
|
+
class SynthesizeTextParseError extends Error {
|
|
657
|
+
constructor(message: string) {
|
|
658
|
+
super(message);
|
|
659
|
+
this.name = 'SynthesizeTextParseError';
|
|
660
|
+
}
|
|
661
|
+
}
|
|
662
|
+
|
|
663
|
+
/**
|
|
664
|
+
* Robustly extract one JSON object from a possibly-prose-wrapped model
|
|
665
|
+
* response. A naive greedy `/\{[\s\S]*\}/` over-captures (first `{` to
|
|
666
|
+
* LAST `}`), throwing on "prose with a brace before the JSON" or "two
|
|
667
|
+
* objects". This tries, in order: (1) parse the trimmed text directly;
|
|
668
|
+
* (2) parse the contents of a ```json fence; (3) scan from the first `{`
|
|
669
|
+
* for the BALANCED closing `}` (string/escape aware). Throws only when
|
|
670
|
+
* nothing parses — the bounded validate-and-repair loop then retries
|
|
671
|
+
* with a corrective note.
|
|
672
|
+
*/
|
|
673
|
+
function extractJsonObject(text: string): unknown {
|
|
674
|
+
const trimmed = text.trim();
|
|
675
|
+
try {
|
|
676
|
+
return JSON.parse(trimmed);
|
|
677
|
+
} catch {
|
|
678
|
+
// fall through
|
|
679
|
+
}
|
|
680
|
+
const fence = trimmed.match(/```(?:json)?\s*([\s\S]*?)```/);
|
|
681
|
+
if (fence && fence[1]) {
|
|
682
|
+
try {
|
|
683
|
+
return JSON.parse(fence[1].trim());
|
|
684
|
+
} catch {
|
|
685
|
+
// fall through
|
|
686
|
+
}
|
|
687
|
+
}
|
|
688
|
+
// Scan each balanced {...} from each '{' and return the FIRST that
|
|
689
|
+
// parses — robust against prose braces BEFORE the JSON (e.g. "{your
|
|
690
|
+
// widget}: {...}") and a trailing second object.
|
|
691
|
+
let searchFrom = 0;
|
|
692
|
+
for (;;) {
|
|
693
|
+
const start = trimmed.indexOf('{', searchFrom);
|
|
694
|
+
if (start < 0) break;
|
|
695
|
+
let depth = 0;
|
|
696
|
+
let inStr = false;
|
|
697
|
+
let esc = false;
|
|
698
|
+
let end = -1;
|
|
699
|
+
for (let i = start; i < trimmed.length; i++) {
|
|
700
|
+
const ch = trimmed[i];
|
|
701
|
+
if (inStr) {
|
|
702
|
+
if (esc) esc = false;
|
|
703
|
+
else if (ch === '\\') esc = true;
|
|
704
|
+
else if (ch === '"') inStr = false;
|
|
705
|
+
} else if (ch === '"') {
|
|
706
|
+
inStr = true;
|
|
707
|
+
} else if (ch === '{') {
|
|
708
|
+
depth++;
|
|
709
|
+
} else if (ch === '}') {
|
|
710
|
+
depth--;
|
|
711
|
+
if (depth === 0) {
|
|
712
|
+
end = i;
|
|
713
|
+
break;
|
|
714
|
+
}
|
|
715
|
+
}
|
|
716
|
+
}
|
|
717
|
+
if (end < 0) break; // unbalanced from here on
|
|
718
|
+
try {
|
|
719
|
+
return JSON.parse(trimmed.slice(start, end + 1));
|
|
720
|
+
} catch {
|
|
721
|
+
// not valid JSON from this '{' — advance and try the next one
|
|
722
|
+
}
|
|
723
|
+
searchFrom = start + 1;
|
|
724
|
+
}
|
|
725
|
+
throw new SynthesizeTextParseError(
|
|
726
|
+
`synthesize text-fallback: no parseable JSON object in response (length ${text.length})`,
|
|
727
|
+
);
|
|
728
|
+
}
|
|
729
|
+
|
|
538
730
|
/**
|
|
539
731
|
* Drop `actionSpec` entries the structural validator flags as
|
|
540
732
|
* `redundant-action` — an empty-payload action whose name is a
|
|
@@ -574,8 +766,14 @@ function pruneRedundantActions(contract: DataContract): DataContract {
|
|
|
574
766
|
* Empty / whitespace intent short-circuits to null with a reason —
|
|
575
767
|
* no contract can be inferred from nothing.
|
|
576
768
|
*
|
|
577
|
-
*
|
|
578
|
-
*
|
|
769
|
+
* Providers without `callStructured` (gemini / openai / openrouter) use
|
|
770
|
+
* a text-JSON fallback (the validate-and-repair loop catches malformed
|
|
771
|
+
* output and retries) rather than skipping synthesis.
|
|
772
|
+
*
|
|
773
|
+
* When `options.draft` is supplied, the loop REPAIRS that draft in
|
|
774
|
+
* place (seeded with the agent's contract + the deterministic findings)
|
|
775
|
+
* instead of synthesizing from `intent` alone — the forgiving-handshake
|
|
776
|
+
* path.
|
|
579
777
|
*
|
|
580
778
|
* Each attempt is self-checked against the validation gate; a failure
|
|
581
779
|
* feeds the precise error back for up to {@link MAX_SYNTH_ATTEMPTS}
|
|
@@ -602,6 +800,30 @@ export async function synthesizeContract(
|
|
|
602
800
|
* no-app-registry path).
|
|
603
801
|
*/
|
|
604
802
|
readonly appGadgets?: readonly GadgetDescriptor[];
|
|
803
|
+
/**
|
|
804
|
+
* Repair-in-place seed. When provided, the synthesizer does NOT
|
|
805
|
+
* synthesize from `intent` alone — it starts from the agent's
|
|
806
|
+
* proposed `draft` and the deterministic findings that rejected it,
|
|
807
|
+
* and the validate-and-repair loop corrects exactly those problems
|
|
808
|
+
* while preserving the agent's intent + structure. This is the
|
|
809
|
+
* forgiving-handshake path: an invalid agent draft is the loop's
|
|
810
|
+
* SEED rather than a thrown error. Absent ⇒ classic
|
|
811
|
+
* synthesize-from-intent (cold path). Typed `unknown` because the
|
|
812
|
+
* agent's draft is untrusted — it may not be a valid DataContract
|
|
813
|
+
* (that's the whole point of repairing it).
|
|
814
|
+
*/
|
|
815
|
+
readonly draft?: unknown;
|
|
816
|
+
/**
|
|
817
|
+
* Deterministic validation findings that rejected {@link draft}
|
|
818
|
+
* (from `lintContract(draft).errors`). Fed into the first repair
|
|
819
|
+
* note so the model corrects the precise problems. Ignored when
|
|
820
|
+
* `draft` is absent.
|
|
821
|
+
*/
|
|
822
|
+
readonly draftFindings?: readonly {
|
|
823
|
+
readonly code: string;
|
|
824
|
+
readonly path: string;
|
|
825
|
+
readonly message: string;
|
|
826
|
+
}[];
|
|
605
827
|
},
|
|
606
828
|
): Promise<SynthesizeContractResult> {
|
|
607
829
|
const startedAt = Date.now();
|
|
@@ -616,17 +838,6 @@ export async function synthesizeContract(
|
|
|
616
838
|
};
|
|
617
839
|
}
|
|
618
840
|
|
|
619
|
-
if (typeof deps.llm.callStructured !== 'function') {
|
|
620
|
-
return {
|
|
621
|
-
contract: null,
|
|
622
|
-
reason:
|
|
623
|
-
'synthesize-skip: provider does not support callStructured. Bind a structured-capable LLMCaller (Anthropic adapter) to enable contract synthesis.',
|
|
624
|
-
latencyMs: Date.now() - startedAt,
|
|
625
|
-
attempts: 0,
|
|
626
|
-
findings: [],
|
|
627
|
-
};
|
|
628
|
-
}
|
|
629
|
-
|
|
630
841
|
const gadgetsSection = composeAvailableGadgetsSection(
|
|
631
842
|
options?.appGadgets,
|
|
632
843
|
);
|
|
@@ -642,10 +853,30 @@ export async function synthesizeContract(
|
|
|
642
853
|
// small enough that a full re-emit IS the surgical edit. Budget
|
|
643
854
|
// exhausted → decline exactly as the one-shot path did (caller falls
|
|
644
855
|
// back to an empty contract stub).
|
|
645
|
-
|
|
856
|
+
// Repair-in-place seed: when the caller passed the agent's rejected
|
|
857
|
+
// draft, the loop's FIRST attempt repairs that draft (not a blank
|
|
858
|
+
// synthesis). Later attempts overwrite this via buildRepairNote.
|
|
859
|
+
let repairNote: string | undefined =
|
|
860
|
+
options?.draft !== undefined
|
|
861
|
+
? buildDraftSeedNote(options.draft, options.draftFindings ?? [])
|
|
862
|
+
: undefined;
|
|
646
863
|
let lastReason = 'synthesize-fail: exhausted repair attempts';
|
|
647
864
|
let lastFindings: readonly ContractValidationFinding[] = [];
|
|
648
865
|
|
|
866
|
+
// L2 — repair path uses the PATCH-MODE preamble (minimal patch, no
|
|
867
|
+
// re-author); cold path uses the bare authoring prompt.
|
|
868
|
+
const systemPrompt =
|
|
869
|
+
options?.draft !== undefined
|
|
870
|
+
? `${REPAIR_PREAMBLE}\n\n${SYNTHESIZE_SYSTEM_PROMPT}`
|
|
871
|
+
: SYNTHESIZE_SYSTEM_PROMPT;
|
|
872
|
+
|
|
873
|
+
// L1 — the agent-owned seed surfaces the repaired contract MUST keep,
|
|
874
|
+
// and the best valid-but-non-preserving candidate to fall back on so
|
|
875
|
+
// preservation never makes the result WORSE than the validity-only gate.
|
|
876
|
+
const draftSeedKeys =
|
|
877
|
+
options?.draft !== undefined ? draftSeedPropKeys(options.draft) : [];
|
|
878
|
+
let lastValidContract: DataContract | null = null;
|
|
879
|
+
|
|
649
880
|
for (let attempt = 1; attempt <= MAX_SYNTH_ATTEMPTS; attempt++) {
|
|
650
881
|
const userPrompt =
|
|
651
882
|
repairNote === undefined
|
|
@@ -654,16 +885,22 @@ export async function synthesizeContract(
|
|
|
654
885
|
|
|
655
886
|
let toolInput: unknown;
|
|
656
887
|
try {
|
|
657
|
-
toolInput = await
|
|
658
|
-
|
|
888
|
+
toolInput = await callSynthesizeTool(
|
|
889
|
+
deps.llm,
|
|
890
|
+
systemPrompt,
|
|
659
891
|
userPrompt,
|
|
660
|
-
SYNTHESIZE_TOOL,
|
|
661
|
-
1024,
|
|
662
892
|
);
|
|
663
893
|
} catch (err) {
|
|
664
|
-
|
|
665
|
-
//
|
|
666
|
-
|
|
894
|
+
lastReason = `synthesize-fail: callSynthesizeTool threw — ${err instanceof Error ? err.message : String(err)}`;
|
|
895
|
+
// Distinguish causes: an UNPARSEABLE text-path response (model
|
|
896
|
+
// emitted prose/non-JSON — common on gemini/openai via the text
|
|
897
|
+
// fallback) gets a corrective note so the next attempt emits pure
|
|
898
|
+
// JSON. A transient NETWORK failure re-sends the SAME prompt
|
|
899
|
+
// (nothing to correct) — preserving the retry semantics.
|
|
900
|
+
if (err instanceof SynthesizeTextParseError) {
|
|
901
|
+
repairNote =
|
|
902
|
+
'Your previous response could not be parsed as JSON. Respond with EXACTLY ONE JSON object and nothing else — no prose, no explanation, no markdown code fence.';
|
|
903
|
+
}
|
|
667
904
|
continue;
|
|
668
905
|
}
|
|
669
906
|
|
|
@@ -741,6 +978,24 @@ export async function synthesizeContract(
|
|
|
741
978
|
continue;
|
|
742
979
|
}
|
|
743
980
|
|
|
981
|
+
// L1 — preservation gate (best-effort). The candidate is VALID, but
|
|
982
|
+
// a repair must not silently drop an agent-owned seed surface the
|
|
983
|
+
// draft declared on propsSpec (the valid-but-round-trip-broken
|
|
984
|
+
// reshape lintContract + the placement validators can't see).
|
|
985
|
+
// Deterministic + model-independent: it drives a corrective retry.
|
|
986
|
+
if (draftSeedKeys.length > 0) {
|
|
987
|
+
const dropped = findDroppedSeedSurfaces(options?.draft, validatedContract);
|
|
988
|
+
if (dropped.length > 0) {
|
|
989
|
+
// Remember the best VALID candidate so an exhausted budget never
|
|
990
|
+
// returns WORSE than the validity-only gate did (a valid contract).
|
|
991
|
+
lastValidContract = validatedContract;
|
|
992
|
+
lastReason = `synthesize-preservation: candidate dropped agent-owned propsSpec seed surface(s) [${dropped.join(', ')}]`;
|
|
993
|
+
lastFindings = allFindings;
|
|
994
|
+
repairNote = buildPreservationRepairNote(validatedContract, dropped);
|
|
995
|
+
continue;
|
|
996
|
+
}
|
|
997
|
+
}
|
|
998
|
+
|
|
744
999
|
const findingsSuffix =
|
|
745
1000
|
allFindings.length > 0
|
|
746
1001
|
? ` — validator: ${formatValidationFindings({ findings: allFindings })}`
|
|
@@ -757,9 +1012,16 @@ export async function synthesizeContract(
|
|
|
757
1012
|
};
|
|
758
1013
|
}
|
|
759
1014
|
|
|
1015
|
+
// Budget exhausted. If preservation retries never landed a contract
|
|
1016
|
+
// that kept every seed surface, fall back to the best VALID candidate
|
|
1017
|
+
// we did produce (never worse than the validity-only gate). Only when
|
|
1018
|
+
// no valid candidate ever appeared do we decline with `null`.
|
|
760
1019
|
return {
|
|
761
|
-
contract:
|
|
762
|
-
reason:
|
|
1020
|
+
contract: lastValidContract,
|
|
1021
|
+
reason:
|
|
1022
|
+
lastValidContract !== null
|
|
1023
|
+
? `${lastReason}; returned best valid candidate (preservation retries exhausted)`
|
|
1024
|
+
: lastReason,
|
|
763
1025
|
latencyMs: Date.now() - startedAt,
|
|
764
1026
|
attempts: MAX_SYNTH_ATTEMPTS,
|
|
765
1027
|
findings: lastFindings,
|