@ggui-ai/negotiator 0.2.0-alpha.3 → 0.2.0-alpha.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ensure-conforming-contract.d.ts +70 -0
- package/dist/ensure-conforming-contract.d.ts.map +1 -0
- package/dist/ensure-conforming-contract.js +115 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/normalize-draft.d.ts +33 -0
- package/dist/normalize-draft.d.ts.map +1 -0
- package/dist/normalize-draft.js +143 -0
- package/dist/preserve-seed-surfaces.d.ts +40 -0
- package/dist/preserve-seed-surfaces.d.ts.map +1 -0
- package/dist/preserve-seed-surfaces.js +57 -0
- package/dist/synth-bench/cli-llm.d.ts +20 -0
- package/dist/synth-bench/cli-llm.d.ts.map +1 -0
- package/dist/synth-bench/cli-llm.js +97 -0
- package/dist/synth-bench/corpus.d.ts +52 -0
- package/dist/synth-bench/corpus.d.ts.map +1 -1
- package/dist/synth-bench/corpus.js +306 -5
- package/dist/synth-bench/round-trip-score.d.ts +87 -0
- package/dist/synth-bench/round-trip-score.d.ts.map +1 -0
- package/dist/synth-bench/round-trip-score.js +105 -0
- package/dist/synth-bench/run-bench-cli.js +6 -82
- package/dist/synth-bench/run-repair-bench-cli.d.ts +3 -0
- package/dist/synth-bench/run-repair-bench-cli.d.ts.map +1 -0
- package/dist/synth-bench/run-repair-bench-cli.js +86 -0
- package/dist/synth-bench/run-repair-bench.d.ts +94 -0
- package/dist/synth-bench/run-repair-bench.d.ts.map +1 -0
- package/dist/synth-bench/run-repair-bench.js +172 -0
- package/dist/synthesize-contract.d.ts +38 -6
- package/dist/synthesize-contract.d.ts.map +1 -1
- package/dist/synthesize-contract.js +246 -32
- package/package.json +5 -4
- package/src/ensure-conforming-contract.ts +175 -0
- package/src/index.ts +2 -0
- package/src/normalize-draft.ts +156 -0
- package/src/preserve-seed-surfaces.ts +61 -0
- package/src/synth-bench/cli-llm.ts +140 -0
- package/src/synth-bench/corpus.ts +335 -5
- package/src/synth-bench/round-trip-score.ts +169 -0
- package/src/synth-bench/run-bench-cli.ts +13 -115
- package/src/synth-bench/run-repair-bench-cli.ts +119 -0
- package/src/synth-bench/run-repair-bench.ts +266 -0
- package/src/synthesize-contract.ts +299 -37
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Deterministic draft normalization — the cheap, faithful repair tier.
|
|
3
|
+
*
|
|
4
|
+
* Most agent-draft malformations are MECHANICAL: a stray illegal key on
|
|
5
|
+
* a spec wrapper (a JSON-Schema-reflex `required: [...]` array on the
|
|
6
|
+
* propsSpec wrapper, an `additionalProperties` key), or a non-canonical
|
|
7
|
+
* schema `type` spelling (`"enum"`, `"integer"`). None of these needs an
|
|
8
|
+
* LLM to fix — and routing them through the LLM repair loop is both
|
|
9
|
+
* wasteful (a full regeneration) and RISKY (the model may re-author and
|
|
10
|
+
* reshape a 95%-correct draft, e.g. drop a propsSpec seed surface).
|
|
11
|
+
*
|
|
12
|
+
* This pass fixes the mechanical classes deterministically, preserving
|
|
13
|
+
* the agent's intent exactly:
|
|
14
|
+
* - strips keys the protocol's `.strict()` spec schemas would reject,
|
|
15
|
+
* keeping only the allowed keys at each wrapper / entry level;
|
|
16
|
+
* - canonicalizes every inner JSON Schema via {@link normalizeSchema}
|
|
17
|
+
* (the same normalizer `buildContract` runs on synth output).
|
|
18
|
+
*
|
|
19
|
+
* The caller (`ensureConformingContract`) re-lints the result: if it now
|
|
20
|
+
* passes the gate, the draft is returned WITHOUT ever calling the LLM.
|
|
21
|
+
* Semantic deficiencies (wrong placement, missing data surface, dangling
|
|
22
|
+
* cross-refs) are deliberately out of scope — those still go to the
|
|
23
|
+
* repair loop, where reasoning earns its keep.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import { normalizeSchema } from './normalize-schema.js';
|
|
27
|
+
|
|
28
|
+
// Allowed-key sets mirror the protocol `.strict()` schemas
|
|
29
|
+
// (schemas/data-contract.ts). Stripping anything outside these is safe:
|
|
30
|
+
// the strict schema would reject it as CTR_SHAPE_UNRECOGNIZED_KEYS.
|
|
31
|
+
const PROPS_WRAPPER_KEYS = new Set(['description', 'properties']);
|
|
32
|
+
const PROP_ENTRY_KEYS = new Set([
|
|
33
|
+
'description',
|
|
34
|
+
'schema',
|
|
35
|
+
'required',
|
|
36
|
+
'default',
|
|
37
|
+
'example',
|
|
38
|
+
'sourceTool',
|
|
39
|
+
]);
|
|
40
|
+
const CONTEXT_ENTRY_KEYS = new Set([
|
|
41
|
+
'description',
|
|
42
|
+
'schema',
|
|
43
|
+
'default',
|
|
44
|
+
'debounceMs',
|
|
45
|
+
'example',
|
|
46
|
+
]);
|
|
47
|
+
const ACTION_ENTRY_KEYS = new Set([
|
|
48
|
+
'description',
|
|
49
|
+
'label',
|
|
50
|
+
'schema',
|
|
51
|
+
'example',
|
|
52
|
+
'icon',
|
|
53
|
+
'confirm',
|
|
54
|
+
'nextStep',
|
|
55
|
+
]);
|
|
56
|
+
const STREAM_ENTRY_KEYS = new Set(['description', 'schema', 'source']);
|
|
57
|
+
const AGENT_TOOL_KEYS = new Set([
|
|
58
|
+
'description',
|
|
59
|
+
'usage',
|
|
60
|
+
'inputSchema',
|
|
61
|
+
'outputSchema',
|
|
62
|
+
'required',
|
|
63
|
+
'example',
|
|
64
|
+
]);
|
|
65
|
+
|
|
66
|
+
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
67
|
+
return typeof value === 'object' && value !== null && !Array.isArray(value);
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** Keep only `allowed` keys; normalize the schema-bearing fields named
|
|
71
|
+
* in `schemaFields`. Non-record entries pass through untouched. */
|
|
72
|
+
function cleanEntry(
|
|
73
|
+
entry: unknown,
|
|
74
|
+
allowed: ReadonlySet<string>,
|
|
75
|
+
schemaFields: readonly string[],
|
|
76
|
+
): unknown {
|
|
77
|
+
if (!isRecord(entry)) return entry;
|
|
78
|
+
const out: Record<string, unknown> = {};
|
|
79
|
+
for (const [key, value] of Object.entries(entry)) {
|
|
80
|
+
if (!allowed.has(key)) continue; // strip the illegal key
|
|
81
|
+
out[key] =
|
|
82
|
+
schemaFields.includes(key) && value !== undefined
|
|
83
|
+
? normalizeSchema(value)
|
|
84
|
+
: value;
|
|
85
|
+
}
|
|
86
|
+
return out;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/** Apply {@link cleanEntry} across a `Record<name, entry>` spec map. */
|
|
90
|
+
function cleanEntryMap(
|
|
91
|
+
map: Record<string, unknown>,
|
|
92
|
+
allowed: ReadonlySet<string>,
|
|
93
|
+
schemaFields: readonly string[],
|
|
94
|
+
): Record<string, unknown> {
|
|
95
|
+
const out: Record<string, unknown> = {};
|
|
96
|
+
for (const [name, entry] of Object.entries(map)) {
|
|
97
|
+
out[name] = cleanEntry(entry, allowed, schemaFields);
|
|
98
|
+
}
|
|
99
|
+
return out;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Return a structurally-normalized copy of an untrusted draft: illegal
|
|
104
|
+
* wrapper/entry keys stripped, inner schemas canonicalized. Pure — never
|
|
105
|
+
* mutates the input, never throws. Unknown top-level fields ride through
|
|
106
|
+
* (the top-level DataContract schema is `.passthrough()`); only the
|
|
107
|
+
* `.strict()` spec wrappers and entries are cleaned.
|
|
108
|
+
*/
|
|
109
|
+
export function normalizeDraft(draft: unknown): unknown {
|
|
110
|
+
if (!isRecord(draft)) return draft;
|
|
111
|
+
const out: Record<string, unknown> = { ...draft };
|
|
112
|
+
|
|
113
|
+
// propsSpec wrapper: keep {description, properties}; clean each PropEntry.
|
|
114
|
+
if (isRecord(out['propsSpec'])) {
|
|
115
|
+
const ps: Record<string, unknown> = {};
|
|
116
|
+
for (const [key, value] of Object.entries(out['propsSpec'])) {
|
|
117
|
+
if (PROPS_WRAPPER_KEYS.has(key)) ps[key] = value;
|
|
118
|
+
}
|
|
119
|
+
if (isRecord(ps['properties'])) {
|
|
120
|
+
ps['properties'] = cleanEntryMap(ps['properties'], PROP_ENTRY_KEYS, [
|
|
121
|
+
'schema',
|
|
122
|
+
]);
|
|
123
|
+
}
|
|
124
|
+
out['propsSpec'] = ps;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
if (isRecord(out['contextSpec'])) {
|
|
128
|
+
out['contextSpec'] = cleanEntryMap(out['contextSpec'], CONTEXT_ENTRY_KEYS, [
|
|
129
|
+
'schema',
|
|
130
|
+
]);
|
|
131
|
+
}
|
|
132
|
+
if (isRecord(out['actionSpec'])) {
|
|
133
|
+
out['actionSpec'] = cleanEntryMap(out['actionSpec'], ACTION_ENTRY_KEYS, [
|
|
134
|
+
'schema',
|
|
135
|
+
]);
|
|
136
|
+
}
|
|
137
|
+
if (isRecord(out['streamSpec'])) {
|
|
138
|
+
out['streamSpec'] = cleanEntryMap(out['streamSpec'], STREAM_ENTRY_KEYS, [
|
|
139
|
+
'schema',
|
|
140
|
+
]);
|
|
141
|
+
}
|
|
142
|
+
if (isRecord(out['agentCapabilities'])) {
|
|
143
|
+
const ac = out['agentCapabilities'];
|
|
144
|
+
if (isRecord(ac['tools'])) {
|
|
145
|
+
out['agentCapabilities'] = {
|
|
146
|
+
...ac,
|
|
147
|
+
tools: cleanEntryMap(ac['tools'], AGENT_TOOL_KEYS, [
|
|
148
|
+
'inputSchema',
|
|
149
|
+
'outputSchema',
|
|
150
|
+
]),
|
|
151
|
+
};
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
return out;
|
|
156
|
+
}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Seed-surface preservation — the deterministic backstop that keeps the
|
|
3
|
+
* repair loop FAITHFUL.
|
|
4
|
+
*
|
|
5
|
+
* The agent's draft declares its agent-owned render-time data as
|
|
6
|
+
* `propsSpec.properties` (the only agent→client seed channel —
|
|
7
|
+
* `contextSpec` has no runtime seed path). A repair that drops or
|
|
8
|
+
* reshapes one of those keys away from propsSpec (the canonical
|
|
9
|
+
* `propsSpec.X → contextSpec.X` regression) produces a contract that is
|
|
10
|
+
* VALID (`lintContract` passes) yet round-trip-BROKEN: the agent can no
|
|
11
|
+
* longer seed X at render, so the UI renders empty.
|
|
12
|
+
*
|
|
13
|
+
* `lintContract` + the placement validators can't catch this — none of
|
|
14
|
+
* them sees the agent's DRAFT. These helpers do: they compare the
|
|
15
|
+
* repaired candidate against the draft and report which agent-owned seed
|
|
16
|
+
* surfaces went missing, so the synth loop can drive a corrective retry.
|
|
17
|
+
* Model-independent — it works even when a weak model keeps reshaping.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
import type { DataContract } from '@ggui-ai/protocol';
|
|
21
|
+
|
|
22
|
+
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
23
|
+
return typeof value === 'object' && value !== null && !Array.isArray(value);
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* The `propsSpec.properties` keys an agent declared on a (possibly
|
|
28
|
+
* malformed) draft — its agent-owned render-time SEED surfaces.
|
|
29
|
+
* Defensive: the draft is untrusted, so every level is probed before
|
|
30
|
+
* access. Returns `[]` for any non-propsSpec-bearing draft.
|
|
31
|
+
*/
|
|
32
|
+
export function draftSeedPropKeys(draft: unknown): string[] {
|
|
33
|
+
if (!isRecord(draft)) return [];
|
|
34
|
+
const propsSpec = draft['propsSpec'];
|
|
35
|
+
if (!isRecord(propsSpec)) return [];
|
|
36
|
+
const properties = propsSpec['properties'];
|
|
37
|
+
if (!isRecord(properties)) return [];
|
|
38
|
+
return Object.keys(properties);
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Agent-owned seed surfaces (propsSpec property keys) present in `draft`
|
|
43
|
+
* that the repaired `candidate` DROPPED — i.e. they are no longer
|
|
44
|
+
* seedable as a propsSpec property. Reshaping `propsSpec.X` to
|
|
45
|
+
* `contextSpec.X`, or dropping it entirely, both surface here (contextSpec
|
|
46
|
+
* is not an agent-seedable home). Returns `[]` when every declared seed
|
|
47
|
+
* surface survived (preservation holds).
|
|
48
|
+
*
|
|
49
|
+
* Preservation-biased on purpose: keeping a seed key the intent turned
|
|
50
|
+
* out not to need is harmless (an unsent optional prop); DROPPING one the
|
|
51
|
+
* agent relies on is the round-trip break. So we only ever flag drops.
|
|
52
|
+
*/
|
|
53
|
+
export function findDroppedSeedSurfaces(
|
|
54
|
+
draft: unknown,
|
|
55
|
+
candidate: DataContract,
|
|
56
|
+
): string[] {
|
|
57
|
+
const seedKeys = draftSeedPropKeys(draft);
|
|
58
|
+
if (seedKeys.length === 0) return [];
|
|
59
|
+
const candidateProps = candidate.propsSpec?.properties ?? {};
|
|
60
|
+
return seedKeys.filter((key) => candidateProps[key] === undefined);
|
|
61
|
+
}
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared CLI LLM plumbing for the synth-bench family of live probes
|
|
3
|
+
* (bench-synth, bench-repair). Builds an Anthropic-backed
|
|
4
|
+
* {@link LLMCaller} with forced tool-use, resolves the API key from env
|
|
5
|
+
* or `~/.ggui/credentials.json`, and accumulates token usage for the
|
|
6
|
+
* per-run cost line. Bench-only — not exported from the package index.
|
|
7
|
+
*/
|
|
8
|
+
import { readFileSync } from 'node:fs';
|
|
9
|
+
import { homedir } from 'node:os';
|
|
10
|
+
import { resolve as pathResolve } from 'node:path';
|
|
11
|
+
import type { LLMCaller, ToolSchema } from '../llm-caller.js';
|
|
12
|
+
|
|
13
|
+
const ANTHROPIC_API = 'https://api.anthropic.com/v1/messages';
|
|
14
|
+
|
|
15
|
+
/** Default bench model — matches the `ui-gen-default-haiku-4-5` slug. */
|
|
16
|
+
export const DEFAULT_MODEL = 'claude-haiku-4-5';
|
|
17
|
+
|
|
18
|
+
/** Haiku 4.5 token pricing (USD per token) for the cost report. */
|
|
19
|
+
export const HAIKU_4_5_PRICE_INPUT_PER_TOKEN = 1.0 / 1_000_000;
|
|
20
|
+
export const HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN = 5.0 / 1_000_000;
|
|
21
|
+
|
|
22
|
+
interface CredsFile {
|
|
23
|
+
apps?: { global?: { anthropic?: string } };
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Resolve the Anthropic API key: `ANTHROPIC_API_KEY` env var first, then
|
|
28
|
+
* `~/.ggui/credentials.json` at `apps.global.anthropic`. `scriptName`
|
|
29
|
+
* prefixes the error hints so a failure names the bench that needs it.
|
|
30
|
+
*/
|
|
31
|
+
export function resolveAnthropicKey(scriptName: string): string {
|
|
32
|
+
const envKey = process.env['ANTHROPIC_API_KEY'];
|
|
33
|
+
if (envKey && envKey.length > 0) return envKey;
|
|
34
|
+
const credsPath = pathResolve(homedir(), '.ggui', 'credentials.json');
|
|
35
|
+
let parsed: CredsFile;
|
|
36
|
+
try {
|
|
37
|
+
parsed = JSON.parse(readFileSync(credsPath, 'utf8')) as CredsFile;
|
|
38
|
+
} catch (err) {
|
|
39
|
+
throw new Error(
|
|
40
|
+
`${scriptName}: could not read ${credsPath} (${err instanceof Error ? err.message : String(err)}). Set ANTHROPIC_API_KEY env var or run \`ggui auth set anthropic\`.`,
|
|
41
|
+
);
|
|
42
|
+
}
|
|
43
|
+
const key = parsed.apps?.global?.anthropic;
|
|
44
|
+
if (typeof key !== 'string' || key.length === 0) {
|
|
45
|
+
throw new Error(
|
|
46
|
+
`${scriptName}: no anthropic key found at apps.global.anthropic in ${credsPath}.`,
|
|
47
|
+
);
|
|
48
|
+
}
|
|
49
|
+
return key;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
interface AnthropicContentBlock {
|
|
53
|
+
type: string;
|
|
54
|
+
name?: string;
|
|
55
|
+
input?: unknown;
|
|
56
|
+
text?: string;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
interface AnthropicResponse {
|
|
60
|
+
content?: AnthropicContentBlock[];
|
|
61
|
+
usage?: { input_tokens?: number; output_tokens?: number };
|
|
62
|
+
stop_reason?: string;
|
|
63
|
+
error?: { type?: string; message?: string };
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
let totalInputTokens = 0;
|
|
67
|
+
let totalOutputTokens = 0;
|
|
68
|
+
|
|
69
|
+
/** Running token usage accumulated across `callStructured` calls in this
|
|
70
|
+
* process — read after a run for the cost line. */
|
|
71
|
+
export function getTokenUsage(): {
|
|
72
|
+
readonly input: number;
|
|
73
|
+
readonly output: number;
|
|
74
|
+
} {
|
|
75
|
+
return { input: totalInputTokens, output: totalOutputTokens };
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
export function buildAnthropicLlmCaller(
|
|
79
|
+
apiKey: string,
|
|
80
|
+
model: string,
|
|
81
|
+
): LLMCaller {
|
|
82
|
+
return {
|
|
83
|
+
async call(): Promise<string> {
|
|
84
|
+
throw new Error(
|
|
85
|
+
'synth-bench: text-mode not exercised — synth uses callStructured',
|
|
86
|
+
);
|
|
87
|
+
},
|
|
88
|
+
async callStructured<T>(
|
|
89
|
+
systemPrompt: string,
|
|
90
|
+
userMessage: string,
|
|
91
|
+
tool: ToolSchema,
|
|
92
|
+
maxTokens?: number,
|
|
93
|
+
): Promise<T> {
|
|
94
|
+
// `temperature` deprecated on Haiku 4.5+ — Anthropic rejects with
|
|
95
|
+
// HTTP 400. `tool_choice: { type: 'tool', name }` below already
|
|
96
|
+
// binds output to the input_schema; residual stochasticity stays
|
|
97
|
+
// bounded via canonical-key normalization downstream.
|
|
98
|
+
const body = {
|
|
99
|
+
model,
|
|
100
|
+
max_tokens: maxTokens ?? 1024,
|
|
101
|
+
system: systemPrompt,
|
|
102
|
+
messages: [{ role: 'user', content: userMessage }],
|
|
103
|
+
tools: [
|
|
104
|
+
{
|
|
105
|
+
name: tool.name,
|
|
106
|
+
description: tool.description,
|
|
107
|
+
input_schema: tool.input_schema,
|
|
108
|
+
},
|
|
109
|
+
],
|
|
110
|
+
tool_choice: { type: 'tool', name: tool.name },
|
|
111
|
+
};
|
|
112
|
+
const res = await fetch(ANTHROPIC_API, {
|
|
113
|
+
method: 'POST',
|
|
114
|
+
headers: {
|
|
115
|
+
'content-type': 'application/json',
|
|
116
|
+
'x-api-key': apiKey,
|
|
117
|
+
'anthropic-version': '2023-06-01',
|
|
118
|
+
},
|
|
119
|
+
body: JSON.stringify(body),
|
|
120
|
+
});
|
|
121
|
+
const json = (await res.json()) as AnthropicResponse;
|
|
122
|
+
if (!res.ok) {
|
|
123
|
+
const errType = json.error?.type ?? 'unknown';
|
|
124
|
+
const errMsg = json.error?.message ?? `HTTP ${res.status}`;
|
|
125
|
+
throw new Error(`anthropic ${errType}: ${errMsg}`);
|
|
126
|
+
}
|
|
127
|
+
if (json.usage) {
|
|
128
|
+
totalInputTokens += json.usage.input_tokens ?? 0;
|
|
129
|
+
totalOutputTokens += json.usage.output_tokens ?? 0;
|
|
130
|
+
}
|
|
131
|
+
const toolBlock = json.content?.find((b) => b.type === 'tool_use');
|
|
132
|
+
if (!toolBlock || toolBlock.input === undefined) {
|
|
133
|
+
throw new Error(
|
|
134
|
+
`anthropic: no tool_use block in response (stop_reason=${json.stop_reason ?? 'unknown'})`,
|
|
135
|
+
);
|
|
136
|
+
}
|
|
137
|
+
return toolBlock.input as T;
|
|
138
|
+
},
|
|
139
|
+
};
|
|
140
|
+
}
|