@ggui-ai/negotiator 0.2.0-alpha.3 → 0.2.0-alpha.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/dist/ensure-conforming-contract.d.ts +70 -0
  2. package/dist/ensure-conforming-contract.d.ts.map +1 -0
  3. package/dist/ensure-conforming-contract.js +115 -0
  4. package/dist/index.d.ts +2 -0
  5. package/dist/index.d.ts.map +1 -1
  6. package/dist/index.js +1 -0
  7. package/dist/normalize-draft.d.ts +33 -0
  8. package/dist/normalize-draft.d.ts.map +1 -0
  9. package/dist/normalize-draft.js +143 -0
  10. package/dist/preserve-seed-surfaces.d.ts +40 -0
  11. package/dist/preserve-seed-surfaces.d.ts.map +1 -0
  12. package/dist/preserve-seed-surfaces.js +57 -0
  13. package/dist/synth-bench/cli-llm.d.ts +20 -0
  14. package/dist/synth-bench/cli-llm.d.ts.map +1 -0
  15. package/dist/synth-bench/cli-llm.js +97 -0
  16. package/dist/synth-bench/corpus.d.ts +52 -0
  17. package/dist/synth-bench/corpus.d.ts.map +1 -1
  18. package/dist/synth-bench/corpus.js +306 -5
  19. package/dist/synth-bench/round-trip-score.d.ts +87 -0
  20. package/dist/synth-bench/round-trip-score.d.ts.map +1 -0
  21. package/dist/synth-bench/round-trip-score.js +105 -0
  22. package/dist/synth-bench/run-bench-cli.js +6 -82
  23. package/dist/synth-bench/run-repair-bench-cli.d.ts +3 -0
  24. package/dist/synth-bench/run-repair-bench-cli.d.ts.map +1 -0
  25. package/dist/synth-bench/run-repair-bench-cli.js +86 -0
  26. package/dist/synth-bench/run-repair-bench.d.ts +94 -0
  27. package/dist/synth-bench/run-repair-bench.d.ts.map +1 -0
  28. package/dist/synth-bench/run-repair-bench.js +172 -0
  29. package/dist/synthesize-contract.d.ts +38 -6
  30. package/dist/synthesize-contract.d.ts.map +1 -1
  31. package/dist/synthesize-contract.js +246 -32
  32. package/package.json +5 -4
  33. package/src/ensure-conforming-contract.ts +175 -0
  34. package/src/index.ts +2 -0
  35. package/src/normalize-draft.ts +156 -0
  36. package/src/preserve-seed-surfaces.ts +61 -0
  37. package/src/synth-bench/cli-llm.ts +140 -0
  38. package/src/synth-bench/corpus.ts +335 -5
  39. package/src/synth-bench/round-trip-score.ts +169 -0
  40. package/src/synth-bench/run-bench-cli.ts +13 -115
  41. package/src/synth-bench/run-repair-bench-cli.ts +119 -0
  42. package/src/synth-bench/run-repair-bench.ts +266 -0
  43. package/src/synthesize-contract.ts +299 -37
@@ -0,0 +1,119 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Repair bench CLI — live round-trip-quality probe.
4
+ *
5
+ * Runs the production forgiving-handshake create-path
6
+ * (`ensureConformingContract`) over REPAIR_CORPUS and reports the
7
+ * round-trip-usable rate — the contract-quality number the shape bench
8
+ * (bench-synth) cannot see. Reads ~/.ggui/credentials.json (or
9
+ * ANTHROPIC_API_KEY) for the key.
10
+ *
11
+ * Costs ~$0.001 per repaired entry on Haiku 4.5 (clean drafts hit the
12
+ * lint-clean fast path and make NO LLM call). NOT run in CI; opt-in.
13
+ *
14
+ * Usage:
15
+ * pnpm -F @ggui-ai/negotiator bench-repair
16
+ * pnpm -F @ggui-ai/negotiator bench-repair -- --limit 1
17
+ * pnpm -F @ggui-ai/negotiator bench-repair -- --json > repair-report.json
18
+ * ANTHROPIC_API_KEY=sk-... pnpm -F @ggui-ai/negotiator bench-repair
19
+ *
20
+ * Bench-only — not exported from the package index.
21
+ */
22
+ import {
23
+ evaluateRepairCorpus,
24
+ formatRepairBenchReport,
25
+ } from './run-repair-bench.js';
26
+ import {
27
+ DEFAULT_MODEL,
28
+ HAIKU_4_5_PRICE_INPUT_PER_TOKEN,
29
+ HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN,
30
+ buildAnthropicLlmCaller,
31
+ getTokenUsage,
32
+ resolveAnthropicKey,
33
+ } from './cli-llm.js';
34
+
35
+ interface CliArgs {
36
+ limit?: number;
37
+ model: string;
38
+ json: boolean;
39
+ }
40
+
41
+ function parseArgs(argv: readonly string[]): CliArgs {
42
+ let limit: number | undefined;
43
+ let model = DEFAULT_MODEL;
44
+ let json = false;
45
+ for (let i = 0; i < argv.length; i++) {
46
+ const a = argv[i];
47
+ if (a === '--limit' && argv[i + 1]) {
48
+ limit = Number(argv[++i]);
49
+ } else if (a === '--model' && argv[i + 1]) {
50
+ model = argv[++i]!;
51
+ } else if (a === '--json') {
52
+ json = true;
53
+ }
54
+ }
55
+ const result: CliArgs = { model, json };
56
+ if (limit !== undefined) result.limit = limit;
57
+ return result;
58
+ }
59
+
60
+ async function main(): Promise<void> {
61
+ const args = parseArgs(process.argv.slice(2));
62
+ const apiKey = resolveAnthropicKey('bench-repair');
63
+ const llm = buildAnthropicLlmCaller(apiKey, args.model);
64
+
65
+ if (!args.json) {
66
+ process.stdout.write(`bench-repair: model=${args.model}\n\n`);
67
+ }
68
+ const report = await evaluateRepairCorpus(
69
+ { llm },
70
+ {
71
+ ...(args.limit !== undefined ? { limit: args.limit } : {}),
72
+ onProgress: args.json
73
+ ? undefined
74
+ : (outcome, idx, total) => {
75
+ const pass =
76
+ outcome.roundTrip !== null
77
+ ? outcome.roundTrip.pass
78
+ : outcome.shape.pass;
79
+ const status = pass ? 'OK ' : 'NO ';
80
+ const id = outcome.entry.id.padEnd(22);
81
+ const origin = `origin:${outcome.origin}`.padEnd(13);
82
+ process.stdout.write(
83
+ `[${status}] ${(idx + 1).toString().padStart(2)}/${total} ${origin} ${id} ${outcome.latencyMs}ms\n`,
84
+ );
85
+ },
86
+ },
87
+ );
88
+
89
+ if (args.json) {
90
+ process.stdout.write(JSON.stringify(report, null, 2));
91
+ process.stdout.write('\n');
92
+ return;
93
+ }
94
+
95
+ process.stdout.write('\n');
96
+ process.stdout.write(formatRepairBenchReport(report));
97
+ process.stdout.write('\n\n');
98
+
99
+ const usage = getTokenUsage();
100
+ const totalCost =
101
+ usage.input * HAIKU_4_5_PRICE_INPUT_PER_TOKEN +
102
+ usage.output * HAIKU_4_5_PRICE_OUTPUT_PER_TOKEN;
103
+ // Clean drafts (origin agent) make no LLM call — cost is per repaired entry.
104
+ const callsMade = report.totals.originSynth;
105
+ const costPerCall = callsMade === 0 ? 0 : totalCost / callsMade;
106
+ process.stdout.write(
107
+ `Tokens: input=${usage.input} output=${usage.output}\n`,
108
+ );
109
+ process.stdout.write(
110
+ `Cost: total=$${totalCost.toFixed(4)} per-repair=$${costPerCall.toFixed(4)}\n`,
111
+ );
112
+ }
113
+
114
+ main().catch((err) => {
115
+ process.stderr.write(
116
+ `bench-repair failed: ${err instanceof Error ? err.message : String(err)}\n`,
117
+ );
118
+ process.exit(1);
119
+ });
@@ -0,0 +1,266 @@
1
+ /**
2
+ * Repair-path bench runner — the round-trip QUALITY probe.
3
+ *
4
+ * Where {@link evaluateAgainstCorpus} runs synthesize-from-intent and
5
+ * scores SHAPE, this runs the production forgiving-handshake create-path
6
+ * — `ensureConformingContract(draft, intent)` — over {@link
7
+ * REPAIR_CORPUS} and scores whether the produced contract is round-trip
8
+ * USABLE, not merely valid.
9
+ *
10
+ * Per entry it records:
11
+ * - `origin` — `agent` (draft was clean, returned verbatim via the
12
+ * fast path) vs `synth` (repaired in-place). The repair discriminator.
13
+ * - `shape` — {@link scoreSynthesizedContract} ride-along (which specs).
14
+ * - `roundTrip` — {@link scoreContractRoundTrip}: the headline signal.
15
+ * A reshape that breaks the agent's seed round-trip fails here even
16
+ * though `lintContract` passes the contract.
17
+ *
18
+ * The bench's pass verdict is the ROUND-TRIP score when an entry carries
19
+ * a round-trip expectation, falling back to shape otherwise. So the
20
+ * top-line precision answers "how often does the negotiator produce a
21
+ * round-trip-usable contract from an agent draft?" — the contract-
22
+ * quality number the shape bench cannot see.
23
+ *
24
+ * Live LLM probe — opt-in CLI (run-repair-bench-cli.ts), NOT in CI. The
25
+ * deterministic scorer pinning lives in round-trip-score.test.ts.
26
+ */
27
+
28
+ import type { DataContract, SuggestionFinding } from '@ggui-ai/protocol';
29
+ import { ensureConformingContract } from '../ensure-conforming-contract.js';
30
+ import type { LLMCaller } from '../llm-caller.js';
31
+ import { scoreSynthesizedContract, type ScoreResult } from './run-bench.js';
32
+ import {
33
+ scoreContractRoundTrip,
34
+ type RoundTripScore,
35
+ } from './round-trip-score.js';
36
+ import { REPAIR_CORPUS, type BenchEntry } from './corpus.js';
37
+
38
+ export interface RepairBenchOutcome {
39
+ readonly entry: BenchEntry;
40
+ /** ensureConformingContract always returns a contract (possibly `{}`). */
41
+ readonly contract: DataContract;
42
+ /** `agent` = clean draft returned verbatim; `synth` = repaired in-place. */
43
+ readonly origin: 'agent' | 'synth';
44
+ /** How the contract was produced (the efficiency tier): verbatim /
45
+ * normalized (deterministic, no LLM) / llm-repair / fallback-empty. */
46
+ readonly method: 'verbatim' | 'normalized' | 'llm-repair' | 'fallback-empty';
47
+ /** Structural shape score (ride-along secondary signal). */
48
+ readonly shape: ScoreResult;
49
+ /** Round-trip usability — null when the entry declares no round-trip
50
+ * expectation (then `shape` carries the verdict). */
51
+ readonly roundTrip: RoundTripScore | null;
52
+ /** The error/warn findings the negotiator surfaced back to the agent. */
53
+ readonly findings: readonly SuggestionFinding[];
54
+ readonly reasoning: string;
55
+ readonly latencyMs: number;
56
+ }
57
+
58
+ /**
59
+ * An outcome passes on its ROUND-TRIP score when one exists (the sharper
60
+ * gate), else on its shape score. Round-trip is the point of this bench.
61
+ */
62
+ export function repairOutcomePass(outcome: RepairBenchOutcome): boolean {
63
+ return outcome.roundTrip !== null
64
+ ? outcome.roundTrip.pass
65
+ : outcome.shape.pass;
66
+ }
67
+
68
+ export interface RepairBenchReport {
69
+ readonly outcomes: readonly RepairBenchOutcome[];
70
+ readonly totals: {
71
+ readonly all: number;
72
+ readonly pass: number;
73
+ readonly fail: number;
74
+ /** Drafts returned verbatim by the lint-clean fast path. */
75
+ readonly originAgent: number;
76
+ /** Drafts repaired in-place by the synth loop. */
77
+ readonly originSynth: number;
78
+ /** Entries carrying a round-trip expectation. */
79
+ readonly roundTripScored: number;
80
+ /** Of those, how many round-trip cleanly. */
81
+ readonly roundTripPass: number;
82
+ /** roundTripPass / roundTripScored — the contract-quality headline. */
83
+ readonly roundTripPrecision: number;
84
+ /** pass / all across all entries. */
85
+ readonly precision: number;
86
+ };
87
+ /** Histogram of round-trip failure kinds across the run. */
88
+ readonly byFailureKind: Readonly<Record<string, number>>;
89
+ readonly latency: { readonly p50Ms: number; readonly p95Ms: number };
90
+ }
91
+
92
+ export interface RunRepairBenchOptions {
93
+ readonly limit?: number;
94
+ readonly onProgress?: (
95
+ outcome: RepairBenchOutcome,
96
+ index: number,
97
+ total: number,
98
+ ) => void;
99
+ }
100
+
101
+ export async function evaluateRepairCorpus(
102
+ deps: { readonly llm: LLMCaller },
103
+ options: RunRepairBenchOptions = {},
104
+ corpus: readonly BenchEntry[] = REPAIR_CORPUS,
105
+ ): Promise<RepairBenchReport> {
106
+ let subset: readonly BenchEntry[] = corpus;
107
+ if (options.limit !== undefined) {
108
+ subset = subset.slice(0, options.limit);
109
+ }
110
+
111
+ const outcomes: RepairBenchOutcome[] = [];
112
+ for (let i = 0; i < subset.length; i++) {
113
+ const entry = subset[i]!;
114
+ const startedAt = Date.now();
115
+ // The real production create-path: lint the draft → verbatim if clean
116
+ // (origin agent), repair-in-place otherwise (origin synth). NEVER
117
+ // throws; an unrepairable draft yields the empty `{}` contract.
118
+ const result = await ensureConformingContract(
119
+ { llm: deps.llm },
120
+ {
121
+ draft: entry.draft,
122
+ intent: entry.intent,
123
+ ...(entry.appGadgets !== undefined
124
+ ? { appGadgets: entry.appGadgets }
125
+ : {}),
126
+ },
127
+ );
128
+ const latencyMs = Date.now() - startedAt;
129
+ const shape = scoreSynthesizedContract(result.contract, entry.expected);
130
+ const roundTrip =
131
+ entry.roundTrip !== undefined
132
+ ? scoreContractRoundTrip(result.contract, entry.roundTrip)
133
+ : null;
134
+ const outcome: RepairBenchOutcome = {
135
+ entry,
136
+ contract: result.contract,
137
+ origin: result.origin,
138
+ method: result.method,
139
+ shape,
140
+ roundTrip,
141
+ findings: result.findings,
142
+ reasoning: result.reasoning,
143
+ latencyMs,
144
+ };
145
+ outcomes.push(outcome);
146
+ options.onProgress?.(outcome, i, subset.length);
147
+ }
148
+
149
+ return summarizeRepair(outcomes);
150
+ }
151
+
152
+ export function summarizeRepair(
153
+ outcomes: readonly RepairBenchOutcome[],
154
+ ): RepairBenchReport {
155
+ const all = outcomes.length;
156
+ const pass = outcomes.filter(repairOutcomePass).length;
157
+ const fail = all - pass;
158
+ const originAgent = outcomes.filter((o) => o.origin === 'agent').length;
159
+ const originSynth = outcomes.filter((o) => o.origin === 'synth').length;
160
+
161
+ const scored = outcomes.filter((o) => o.roundTrip !== null);
162
+ const roundTripScored = scored.length;
163
+ const roundTripPass = scored.filter((o) => o.roundTrip?.pass === true).length;
164
+ const roundTripPrecision =
165
+ roundTripScored === 0 ? 0 : roundTripPass / roundTripScored;
166
+
167
+ const byFailureKind: Record<string, number> = {};
168
+ for (const o of outcomes) {
169
+ for (const f of o.roundTrip?.failures ?? []) {
170
+ byFailureKind[f.kind] = (byFailureKind[f.kind] ?? 0) + 1;
171
+ }
172
+ }
173
+
174
+ const latencies = outcomes.map((o) => o.latencyMs).sort((a, b) => a - b);
175
+ return {
176
+ outcomes,
177
+ totals: {
178
+ all,
179
+ pass,
180
+ fail,
181
+ originAgent,
182
+ originSynth,
183
+ roundTripScored,
184
+ roundTripPass,
185
+ roundTripPrecision,
186
+ precision: all === 0 ? 0 : pass / all,
187
+ },
188
+ byFailureKind,
189
+ latency: {
190
+ p50Ms: percentile(latencies, 0.5),
191
+ p95Ms: percentile(latencies, 0.95),
192
+ },
193
+ };
194
+ }
195
+
196
+ function percentile(sorted: readonly number[], p: number): number {
197
+ if (sorted.length === 0) return 0;
198
+ const idx = Math.min(sorted.length - 1, Math.floor(p * sorted.length));
199
+ return sorted[idx] ?? 0;
200
+ }
201
+
202
+ export function formatRepairBenchReport(report: RepairBenchReport): string {
203
+ const lines: string[] = [];
204
+ const t = report.totals;
205
+ lines.push('=== repair bench report (round-trip quality) ===');
206
+ lines.push('');
207
+ lines.push(
208
+ `Round-trip usable: ${t.roundTripPass}/${t.roundTripScored} (${(t.roundTripPrecision * 100).toFixed(1)}%)`,
209
+ );
210
+ lines.push(`Repair origin: agent×${t.originAgent} synth×${t.originSynth}`);
211
+ // Efficiency tiers — verbatim + normalized are FREE (no LLM); only
212
+ // llm-repair pays a model call. A high normalized count = the cheap
213
+ // deterministic tier doing the work the LLM loop used to.
214
+ const byMethod = new Map<string, number>();
215
+ for (const o of report.outcomes) {
216
+ byMethod.set(o.method, (byMethod.get(o.method) ?? 0) + 1);
217
+ }
218
+ const methodStr = ['verbatim', 'normalized', 'llm-repair', 'fallback-empty']
219
+ .filter((m) => byMethod.has(m))
220
+ .map((m) => `${m}×${byMethod.get(m)}`)
221
+ .join(' ');
222
+ lines.push(`Method (LLM cost): ${methodStr}`);
223
+ lines.push(
224
+ `Overall pass: ${t.pass}/${t.all} (${(t.precision * 100).toFixed(1)}%)`,
225
+ );
226
+ lines.push(
227
+ `Latency: p50=${report.latency.p50Ms}ms p95=${report.latency.p95Ms}ms`,
228
+ );
229
+
230
+ const kinds = Object.entries(report.byFailureKind);
231
+ if (kinds.length > 0) {
232
+ lines.push('');
233
+ lines.push('Round-trip failures by kind:');
234
+ for (const [kind, count] of kinds.sort((a, b) => b[1] - a[1])) {
235
+ lines.push(` ${kind.padEnd(20)} ×${count}`);
236
+ }
237
+ }
238
+
239
+ const failed = report.outcomes.filter((o) => !repairOutcomePass(o));
240
+ if (failed.length > 0) {
241
+ lines.push('');
242
+ lines.push('Failures:');
243
+ for (const o of failed) {
244
+ lines.push(
245
+ ` [origin ${o.origin}] ${o.entry.id}: ${o.entry.intent.slice(0, 56)}`,
246
+ );
247
+ // Shape mismatches (gating only — advisory name checks omitted here).
248
+ for (const f of o.shape.failures) {
249
+ lines.push(` shape ${f.kind}: ${f.hint.slice(0, 120)}`);
250
+ }
251
+ for (const f of o.roundTrip?.failures ?? []) {
252
+ lines.push(` round-trip ${f.kind}: ${f.hint.slice(0, 160)}`);
253
+ }
254
+ lines.push(` reasoning: ${o.reasoning.slice(0, 120)}`);
255
+ }
256
+ }
257
+
258
+ return lines.join('\n');
259
+ }
260
+
261
+ export function runRepairBench(
262
+ deps: { readonly llm: LLMCaller },
263
+ options: RunRepairBenchOptions = {},
264
+ ): Promise<RepairBenchReport> {
265
+ return evaluateRepairCorpus(deps, options);
266
+ }