npm - @tangle-network/agent-runtime - Versions diffs - 0.37.0 → 0.38.0 - Mend

@tangle-network/agent-runtime 0.37.0 → 0.38.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (29) hide show

package/dist/agent.d.ts +3 -3
package/dist/analyst-loop.d.ts +2 -2
package/dist/analyst-loop.js +3 -257
package/dist/analyst-loop.js.map +1 -1
package/dist/chunk-VOX6Z3II.js +90 -0
package/dist/chunk-VOX6Z3II.js.map +1 -0
package/dist/chunk-XBUG326M.js +261 -0
package/dist/chunk-XBUG326M.js.map +1 -0
package/dist/{chunk-T3GJBKHA.js → chunk-Z523NPJK.js} +58 -1
package/dist/chunk-Z523NPJK.js.map +1 -0
package/dist/dynamic-DeOPeeAw.d.ts +106 -0
package/dist/{improvement-adapter-CaZxFxTd.d.ts → improvement-adapter-BC4HhuAR.d.ts} +1 -1
package/dist/improvement.d.ts +6 -130
package/dist/improvement.js +4 -85
package/dist/improvement.js.map +1 -1
package/dist/index.d.ts +67 -5
package/dist/index.js +61 -2
package/dist/index.js.map +1 -1
package/dist/loops.d.ts +5 -106
package/dist/mcp/index.d.ts +4 -79
package/dist/mcp/index.js +2 -57
package/dist/mcp/index.js.map +1 -1
package/dist/optimize-prompt-cmH9wZdH.d.ts +129 -0
package/dist/{otel-export-DgFMwsVy.d.ts → otel-export-CNmeg_7B.d.ts} +77 -2
package/dist/profiles.d.ts +1 -1
package/dist/{types-CmTjKLyB.d.ts → types-CmkQl8qE.d.ts} +1 -1
package/dist/{types-D_MXrmJP.d.ts → types-p8dWBIXL.d.ts} +1 -1
package/package.json +1 -1
package/dist/chunk-T3GJBKHA.js.map +0 -1

package/dist/improvement.d.ts CHANGED Viewed

@@ -1,8 +1,9 @@
-import { AnalystFinding, LlmClientOptions } from '@tangle-network/agent-eval';
+import { AnalystFinding } from '@tangle-network/agent-eval';
 import { L as LocalHarness, r as runLocalHarness } from './local-harness-KrdFTY5R.js';
-import { LabeledScenarioStore, WorktreeAdapter, ImprovementDriver, Scenario, DispatchContext, JudgeConfig, Gate, CampaignStorage, GateResult, RunImprovementLoopResult } from '@tangle-network/agent-eval/campaign';
-import { S as SurfaceImprovementEdit } from './improvement-adapter-CaZxFxTd.js';
-import { I as ImprovementAdapter } from './types-D_MXrmJP.js';
+import { LabeledScenarioStore, WorktreeAdapter, ImprovementDriver } from '@tangle-network/agent-eval/campaign';
+export { O as OptimizePromptOptions, a as OptimizePromptReflection, b as OptimizePromptResult, o as optimizePrompt } from './optimize-prompt-cmH9wZdH.js';
+import { S as SurfaceImprovementEdit } from './improvement-adapter-BC4HhuAR.js';
+import { I as ImprovementAdapter } from './types-p8dWBIXL.js';
 import 'node:child_process';
 /**
@@ -98,131 +99,6 @@ interface AgenticGeneratorOptions {
 }
 declare function agenticGenerator(opts?: AgenticGeneratorOptions): CandidateGenerator;
-/**
- * @experimental
- *
- * `optimizePrompt` — identity-gated optimization for any TEXT prompt surface
- * (system prompt, planner prompt, judge rubric, skill doc).
- *
- * The text-surface sibling to this module's `improvementDriver` (the
- * CODE-surface / worktree path). Both feed agent-eval's `runImprovementLoop`;
- * this one defaults the driver to agent-eval's `gepaDriver` (reflective text
- * mutator) and the gate to `heldOutGate`.
- *
- * IDENTITY-GATED BY CONSTRUCTION — the whole point. The loop runs evals,
- * collects per-scenario signal, proposes candidates, and the gate compares
- * candidate-vs-baseline ON THE HELDOUT. `result.prompt` is the baseline
- * (identity) UNLESS the gate decided `'ship'`. So wiring a surface up is safe:
- * a surface with no beneficial mutation simply keeps its baseline. You never
- * regress by registering a prompt — you only ever improve when the held-out
- * data earns it.
- *
- * Generic over the runtime: `runWithPrompt` is the only domain seam — given a
- * candidate prompt + scenario, run it however the surface runs (sandbox
- * `streamPrompt`, a `runLoop`, a direct model call) and return the artifact the
- * judges score. The optimizer never assumes how a prompt is executed.
- */
-/** Reflection config for the default `gepaDriver`. Omit when passing a custom
- *  `driver`. */
-interface OptimizePromptReflection {
-    /** Router transport for the reflection model. */
-    llm: LlmClientOptions;
-    /** Model that performs the reflective rewrite. */
-    model: string;
-    /** What is being optimized — orients the reflection prompt. Default
-     *  `'system prompt'`. */
-    target?: string;
-    /** Surface-specific mutation levers offered to the reflector. */
-    mutationPrimitives?: string[];
-    /** H2 (`## Foo`) headings that MUST survive every candidate. gepaDriver's
-     *  only structural guard — load-bearing sections of the prompt should be
-     *  `##` headings so a rewrite cannot drop them. */
-    preserveSections?: string[];
-    /** Max sentence-level edits per candidate vs the parent (a textual learning
-     *  rate). Caps a rewrite from wiping prior rules in one generation. */
-    maxSentenceEdits?: number;
-}
-/** @experimental */
-interface OptimizePromptOptions<TScenario extends Scenario, TArtifact> {
-    /** The prompt being optimized — the identity baseline the gate protects. */
-    baselinePrompt: string;
-    /** Domain seam: run a candidate prompt against a scenario → artifact the
-     *  judges score. The optimizer is agnostic to HOW the prompt runs. */
-    runWithPrompt: (prompt: string, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
-    /** Training pool — scored each generation to rank candidates. */
-    scenarios: TScenario[];
-    /** Held out of training — scored ONLY for the gate's baseline-vs-winner
-     *  delta. Disjoint from `scenarios`; this is what makes promotion measure
-     *  generalization, not memorization. */
-    holdoutScenarios: TScenario[];
-    /** Scorers — deterministic checks or LLM judges. */
-    judges: JudgeConfig<TArtifact, TScenario>[];
-    /** Where artifacts + traces land (opaque key under in-memory storage). */
-    runDir: string;
-    /** Default driver = `gepaDriver` built from this. Required UNLESS `driver`
-     *  is supplied. */
-    reflection?: OptimizePromptReflection;
-    /** Override the improvement strategy (custom driver / deterministic tests). */
-    driver?: ImprovementDriver;
-    /** Override the promotion gate. Default `heldOutGate` over `holdoutScenarios`
-     *  — zero extra LLM. Wrap `defaultProductionGate` for red-team/reward-hacking
-     *  hardening on production wiring. */
-    gate?: Gate<TArtifact, TScenario>;
-    /** Minimum held-out composite lift to ship, forwarded to the default
-     *  `heldOutGate`. When omitted the gate uses its own default. */
-    deltaThreshold?: number;
-    /** Candidates proposed per generation. Default 4. */
-    populationSize?: number;
-    /** Generations to run. Default 3. */
-    maxGenerations?: number;
-    /** Candidates carried to the next generation. Default 2. */
-    promoteTopK?: number;
-    /** Storage backend. Pass `inMemoryCampaignStorage()` for filesystem-less /
-     *  test runs. Default: Node filesystem. */
-    storage?: CampaignStorage;
-    /** Reproducibility seed. Default 42. */
-    seed?: number;
-    /** Per-scenario replicates for CI bands. Default 1. */
-    reps?: number;
-    /** Max concurrent cells. Default 2. */
-    maxConcurrency?: number;
-    /** Test seam — override the wall clock. */
-    now?: () => Date;
-    /** On a shipped gate: `'pr'` opens a PR, `'none'` just reports. Default
-     *  `'none'`. */
-    autoOnPromote?: 'pr' | 'none';
-    ghOwner?: string;
-    ghRepo?: string;
-}
-/** @experimental */
-interface OptimizePromptResult<TArtifact, TScenario extends Scenario> {
-    /** The prompt to USE. Identity (the baseline) unless the gate shipped a
-     *  winner — so a caller can always assign `result.prompt` unconditionally. */
-    prompt: string;
-    /** True only when the gate promoted a candidate over baseline on holdout. */
-    improved: boolean;
-    /** The gate's verdict (`'ship' | 'hold' | 'need_more_work' | ...`). */
-    decision: GateResult['decision'];
-    /** Human-readable reasons the gate gave. */
-    reasons: string[];
-    /** Mean held-out composite of the baseline. */
-    baselineComposite: number;
-    /** Mean held-out composite of the winner candidate. */
-    winnerComposite: number;
-    /** Held-out lift (winner − baseline); the gate's `delta` when it reported one. */
-    delta: number;
-    /** Why the winner was proposed — present when a shipped winner carried a
-     *  driver rationale. */
-    rationale?: string;
-    /** Unified baseline→winner diff (empty when the winner is the baseline). */
-    diff: string;
-    /** The full loop result for callers that need generations / campaigns. */
-    raw: RunImprovementLoopResult<TArtifact, TScenario>;
-}
-/** @experimental */
-declare function optimizePrompt<TScenario extends Scenario, TArtifact>(opts: OptimizePromptOptions<TScenario, TArtifact>): Promise<OptimizePromptResult<TArtifact, TScenario>>;
 /**
  * @experimental
  *
@@ -242,4 +118,4 @@ interface ReflectiveGeneratorOptions {
 }
 declare function reflectiveGenerator(opts: ReflectiveGeneratorOptions): CandidateGenerator;
-export { type AgenticGeneratorOptions, type CandidateGenerator, type ImprovementDriverOptions, type OptimizePromptOptions, type OptimizePromptReflection, type OptimizePromptResult, type ReflectiveGeneratorOptions, agenticGenerator, improvementDriver, optimizePrompt, reflectiveGenerator };
+export { type AgenticGeneratorOptions, type CandidateGenerator, type ImprovementDriverOptions, type ReflectiveGeneratorOptions, agenticGenerator, improvementDriver, reflectiveGenerator };

package/dist/improvement.js CHANGED Viewed

@@ -1,9 +1,10 @@
+import {
+  optimizePrompt
+} from "./chunk-VOX6Z3II.js";
 import {
   runLocalHarness
 } from "./chunk-GLR25NG7.js";
-import {
-  ConfigError
-} from "./chunk-SQSCRJ7U.js";
+import "./chunk-SQSCRJ7U.js";
 import "./chunk-DGUM43GV.js";
 // src/improvement/agentic-generator.ts
@@ -130,88 +131,6 @@ function resolveFindings(ctx) {
   return ctx.findings;
 }
-// src/improvement/optimize-prompt.ts
-import { gepaDriver, heldOutGate, runImprovementLoop } from "@tangle-network/agent-eval/campaign";
-async function optimizePrompt(opts) {
-  if (!opts.driver && !opts.reflection) {
-    throw new ConfigError(
-      "optimizePrompt: pass `reflection` (builds the default gepaDriver) or a custom `driver`"
-    );
-  }
-  if (opts.scenarios.length === 0) {
-    throw new ConfigError("optimizePrompt: `scenarios` must be non-empty");
-  }
-  if (opts.holdoutScenarios.length === 0) {
-    throw new ConfigError(
-      "optimizePrompt: `holdoutScenarios` must be non-empty (the gate needs it)"
-    );
-  }
-  const driver = opts.driver ?? gepaDriver({
-    llm: opts.reflection.llm,
-    model: opts.reflection.model,
-    target: opts.reflection.target ?? "system prompt",
-    mutationPrimitives: opts.reflection.mutationPrimitives,
-    constraints: opts.reflection.preserveSections || opts.reflection.maxSentenceEdits !== void 0 ? {
-      preserveSections: opts.reflection.preserveSections,
-      maxSentenceEdits: opts.reflection.maxSentenceEdits
-    } : void 0
-  });
-  const gate = opts.gate ?? heldOutGate({
-    scenarios: opts.holdoutScenarios,
-    ...opts.deltaThreshold !== void 0 ? { deltaThreshold: opts.deltaThreshold } : {}
-  });
-  const result = await runImprovementLoop({
-    baselineSurface: opts.baselinePrompt,
-    dispatchWithSurface: (surface, scenario, ctx) => {
-      if (typeof surface !== "string") {
-        throw new ConfigError(
-          "optimizePrompt: received a CodeSurface \u2014 this entry point optimizes string prompts only"
-        );
-      }
-      return opts.runWithPrompt(surface, scenario, ctx);
-    },
-    driver,
-    populationSize: opts.populationSize ?? 4,
-    maxGenerations: opts.maxGenerations ?? 3,
-    ...opts.promoteTopK !== void 0 ? { promoteTopK: opts.promoteTopK } : {},
-    scenarios: opts.scenarios,
-    holdoutScenarios: opts.holdoutScenarios,
-    judges: opts.judges,
-    gate,
-    autoOnPromote: opts.autoOnPromote ?? "none",
-    ...opts.ghOwner !== void 0 ? { ghOwner: opts.ghOwner } : {},
-    ...opts.ghRepo !== void 0 ? { ghRepo: opts.ghRepo } : {},
-    runDir: opts.runDir,
-    ...opts.storage !== void 0 ? { storage: opts.storage } : {},
-    ...opts.seed !== void 0 ? { seed: opts.seed } : {},
-    ...opts.reps !== void 0 ? { reps: opts.reps } : {},
-    ...opts.maxConcurrency !== void 0 ? { maxConcurrency: opts.maxConcurrency } : {},
-    ...opts.now !== void 0 ? { now: opts.now } : {}
-  });
-  const improved = result.gateResult.decision === "ship";
-  const winnerSurface = typeof result.winnerSurface === "string" ? result.winnerSurface : opts.baselinePrompt;
-  const baselineComposite = meanComposite(result.baselineOnHoldout);
-  const winnerComposite = meanComposite(result.winnerOnHoldout);
-  return {
-    prompt: improved ? winnerSurface : opts.baselinePrompt,
-    improved,
-    decision: result.gateResult.decision,
-    reasons: result.gateResult.reasons,
-    baselineComposite,
-    winnerComposite,
-    delta: result.gateResult.delta ?? winnerComposite - baselineComposite,
-    ...improved && result.winnerRationale ? { rationale: result.winnerRationale } : {},
-    diff: result.promotedDiff,
-    raw: result
-  };
-}
-function meanComposite(campaign) {
-  const scenarios = Object.values(campaign.aggregates.byScenario);
-  if (scenarios.length === 0) return 0;
-  const sum = scenarios.reduce((acc, s) => acc + s.meanComposite, 0);
-  return sum / scenarios.length;
-}
 // src/improvement/reflective-generator.ts
 import { spawnSync as spawnSync2 } from "child_process";
 function reflectiveGenerator(opts) {

package/dist/improvement.js.map CHANGED Viewed

	@@ -1 +1 @@
1	- {"version":3,"sources":["../src/improvement/agentic-generator.ts","../src/improvement/improvement-driver.ts","../src/improvement/optimize-prompt.ts","../src/improvement/reflective-generator.ts"],"sourcesContent":["/*\n @experimental\n \n `agenticGenerator` — the full-agentic `CandidateGenerator`: the\n * `shots=N, sandbox=on` setting of the one `improvementDriver`. It runs a real\n * coding harness (claude / codex / opencode) inside the candidate worktree the\n * driver already created, letting the agent read the codebase + the research\n * report and make the change in place. The driver then commits the worktree\n * into a `CodeSurface`.\n \n Mechanism: identical to the proven Phase-2.8 in-process executor — spawn the\n * harness as a subprocess with `cwd` = the worktree, on the same filesystem,\n * so edits land in place (no sandbox-mount round-trip). `runLocalHarness` is\n * the verified primitive. The OUTER sandbox is the improvement loop's own\n * execution context; the generator does not nest a second sandbox per\n * candidate (which would reintroduce a host↔sandbox worktree-transport\n * problem that does not need solving here).\n \n `maxShots` is the DEPTH dial: the harness runs once; if it produced no change\n * (the worktree stays clean), the generator refines the prompt and retries, up\n * to `maxShots` times. A harness that already changed files returns on shot 1.\n /\n\nimport { spawnSync } from 'node:child_process'\nimport type { AnalystFinding } from '@tangle-network/agent-eval'\nimport { type LocalHarness, runLocalHarness } from '../mcp/local-harness'\nimport type { CandidateGenerator } from './improvement-driver'\n\nexport interface AgenticGeneratorOptions {\n /* Local coding harness to run in the worktree. Default `claude`. /\n harness?: LocalHarness\n /* Per-shot wall-clock timeout (ms). Default = `runLocalHarness` default (5m). /\n timeoutMs?: number\n /* Build the harness task prompt from the report + findings. Override for\n * domain phrasing; the default turns findings into a concrete coder task. /\n buildPrompt?: (args: { report: unknown; findings: AnalystFinding[] }) => string\n /* Test seam — inject the harness runner (defaults to `runLocalHarness`). /\n runHarness?: typeof runLocalHarness\n /* Test seam — inject the worktree-dirty check (defaults to `git status`). /\n isDirty?: (worktreePath: string) => boolean\n}\n\nexport function agenticGenerator(opts: AgenticGeneratorOptions = {}): CandidateGenerator {\n const harness = opts.harness ?? 'claude'\n const buildPrompt = opts.buildPrompt ?? defaultBuildPrompt\n const run = opts.runHarness ?? runLocalHarness\n const dirty = opts.isDirty ?? worktreeDirty\n\n return {\n kind: `agentic:${harness}`,\n async generate({ worktreePath, report, findings, maxShots, signal }) {\n let prompt = buildPrompt({ report, findings })\n const shots = Math.max(1, maxShots)\n\n for (let shot = 0; shot < shots; shot++) {\n if (signal.aborted) break\n await run({\n harness,\n cwd: worktreePath,\n taskPrompt: prompt,\n timeoutMs: opts.timeoutMs,\n signal,\n })\n // The worktree IS the signal: if the harness touched files, we have a\n // candidate. We don't trust the harness's stdout — we trust the diff.\n if (dirty(worktreePath)) {\n return { applied: true, summary: summarize(findings) }\n }\n // No change this shot — give the next attempt explicit feedback.\n prompt = refine(prompt)\n }\n return { applied: false, summary: '' }\n },\n }\n}\n\n/* Turn the analyst's findings (+ optional report) into a concrete coder task. /\nfunction defaultBuildPrompt(args: { report: unknown; findings: AnalystFinding[] }): string {\n const lines: string[] = [\n 'You are improving this codebase based on an evaluation analysis.',\n 'Make the smallest set of edits that addresses the findings below, then stop.',\n 'Do not change unrelated code. Do not commit — leave changes in the working tree.',\n '',\n 'Findings:',\n ]\n for (const f of args.findings) {\n const where = f.subject ? ` [${f.subject}]` : ''\n lines.push(`- (${f.severity})${where} ${f.claim}`)\n if (f.recommended_action) lines.push(` → ${f.recommended_action}`)\n }\n return lines.join('\\n')\n}\n\nfunction refine(prompt: string): string {\n return `${prompt}\\n\\nNOTE: your previous attempt left the working tree unchanged. Make the concrete file edits now.`\n}\n\n/* A one-line summary for the commit message, derived from the findings. /\nfunction summarize(findings: AnalystFinding[]): string {\n if (findings.length === 0) return 'agentic improvement'\n if (findings.length === 1) return `agentic: ${truncate(findings[0]!.claim, 64)}`\n return `agentic: ${findings.length} findings addressed`\n}\n\nfunction truncate(s: string, n: number): string {\n return s.length <= n ? s : `${s.slice(0, n - 1)}…`\n}\n\n/* Non-empty `git status --porcelain` ⇒ the harness changed the worktree.\n * Fails loud: the worktree is a fresh checkout, so a git error here means\n * something is genuinely broken (git missing, corrupt index, killed mid-run).\n * Folding that into `false` would silently discard a candidate and mask the\n * real failure — forbidden by the no-silent-fallbacks doctrine. /\nfunction worktreeDirty(worktreePath: string): boolean {\n const result = spawnSync('git', ['status', '--porcelain'], {\n cwd: worktreePath,\n encoding: 'utf-8',\n })\n if (result.error) {\n throw new Error(\n `agenticGenerator: git status failed to spawn in ${worktreePath}: ${result.error.message}`,\n )\n }\n if (result.status !== 0) {\n throw new Error(\n `agenticGenerator: git status exited ${result.status} in ${worktreePath}: ${result.stderr.trim()}`,\n )\n }\n return result.stdout.trim().length > 0\n}\n","/\n @experimental\n \n `improvementDriver` — the ONE reflective/agentic improvement driver for\n * agent-eval's improvement loop. It implements `ImprovementDriver` and owns\n * the candidate lifecycle (worktree create → generate → finalize/discard,\n * × populationSize); it delegates the only thing that genuinely varies — HOW\n * a candidate change is produced — to a pluggable `CandidateGenerator`.\n \n There is no separate \"analyst driver\" vs \"autoresearch driver\": those are\n * the SAME driver at two settings of a dial.\n * - cheap reflective path → `reflectiveGenerator` (shots=1, no sandbox;\n * applies pre-drafted patches)\n * - full agentic path → `agenticGenerator` (shots=N, sandbox runLoop;\n * an agent reads code + report and edits)\n * Both emit changes into a worktree the driver finalizes into a\n * `CodeSurface{ worktreeRef }` the loop measures on the holdout. See\n * agent-eval's `docs/design/self-improvement-engine.md`.\n /\n\nimport type { AnalystFinding } from '@tangle-network/agent-eval'\nimport type {\n CodeSurface,\n ImprovementDriver,\n LabeledScenarioStore,\n ProposeContext,\n WorktreeAdapter,\n} from '@tangle-network/agent-eval/campaign'\n\n/* The byte-producing seam — the ONE thing that differs between the cheap\n * reflective path and the full agentic path. A generator makes (uncommitted)\n * changes inside `worktreePath`; the driver commits them via the worktree\n * adapter's `finalize`. /\nexport interface CandidateGenerator {\n kind: string\n generate(args: {\n /* The candidate worktree — a fresh checkout of baseRef. Write changes here. /\n worktreePath: string\n /* Phase-2 research report (analyst findings + diff), opaque. /\n report: unknown\n /* Findings resolved from the report or the loop context. /\n findings: AnalystFinding[]\n /* Handle to all captured data, to ground the change. /\n dataset?: LabeledScenarioStore\n /* DEPTH: max iterations the generator may take (agentic uses this; the\n * reflective generator ignores it). /\n maxShots: number\n signal: AbortSignal\n }): Promise<{ applied: boolean; summary: string }>\n}\n\nexport interface ImprovementDriverOptions {\n worktree: WorktreeAdapter\n generator: CandidateGenerator\n /* Base ref candidate worktrees fork from. Default `main`. /\n baseRef?: string\n}\n\nexport function improvementDriver(\n opts: ImprovementDriverOptions,\n): ImprovementDriver<AnalystFinding> {\n const baseRef = opts.baseRef ?? 'main'\n\n return {\n kind: `improvement:${opts.generator.kind}`,\n async propose(ctx) {\n const findings = resolveFindings(ctx)\n // No signal to act on — propose nothing rather than spin up worktrees.\n if (findings.length === 0 && ctx.report === undefined) return []\n\n const surfaces: CodeSurface[] = []\n for (let i = 0; i < ctx.populationSize; i++) {\n if (ctx.signal.aborted) break\n const wt = await opts.worktree.create({\n baseRef,\n label: `${opts.generator.kind}-gen${ctx.generation}-cand${i}`,\n })\n // Once a worktree exists it MUST be accounted for: finalized into a\n // surface, or discarded. A throw from generate()/finalize() must not\n // leak the worktree + branch — discard best-effort, then rethrow loud.\n try {\n const { applied, summary } = await opts.generator.generate({\n worktreePath: wt.path,\n report: ctx.report,\n findings,\n dataset: ctx.dataset,\n maxShots: ctx.maxImprovementShots ?? 1,\n signal: ctx.signal,\n })\n if (!applied) {\n await opts.worktree.discard(wt)\n continue\n }\n surfaces.push(await opts.worktree.finalize(wt, summary))\n } catch (err) {\n // Best-effort cleanup; never mask the original failure.\n await opts.worktree.discard(wt).catch(() => {})\n throw err\n }\n }\n return surfaces\n },\n }\n}\n\n/* Phase-2 report carries `findings` when present; else fall back to the\n * loop's `ctx.findings`. The report is opaque to the substrate, so probe it\n * structurally. /\nfunction resolveFindings(ctx: ProposeContext<AnalystFinding>): AnalystFinding[] {\n const report = ctx.report\n if (report && typeof report === 'object' && 'findings' in report) {\n const f = (report as { findings: unknown }).findings\n if (Array.isArray(f) && f.length > 0) return f as AnalystFinding[]\n }\n return ctx.findings\n}\n","/\n @experimental\n \n `optimizePrompt` — identity-gated optimization for any TEXT prompt surface\n * (system prompt, planner prompt, judge rubric, skill doc).\n \n The text-surface sibling to this module's `improvementDriver` (the\n * CODE-surface / worktree path). Both feed agent-eval's `runImprovementLoop`;\n * this one defaults the driver to agent-eval's `gepaDriver` (reflective text\n * mutator) and the gate to `heldOutGate`.\n \n IDENTITY-GATED BY CONSTRUCTION — the whole point. The loop runs evals,\n * collects per-scenario signal, proposes candidates, and the gate compares\n * candidate-vs-baseline ON THE HELDOUT. `result.prompt` is the baseline\n * (identity) UNLESS the gate decided `'ship'`. So wiring a surface up is safe:\n * a surface with no beneficial mutation simply keeps its baseline. You never\n * regress by registering a prompt — you only ever improve when the held-out\n * data earns it.\n \n Generic over the runtime: `runWithPrompt` is the only domain seam — given a\n * candidate prompt + scenario, run it however the surface runs (sandbox\n * `streamPrompt`, a `runLoop`, a direct model call) and return the artifact the\n * judges score. The optimizer never assumes how a prompt is executed.\n /\n\nimport type { LlmClientOptions } from '@tangle-network/agent-eval'\nimport type {\n CampaignResult,\n CampaignStorage,\n DispatchContext,\n Gate,\n GateResult,\n ImprovementDriver,\n JudgeConfig,\n RunImprovementLoopResult,\n Scenario,\n} from '@tangle-network/agent-eval/campaign'\nimport { gepaDriver, heldOutGate, runImprovementLoop } from '@tangle-network/agent-eval/campaign'\nimport { ConfigError } from '../errors'\n\n/* Reflection config for the default `gepaDriver`. Omit when passing a custom\n * `driver`. /\nexport interface OptimizePromptReflection {\n /* Router transport for the reflection model. /\n llm: LlmClientOptions\n /* Model that performs the reflective rewrite. /\n model: string\n /* What is being optimized — orients the reflection prompt. Default\n * `'system prompt'`. /\n target?: string\n /* Surface-specific mutation levers offered to the reflector. /\n mutationPrimitives?: string[]\n /* H2 (`## Foo`) headings that MUST survive every candidate. gepaDriver's\n * only structural guard — load-bearing sections of the prompt should be\n * `##` headings so a rewrite cannot drop them. /\n preserveSections?: string[]\n /* Max sentence-level edits per candidate vs the parent (a textual learning\n * rate). Caps a rewrite from wiping prior rules in one generation. /\n maxSentenceEdits?: number\n}\n\n/* @experimental /\nexport interface OptimizePromptOptions<TScenario extends Scenario, TArtifact> {\n /* The prompt being optimized — the identity baseline the gate protects. /\n baselinePrompt: string\n /* Domain seam: run a candidate prompt against a scenario → artifact the\n * judges score. The optimizer is agnostic to HOW the prompt runs. /\n runWithPrompt: (prompt: string, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>\n /* Training pool — scored each generation to rank candidates. /\n scenarios: TScenario[]\n /* Held out of training — scored ONLY for the gate's baseline-vs-winner\n * delta. Disjoint from `scenarios`; this is what makes promotion measure\n * generalization, not memorization. /\n holdoutScenarios: TScenario[]\n /* Scorers — deterministic checks or LLM judges. /\n judges: JudgeConfig<TArtifact, TScenario>[]\n /* Where artifacts + traces land (opaque key under in-memory storage). /\n runDir: string\n /* Default driver = `gepaDriver` built from this. Required UNLESS `driver`\n * is supplied. /\n reflection?: OptimizePromptReflection\n /* Override the improvement strategy (custom driver / deterministic tests). /\n driver?: ImprovementDriver\n /* Override the promotion gate. Default `heldOutGate` over `holdoutScenarios`\n * — zero extra LLM. Wrap `defaultProductionGate` for red-team/reward-hacking\n * hardening on production wiring. /\n gate?: Gate<TArtifact, TScenario>\n /* Minimum held-out composite lift to ship, forwarded to the default\n * `heldOutGate`. When omitted the gate uses its own default. /\n deltaThreshold?: number\n /* Candidates proposed per generation. Default 4. /\n populationSize?: number\n /* Generations to run. Default 3. /\n maxGenerations?: number\n /* Candidates carried to the next generation. Default 2. /\n promoteTopK?: number\n /* Storage backend. Pass `inMemoryCampaignStorage()` for filesystem-less /\n * test runs. Default: Node filesystem. /\n storage?: CampaignStorage\n /* Reproducibility seed. Default 42. /\n seed?: number\n /* Per-scenario replicates for CI bands. Default 1. /\n reps?: number\n /* Max concurrent cells. Default 2. /\n maxConcurrency?: number\n /* Test seam — override the wall clock. /\n now?: () => Date\n /* On a shipped gate: `'pr'` opens a PR, `'none'` just reports. Default\n * `'none'`. /\n autoOnPromote?: 'pr' \| 'none'\n ghOwner?: string\n ghRepo?: string\n}\n\n/* @experimental /\nexport interface OptimizePromptResult<TArtifact, TScenario extends Scenario> {\n /* The prompt to USE. Identity (the baseline) unless the gate shipped a\n * winner — so a caller can always assign `result.prompt` unconditionally. /\n prompt: string\n /* True only when the gate promoted a candidate over baseline on holdout. /\n improved: boolean\n /* The gate's verdict (`'ship' \| 'hold' \| 'need_more_work' \| ...`). /\n decision: GateResult['decision']\n /* Human-readable reasons the gate gave. /\n reasons: string[]\n /* Mean held-out composite of the baseline. /\n baselineComposite: number\n /* Mean held-out composite of the winner candidate. /\n winnerComposite: number\n /* Held-out lift (winner − baseline); the gate's `delta` when it reported one. /\n delta: number\n /* Why the winner was proposed — present when a shipped winner carried a\n * driver rationale. /\n rationale?: string\n /* Unified baseline→winner diff (empty when the winner is the baseline). /\n diff: string\n /* The full loop result for callers that need generations / campaigns. /\n raw: RunImprovementLoopResult<TArtifact, TScenario>\n}\n\n/* @experimental /\nexport async function optimizePrompt<TScenario extends Scenario, TArtifact>(\n opts: OptimizePromptOptions<TScenario, TArtifact>,\n): Promise<OptimizePromptResult<TArtifact, TScenario>> {\n if (!opts.driver && !opts.reflection) {\n throw new ConfigError(\n 'optimizePrompt: pass `reflection` (builds the default gepaDriver) or a custom `driver`',\n )\n }\n if (opts.scenarios.length === 0) {\n throw new ConfigError('optimizePrompt: `scenarios` must be non-empty')\n }\n if (opts.holdoutScenarios.length === 0) {\n throw new ConfigError(\n 'optimizePrompt: `holdoutScenarios` must be non-empty (the gate needs it)',\n )\n }\n\n const driver =\n opts.driver ??\n gepaDriver({\n llm: opts.reflection!.llm,\n model: opts.reflection!.model,\n target: opts.reflection!.target ?? 'system prompt',\n mutationPrimitives: opts.reflection!.mutationPrimitives,\n constraints:\n opts.reflection!.preserveSections \|\| opts.reflection!.maxSentenceEdits !== undefined\n ? {\n preserveSections: opts.reflection!.preserveSections,\n maxSentenceEdits: opts.reflection!.maxSentenceEdits,\n }\n : undefined,\n })\n\n const gate =\n opts.gate ??\n heldOutGate<TArtifact, TScenario>({\n scenarios: opts.holdoutScenarios,\n ...(opts.deltaThreshold !== undefined ? { deltaThreshold: opts.deltaThreshold } : {}),\n })\n\n const result = await runImprovementLoop<TScenario, TArtifact>({\n baselineSurface: opts.baselinePrompt,\n dispatchWithSurface: (surface, scenario, ctx) => {\n if (typeof surface !== 'string') {\n // optimizePrompt is the TEXT-surface entry point; a CodeSurface means\n // the caller wired the wrong driver. Fail loud — don't silently run the\n // baseline and report a phantom score.\n throw new ConfigError(\n 'optimizePrompt: received a CodeSurface — this entry point optimizes string prompts only',\n )\n }\n return opts.runWithPrompt(surface, scenario, ctx)\n },\n driver,\n populationSize: opts.populationSize ?? 4,\n maxGenerations: opts.maxGenerations ?? 3,\n ...(opts.promoteTopK !== undefined ? { promoteTopK: opts.promoteTopK } : {}),\n scenarios: opts.scenarios,\n holdoutScenarios: opts.holdoutScenarios,\n judges: opts.judges,\n gate,\n autoOnPromote: opts.autoOnPromote ?? 'none',\n ...(opts.ghOwner !== undefined ? { ghOwner: opts.ghOwner } : {}),\n ...(opts.ghRepo !== undefined ? { ghRepo: opts.ghRepo } : {}),\n runDir: opts.runDir,\n ...(opts.storage !== undefined ? { storage: opts.storage } : {}),\n ...(opts.seed !== undefined ? { seed: opts.seed } : {}),\n ...(opts.reps !== undefined ? { reps: opts.reps } : {}),\n ...(opts.maxConcurrency !== undefined ? { maxConcurrency: opts.maxConcurrency } : {}),\n ...(opts.now !== undefined ? { now: opts.now } : {}),\n })\n\n const improved = result.gateResult.decision === 'ship'\n const winnerSurface =\n typeof result.winnerSurface === 'string' ? result.winnerSurface : opts.baselinePrompt\n const baselineComposite = meanComposite(result.baselineOnHoldout)\n const winnerComposite = meanComposite(result.winnerOnHoldout)\n\n return {\n prompt: improved ? winnerSurface : opts.baselinePrompt,\n improved,\n decision: result.gateResult.decision,\n reasons: result.gateResult.reasons,\n baselineComposite,\n winnerComposite,\n delta: result.gateResult.delta ?? winnerComposite - baselineComposite,\n ...(improved && result.winnerRationale ? { rationale: result.winnerRationale } : {}),\n diff: result.promotedDiff,\n raw: result,\n }\n}\n\n/* Mean composite over a campaign's per-scenario aggregates. The held-out\n * campaigns score one surface across `holdoutScenarios`; averaging the\n * per-scenario means gives the single number the gate's delta is built from. /\nfunction meanComposite(campaign: CampaignResult<unknown, Scenario>): number {\n const scenarios = Object.values(campaign.aggregates.byScenario)\n if (scenarios.length === 0) return 0\n const sum = scenarios.reduce((acc, s) => acc + s.meanComposite, 0)\n return sum / scenarios.length\n}\n","/\n @experimental\n \n `reflectiveGenerator` — the cheap, no-sandbox `CandidateGenerator`. It drafts\n * surface edits via the existing improvement adapter (`proposeFromFindings`,\n * one LLM patch per finding) and applies them as ONE coherent improvement into\n * the candidate worktree. `maxShots` is ignored — reflection is single-shot by\n * construction (the patches are already drafted).\n \n This is the `shots=1, sandbox=off` setting of the one improvement driver.\n * The `agenticGenerator` (sandbox runLoop) is the `shots=N, sandbox=on`\n * setting — both plug into the same `improvementDriver`.\n /\n\nimport { spawnSync } from 'node:child_process'\nimport type { SurfaceImprovementEdit } from '../agent/improvement-adapter'\nimport type { ImprovementAdapter } from '../analyst-loop/types'\nimport type { CandidateGenerator } from './improvement-driver'\n\nexport interface ReflectiveGeneratorOptions {\n improvementAdapter: ImprovementAdapter<SurfaceImprovementEdit>\n}\n\nexport function reflectiveGenerator(opts: ReflectiveGeneratorOptions): CandidateGenerator {\n return {\n kind: 'reflective',\n async generate({ worktreePath, findings }) {\n const batch = await opts.improvementAdapter.proposeFromFindings(findings)\n if (batch.edits.length === 0) return { applied: false, summary: '' }\n\n let applied = 0\n for (const edit of batch.edits) {\n if (applyPatch(edit.patch, worktreePath)) applied++\n }\n if (applied === 0) return { applied: false, summary: '' }\n\n const summary =\n batch.edits.length === 1\n ? batch.edits[0]!.summary\n : `analyst: ${applied} surface edit${applied === 1 ? '' : 's'}`\n return { applied: true, summary }\n },\n }\n}\n\n/* Mirror the improvement adapter's proven apply invocation, run inside the\n * candidate worktree (a fresh checkout of baseRef, so `-p0` paths match). */\nfunction applyPatch(patch: string, cwd: string): boolean {\n const result = spawnSync('git', ['apply', '--whitespace=fix', '-p0', '-'], {\n cwd,\n input: patch,\n encoding: 'utf-8',\n })\n return result.status === 0\n}\n"],"mappings":";;;;;;;;;AAuBA,SAAS,iBAAiB;AAmBnB,SAAS,iBAAiB,OAAgC,CAAC,GAAuB;AACvF,QAAM,UAAU,KAAK,WAAW;AAChC,QAAM,cAAc,KAAK,eAAe;AACxC,QAAM,MAAM,KAAK,cAAc;AAC/B,QAAM,QAAQ,KAAK,WAAW;AAE9B,SAAO;AAAA,IACL,MAAM,WAAW,OAAO;AAAA,IACxB,MAAM,SAAS,EAAE,cAAc,QAAQ,UAAU,UAAU,OAAO,GAAG;AACnE,UAAI,SAAS,YAAY,EAAE,QAAQ,SAAS,CAAC;AAC7C,YAAM,QAAQ,KAAK,IAAI,GAAG,QAAQ;AAElC,eAAS,OAAO,GAAG,OAAO,OAAO,QAAQ;AACvC,YAAI,OAAO,QAAS;AACpB,cAAM,IAAI;AAAA,UACR;AAAA,UACA,KAAK;AAAA,UACL,YAAY;AAAA,UACZ,WAAW,KAAK;AAAA,UAChB;AAAA,QACF,CAAC;AAGD,YAAI,MAAM,YAAY,GAAG;AACvB,iBAAO,EAAE,SAAS,MAAM,SAAS,UAAU,QAAQ,EAAE;AAAA,QACvD;AAEA,iBAAS,OAAO,MAAM;AAAA,MACxB;AACA,aAAO,EAAE,SAAS,OAAO,SAAS,GAAG;AAAA,IACvC;AAAA,EACF;AACF;AAGA,SAAS,mBAAmB,MAA+D;AACzF,QAAM,QAAkB;AAAA,IACtB;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,EACF;AACA,aAAW,KAAK,KAAK,UAAU;AAC7B,UAAM,QAAQ,EAAE,UAAU,KAAK,EAAE,OAAO,MAAM;AAC9C,UAAM,KAAK,MAAM,EAAE,QAAQ,IAAI,KAAK,IAAI,EAAE,KAAK,EAAE;AACjD,QAAI,EAAE,mBAAoB,OAAM,KAAK,cAAS,EAAE,kBAAkB,EAAE;AAAA,EACtE;AACA,SAAO,MAAM,KAAK,IAAI;AACxB;AAEA,SAAS,OAAO,QAAwB;AACtC,SAAO,GAAG,MAAM;AAAA;AAAA;AAClB;AAGA,SAAS,UAAU,UAAoC;AACrD,MAAI,SAAS,WAAW,EAAG,QAAO;AAClC,MAAI,SAAS,WAAW,EAAG,QAAO,YAAY,SAAS,SAAS,CAAC,EAAG,OAAO,EAAE,CAAC;AAC9E,SAAO,YAAY,SAAS,MAAM;AACpC;AAEA,SAAS,SAAS,GAAW,GAAmB;AAC9C,SAAO,EAAE,UAAU,IAAI,IAAI,GAAG,EAAE,MAAM,GAAG,IAAI,CAAC,CAAC;AACjD;AAOA,SAAS,cAAc,cAA+B;AACpD,QAAM,SAAS,UAAU,OAAO,CAAC,UAAU,aAAa,GAAG;AAAA,IACzD,KAAK;AAAA,IACL,UAAU;AAAA,EACZ,CAAC;AACD,MAAI,OAAO,OAAO;AAChB,UAAM,IAAI;AAAA,MACR,mDAAmD,YAAY,KAAK,OAAO,MAAM,OAAO;AAAA,IAC1F;AAAA,EACF;AACA,MAAI,OAAO,WAAW,GAAG;AACvB,UAAM,IAAI;AAAA,MACR,uCAAuC,OAAO,MAAM,OAAO,YAAY,KAAK,OAAO,OAAO,KAAK,CAAC;AAAA,IAClG;AAAA,EACF;AACA,SAAO,OAAO,OAAO,KAAK,EAAE,SAAS;AACvC;;;ACvEO,SAAS,kBACd,MACmC;AACnC,QAAM,UAAU,KAAK,WAAW;AAEhC,SAAO;AAAA,IACL,MAAM,eAAe,KAAK,UAAU,IAAI;AAAA,IACxC,MAAM,QAAQ,KAAK;AACjB,YAAM,WAAW,gBAAgB,GAAG;AAEpC,UAAI,SAAS,WAAW,KAAK,IAAI,WAAW,OAAW,QAAO,CAAC;AAE/D,YAAM,WAA0B,CAAC;AACjC,eAAS,IAAI,GAAG,IAAI,IAAI,gBAAgB,KAAK;AAC3C,YAAI,IAAI,OAAO,QAAS;AACxB,cAAM,KAAK,MAAM,KAAK,SAAS,OAAO;AAAA,UACpC;AAAA,UACA,OAAO,GAAG,KAAK,UAAU,IAAI,OAAO,IAAI,UAAU,QAAQ,CAAC;AAAA,QAC7D,CAAC;AAID,YAAI;AACF,gBAAM,EAAE,SAAS,QAAQ,IAAI,MAAM,KAAK,UAAU,SAAS;AAAA,YACzD,cAAc,GAAG;AAAA,YACjB,QAAQ,IAAI;AAAA,YACZ;AAAA,YACA,SAAS,IAAI;AAAA,YACb,UAAU,IAAI,uBAAuB;AAAA,YACrC,QAAQ,IAAI;AAAA,UACd,CAAC;AACD,cAAI,CAAC,SAAS;AACZ,kBAAM,KAAK,SAAS,QAAQ,EAAE;AAC9B;AAAA,UACF;AACA,mBAAS,KAAK,MAAM,KAAK,SAAS,SAAS,IAAI,OAAO,CAAC;AAAA,QACzD,SAAS,KAAK;AAEZ,gBAAM,KAAK,SAAS,QAAQ,EAAE,EAAE,MAAM,MAAM;AAAA,UAAC,CAAC;AAC9C,gBAAM;AAAA,QACR;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAKA,SAAS,gBAAgB,KAAuD;AAC9E,QAAM,SAAS,IAAI;AACnB,MAAI,UAAU,OAAO,WAAW,YAAY,cAAc,QAAQ;AAChE,UAAM,IAAK,OAAiC;AAC5C,QAAI,MAAM,QAAQ,CAAC,KAAK,EAAE,SAAS,EAAG,QAAO;AAAA,EAC/C;AACA,SAAO,IAAI;AACb;;;AC9EA,SAAS,YAAY,aAAa,0BAA0B;AAwG5D,eAAsB,eACpB,MACqD;AACrD,MAAI,CAAC,KAAK,UAAU,CAAC,KAAK,YAAY;AACpC,UAAM,IAAI;AAAA,MACR;AAAA,IACF;AAAA,EACF;AACA,MAAI,KAAK,UAAU,WAAW,GAAG;AAC/B,UAAM,IAAI,YAAY,+CAA+C;AAAA,EACvE;AACA,MAAI,KAAK,iBAAiB,WAAW,GAAG;AACtC,UAAM,IAAI;AAAA,MACR;AAAA,IACF;AAAA,EACF;AAEA,QAAM,SACJ,KAAK,UACL,WAAW;AAAA,IACT,KAAK,KAAK,WAAY;AAAA,IACtB,OAAO,KAAK,WAAY;AAAA,IACxB,QAAQ,KAAK,WAAY,UAAU;AAAA,IACnC,oBAAoB,KAAK,WAAY;AAAA,IACrC,aACE,KAAK,WAAY,oBAAoB,KAAK,WAAY,qBAAqB,SACvE;AAAA,MACE,kBAAkB,KAAK,WAAY;AAAA,MACnC,kBAAkB,KAAK,WAAY;AAAA,IACrC,IACA;AAAA,EACR,CAAC;AAEH,QAAM,OACJ,KAAK,QACL,YAAkC;AAAA,IAChC,WAAW,KAAK;AAAA,IAChB,GAAI,KAAK,mBAAmB,SAAY,EAAE,gBAAgB,KAAK,eAAe,IAAI,CAAC;AAAA,EACrF,CAAC;AAEH,QAAM,SAAS,MAAM,mBAAyC;AAAA,IAC5D,iBAAiB,KAAK;AAAA,IACtB,qBAAqB,CAAC,SAAS,UAAU,QAAQ;AAC/C,UAAI,OAAO,YAAY,UAAU;AAI/B,cAAM,IAAI;AAAA,UACR;AAAA,QACF;AAAA,MACF;AACA,aAAO,KAAK,cAAc,SAAS,UAAU,GAAG;AAAA,IAClD;AAAA,IACA;AAAA,IACA,gBAAgB,KAAK,kBAAkB;AAAA,IACvC,gBAAgB,KAAK,kBAAkB;AAAA,IACvC,GAAI,KAAK,gBAAgB,SAAY,EAAE,aAAa,KAAK,YAAY,IAAI,CAAC;AAAA,IAC1E,WAAW,KAAK;AAAA,IAChB,kBAAkB,KAAK;AAAA,IACvB,QAAQ,KAAK;AAAA,IACb;AAAA,IACA,eAAe,KAAK,iBAAiB;AAAA,IACrC,GAAI,KAAK,YAAY,SAAY,EAAE,SAAS,KAAK,QAAQ,IAAI,CAAC;AAAA,IAC9D,GAAI,KAAK,WAAW,SAAY,EAAE,QAAQ,KAAK,OAAO,IAAI,CAAC;AAAA,IAC3D,QAAQ,KAAK;AAAA,IACb,GAAI,KAAK,YAAY,SAAY,EAAE,SAAS,KAAK,QAAQ,IAAI,CAAC;AAAA,IAC9D,GAAI,KAAK,SAAS,SAAY,EAAE,MAAM,KAAK,KAAK,IAAI,CAAC;AAAA,IACrD,GAAI,KAAK,SAAS,SAAY,EAAE,MAAM,KAAK,KAAK,IAAI,CAAC;AAAA,IACrD,GAAI,KAAK,mBAAmB,SAAY,EAAE,gBAAgB,KAAK,eAAe,IAAI,CAAC;AAAA,IACnF,GAAI,KAAK,QAAQ,SAAY,EAAE,KAAK,KAAK,IAAI,IAAI,CAAC;AAAA,EACpD,CAAC;AAED,QAAM,WAAW,OAAO,WAAW,aAAa;AAChD,QAAM,gBACJ,OAAO,OAAO,kBAAkB,WAAW,OAAO,gBAAgB,KAAK;AACzE,QAAM,oBAAoB,cAAc,OAAO,iBAAiB;AAChE,QAAM,kBAAkB,cAAc,OAAO,eAAe;AAE5D,SAAO;AAAA,IACL,QAAQ,WAAW,gBAAgB,KAAK;AAAA,IACxC;AAAA,IACA,UAAU,OAAO,WAAW;AAAA,IAC5B,SAAS,OAAO,WAAW;AAAA,IAC3B;AAAA,IACA;AAAA,IACA,OAAO,OAAO,WAAW,SAAS,kBAAkB;AAAA,IACpD,GAAI,YAAY,OAAO,kBAAkB,EAAE,WAAW,OAAO,gBAAgB,IAAI,CAAC;AAAA,IAClF,MAAM,OAAO;AAAA,IACb,KAAK;AAAA,EACP;AACF;AAKA,SAAS,cAAc,UAAqD;AAC1E,QAAM,YAAY,OAAO,OAAO,SAAS,WAAW,UAAU;AAC9D,MAAI,UAAU,WAAW,EAAG,QAAO;AACnC,QAAM,MAAM,UAAU,OAAO,CAAC,KAAK,MAAM,MAAM,EAAE,eAAe,CAAC;AACjE,SAAO,MAAM,UAAU;AACzB;;;ACnOA,SAAS,aAAAA,kBAAiB;AASnB,SAAS,oBAAoB,MAAsD;AACxF,SAAO;AAAA,IACL,MAAM;AAAA,IACN,MAAM,SAAS,EAAE,cAAc,SAAS,GAAG;AACzC,YAAM,QAAQ,MAAM,KAAK,mBAAmB,oBAAoB,QAAQ;AACxE,UAAI,MAAM,MAAM,WAAW,EAAG,QAAO,EAAE,SAAS,OAAO,SAAS,GAAG;AAEnE,UAAI,UAAU;AACd,iBAAW,QAAQ,MAAM,OAAO;AAC9B,YAAI,WAAW,KAAK,OAAO,YAAY,EAAG;AAAA,MAC5C;AACA,UAAI,YAAY,EAAG,QAAO,EAAE,SAAS,OAAO,SAAS,GAAG;AAExD,YAAM,UACJ,MAAM,MAAM,WAAW,IACnB,MAAM,MAAM,CAAC,EAAG,UAChB,YAAY,OAAO,gBAAgB,YAAY,IAAI,KAAK,GAAG;AACjE,aAAO,EAAE,SAAS,MAAM,QAAQ;AAAA,IAClC;AAAA,EACF;AACF;AAIA,SAAS,WAAW,OAAe,KAAsB;AACvD,QAAM,SAASA,WAAU,OAAO,CAAC,SAAS,oBAAoB,OAAO,GAAG,GAAG;AAAA,IACzE;AAAA,IACA,OAAO;AAAA,IACP,UAAU;AAAA,EACZ,CAAC;AACD,SAAO,OAAO,WAAW;AAC3B;","names":["spawnSync"]}
1	+ {"version":3,"sources":["../src/improvement/agentic-generator.ts","../src/improvement/improvement-driver.ts","../src/improvement/reflective-generator.ts"],"sourcesContent":["/*\n @experimental\n \n `agenticGenerator` — the full-agentic `CandidateGenerator`: the\n * `shots=N, sandbox=on` setting of the one `improvementDriver`. It runs a real\n * coding harness (claude / codex / opencode) inside the candidate worktree the\n * driver already created, letting the agent read the codebase + the research\n * report and make the change in place. The driver then commits the worktree\n * into a `CodeSurface`.\n \n Mechanism: identical to the proven Phase-2.8 in-process executor — spawn the\n * harness as a subprocess with `cwd` = the worktree, on the same filesystem,\n * so edits land in place (no sandbox-mount round-trip). `runLocalHarness` is\n * the verified primitive. The OUTER sandbox is the improvement loop's own\n * execution context; the generator does not nest a second sandbox per\n * candidate (which would reintroduce a host↔sandbox worktree-transport\n * problem that does not need solving here).\n \n `maxShots` is the DEPTH dial: the harness runs once; if it produced no change\n * (the worktree stays clean), the generator refines the prompt and retries, up\n * to `maxShots` times. A harness that already changed files returns on shot 1.\n /\n\nimport { spawnSync } from 'node:child_process'\nimport type { AnalystFinding } from '@tangle-network/agent-eval'\nimport { type LocalHarness, runLocalHarness } from '../mcp/local-harness'\nimport type { CandidateGenerator } from './improvement-driver'\n\nexport interface AgenticGeneratorOptions {\n /* Local coding harness to run in the worktree. Default `claude`. /\n harness?: LocalHarness\n /* Per-shot wall-clock timeout (ms). Default = `runLocalHarness` default (5m). /\n timeoutMs?: number\n /* Build the harness task prompt from the report + findings. Override for\n * domain phrasing; the default turns findings into a concrete coder task. /\n buildPrompt?: (args: { report: unknown; findings: AnalystFinding[] }) => string\n /* Test seam — inject the harness runner (defaults to `runLocalHarness`). /\n runHarness?: typeof runLocalHarness\n /* Test seam — inject the worktree-dirty check (defaults to `git status`). /\n isDirty?: (worktreePath: string) => boolean\n}\n\nexport function agenticGenerator(opts: AgenticGeneratorOptions = {}): CandidateGenerator {\n const harness = opts.harness ?? 'claude'\n const buildPrompt = opts.buildPrompt ?? defaultBuildPrompt\n const run = opts.runHarness ?? runLocalHarness\n const dirty = opts.isDirty ?? worktreeDirty\n\n return {\n kind: `agentic:${harness}`,\n async generate({ worktreePath, report, findings, maxShots, signal }) {\n let prompt = buildPrompt({ report, findings })\n const shots = Math.max(1, maxShots)\n\n for (let shot = 0; shot < shots; shot++) {\n if (signal.aborted) break\n await run({\n harness,\n cwd: worktreePath,\n taskPrompt: prompt,\n timeoutMs: opts.timeoutMs,\n signal,\n })\n // The worktree IS the signal: if the harness touched files, we have a\n // candidate. We don't trust the harness's stdout — we trust the diff.\n if (dirty(worktreePath)) {\n return { applied: true, summary: summarize(findings) }\n }\n // No change this shot — give the next attempt explicit feedback.\n prompt = refine(prompt)\n }\n return { applied: false, summary: '' }\n },\n }\n}\n\n/* Turn the analyst's findings (+ optional report) into a concrete coder task. /\nfunction defaultBuildPrompt(args: { report: unknown; findings: AnalystFinding[] }): string {\n const lines: string[] = [\n 'You are improving this codebase based on an evaluation analysis.',\n 'Make the smallest set of edits that addresses the findings below, then stop.',\n 'Do not change unrelated code. Do not commit — leave changes in the working tree.',\n '',\n 'Findings:',\n ]\n for (const f of args.findings) {\n const where = f.subject ? ` [${f.subject}]` : ''\n lines.push(`- (${f.severity})${where} ${f.claim}`)\n if (f.recommended_action) lines.push(` → ${f.recommended_action}`)\n }\n return lines.join('\\n')\n}\n\nfunction refine(prompt: string): string {\n return `${prompt}\\n\\nNOTE: your previous attempt left the working tree unchanged. Make the concrete file edits now.`\n}\n\n/* A one-line summary for the commit message, derived from the findings. /\nfunction summarize(findings: AnalystFinding[]): string {\n if (findings.length === 0) return 'agentic improvement'\n if (findings.length === 1) return `agentic: ${truncate(findings[0]!.claim, 64)}`\n return `agentic: ${findings.length} findings addressed`\n}\n\nfunction truncate(s: string, n: number): string {\n return s.length <= n ? s : `${s.slice(0, n - 1)}…`\n}\n\n/* Non-empty `git status --porcelain` ⇒ the harness changed the worktree.\n * Fails loud: the worktree is a fresh checkout, so a git error here means\n * something is genuinely broken (git missing, corrupt index, killed mid-run).\n * Folding that into `false` would silently discard a candidate and mask the\n * real failure — forbidden by the no-silent-fallbacks doctrine. /\nfunction worktreeDirty(worktreePath: string): boolean {\n const result = spawnSync('git', ['status', '--porcelain'], {\n cwd: worktreePath,\n encoding: 'utf-8',\n })\n if (result.error) {\n throw new Error(\n `agenticGenerator: git status failed to spawn in ${worktreePath}: ${result.error.message}`,\n )\n }\n if (result.status !== 0) {\n throw new Error(\n `agenticGenerator: git status exited ${result.status} in ${worktreePath}: ${result.stderr.trim()}`,\n )\n }\n return result.stdout.trim().length > 0\n}\n","/\n @experimental\n \n `improvementDriver` — the ONE reflective/agentic improvement driver for\n * agent-eval's improvement loop. It implements `ImprovementDriver` and owns\n * the candidate lifecycle (worktree create → generate → finalize/discard,\n * × populationSize); it delegates the only thing that genuinely varies — HOW\n * a candidate change is produced — to a pluggable `CandidateGenerator`.\n \n There is no separate \"analyst driver\" vs \"autoresearch driver\": those are\n * the SAME driver at two settings of a dial.\n * - cheap reflective path → `reflectiveGenerator` (shots=1, no sandbox;\n * applies pre-drafted patches)\n * - full agentic path → `agenticGenerator` (shots=N, sandbox runLoop;\n * an agent reads code + report and edits)\n * Both emit changes into a worktree the driver finalizes into a\n * `CodeSurface{ worktreeRef }` the loop measures on the holdout. See\n * agent-eval's `docs/design/self-improvement-engine.md`.\n /\n\nimport type { AnalystFinding } from '@tangle-network/agent-eval'\nimport type {\n CodeSurface,\n ImprovementDriver,\n LabeledScenarioStore,\n ProposeContext,\n WorktreeAdapter,\n} from '@tangle-network/agent-eval/campaign'\n\n/* The byte-producing seam — the ONE thing that differs between the cheap\n * reflective path and the full agentic path. A generator makes (uncommitted)\n * changes inside `worktreePath`; the driver commits them via the worktree\n * adapter's `finalize`. /\nexport interface CandidateGenerator {\n kind: string\n generate(args: {\n /* The candidate worktree — a fresh checkout of baseRef. Write changes here. /\n worktreePath: string\n /* Phase-2 research report (analyst findings + diff), opaque. /\n report: unknown\n /* Findings resolved from the report or the loop context. /\n findings: AnalystFinding[]\n /* Handle to all captured data, to ground the change. /\n dataset?: LabeledScenarioStore\n /* DEPTH: max iterations the generator may take (agentic uses this; the\n * reflective generator ignores it). /\n maxShots: number\n signal: AbortSignal\n }): Promise<{ applied: boolean; summary: string }>\n}\n\nexport interface ImprovementDriverOptions {\n worktree: WorktreeAdapter\n generator: CandidateGenerator\n /* Base ref candidate worktrees fork from. Default `main`. /\n baseRef?: string\n}\n\nexport function improvementDriver(\n opts: ImprovementDriverOptions,\n): ImprovementDriver<AnalystFinding> {\n const baseRef = opts.baseRef ?? 'main'\n\n return {\n kind: `improvement:${opts.generator.kind}`,\n async propose(ctx) {\n const findings = resolveFindings(ctx)\n // No signal to act on — propose nothing rather than spin up worktrees.\n if (findings.length === 0 && ctx.report === undefined) return []\n\n const surfaces: CodeSurface[] = []\n for (let i = 0; i < ctx.populationSize; i++) {\n if (ctx.signal.aborted) break\n const wt = await opts.worktree.create({\n baseRef,\n label: `${opts.generator.kind}-gen${ctx.generation}-cand${i}`,\n })\n // Once a worktree exists it MUST be accounted for: finalized into a\n // surface, or discarded. A throw from generate()/finalize() must not\n // leak the worktree + branch — discard best-effort, then rethrow loud.\n try {\n const { applied, summary } = await opts.generator.generate({\n worktreePath: wt.path,\n report: ctx.report,\n findings,\n dataset: ctx.dataset,\n maxShots: ctx.maxImprovementShots ?? 1,\n signal: ctx.signal,\n })\n if (!applied) {\n await opts.worktree.discard(wt)\n continue\n }\n surfaces.push(await opts.worktree.finalize(wt, summary))\n } catch (err) {\n // Best-effort cleanup; never mask the original failure.\n await opts.worktree.discard(wt).catch(() => {})\n throw err\n }\n }\n return surfaces\n },\n }\n}\n\n/* Phase-2 report carries `findings` when present; else fall back to the\n * loop's `ctx.findings`. The report is opaque to the substrate, so probe it\n * structurally. /\nfunction resolveFindings(ctx: ProposeContext<AnalystFinding>): AnalystFinding[] {\n const report = ctx.report\n if (report && typeof report === 'object' && 'findings' in report) {\n const f = (report as { findings: unknown }).findings\n if (Array.isArray(f) && f.length > 0) return f as AnalystFinding[]\n }\n return ctx.findings\n}\n","/\n @experimental\n \n `reflectiveGenerator` — the cheap, no-sandbox `CandidateGenerator`. It drafts\n * surface edits via the existing improvement adapter (`proposeFromFindings`,\n * one LLM patch per finding) and applies them as ONE coherent improvement into\n * the candidate worktree. `maxShots` is ignored — reflection is single-shot by\n * construction (the patches are already drafted).\n \n This is the `shots=1, sandbox=off` setting of the one improvement driver.\n * The `agenticGenerator` (sandbox runLoop) is the `shots=N, sandbox=on`\n * setting — both plug into the same `improvementDriver`.\n /\n\nimport { spawnSync } from 'node:child_process'\nimport type { SurfaceImprovementEdit } from '../agent/improvement-adapter'\nimport type { ImprovementAdapter } from '../analyst-loop/types'\nimport type { CandidateGenerator } from './improvement-driver'\n\nexport interface ReflectiveGeneratorOptions {\n improvementAdapter: ImprovementAdapter<SurfaceImprovementEdit>\n}\n\nexport function reflectiveGenerator(opts: ReflectiveGeneratorOptions): CandidateGenerator {\n return {\n kind: 'reflective',\n async generate({ worktreePath, findings }) {\n const batch = await opts.improvementAdapter.proposeFromFindings(findings)\n if (batch.edits.length === 0) return { applied: false, summary: '' }\n\n let applied = 0\n for (const edit of batch.edits) {\n if (applyPatch(edit.patch, worktreePath)) applied++\n }\n if (applied === 0) return { applied: false, summary: '' }\n\n const summary =\n batch.edits.length === 1\n ? batch.edits[0]!.summary\n : `analyst: ${applied} surface edit${applied === 1 ? '' : 's'}`\n return { applied: true, summary }\n },\n }\n}\n\n/* Mirror the improvement adapter's proven apply invocation, run inside the\n * candidate worktree (a fresh checkout of baseRef, so `-p0` paths match). */\nfunction applyPatch(patch: string, cwd: string): boolean {\n const result = spawnSync('git', ['apply', '--whitespace=fix', '-p0', '-'], {\n cwd,\n input: patch,\n encoding: 'utf-8',\n })\n return result.status === 0\n}\n"],"mappings":";;;;;;;;;;AAuBA,SAAS,iBAAiB;AAmBnB,SAAS,iBAAiB,OAAgC,CAAC,GAAuB;AACvF,QAAM,UAAU,KAAK,WAAW;AAChC,QAAM,cAAc,KAAK,eAAe;AACxC,QAAM,MAAM,KAAK,cAAc;AAC/B,QAAM,QAAQ,KAAK,WAAW;AAE9B,SAAO;AAAA,IACL,MAAM,WAAW,OAAO;AAAA,IACxB,MAAM,SAAS,EAAE,cAAc,QAAQ,UAAU,UAAU,OAAO,GAAG;AACnE,UAAI,SAAS,YAAY,EAAE,QAAQ,SAAS,CAAC;AAC7C,YAAM,QAAQ,KAAK,IAAI,GAAG,QAAQ;AAElC,eAAS,OAAO,GAAG,OAAO,OAAO,QAAQ;AACvC,YAAI,OAAO,QAAS;AACpB,cAAM,IAAI;AAAA,UACR;AAAA,UACA,KAAK;AAAA,UACL,YAAY;AAAA,UACZ,WAAW,KAAK;AAAA,UAChB;AAAA,QACF,CAAC;AAGD,YAAI,MAAM,YAAY,GAAG;AACvB,iBAAO,EAAE,SAAS,MAAM,SAAS,UAAU,QAAQ,EAAE;AAAA,QACvD;AAEA,iBAAS,OAAO,MAAM;AAAA,MACxB;AACA,aAAO,EAAE,SAAS,OAAO,SAAS,GAAG;AAAA,IACvC;AAAA,EACF;AACF;AAGA,SAAS,mBAAmB,MAA+D;AACzF,QAAM,QAAkB;AAAA,IACtB;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,EACF;AACA,aAAW,KAAK,KAAK,UAAU;AAC7B,UAAM,QAAQ,EAAE,UAAU,KAAK,EAAE,OAAO,MAAM;AAC9C,UAAM,KAAK,MAAM,EAAE,QAAQ,IAAI,KAAK,IAAI,EAAE,KAAK,EAAE;AACjD,QAAI,EAAE,mBAAoB,OAAM,KAAK,cAAS,EAAE,kBAAkB,EAAE;AAAA,EACtE;AACA,SAAO,MAAM,KAAK,IAAI;AACxB;AAEA,SAAS,OAAO,QAAwB;AACtC,SAAO,GAAG,MAAM;AAAA;AAAA;AAClB;AAGA,SAAS,UAAU,UAAoC;AACrD,MAAI,SAAS,WAAW,EAAG,QAAO;AAClC,MAAI,SAAS,WAAW,EAAG,QAAO,YAAY,SAAS,SAAS,CAAC,EAAG,OAAO,EAAE,CAAC;AAC9E,SAAO,YAAY,SAAS,MAAM;AACpC;AAEA,SAAS,SAAS,GAAW,GAAmB;AAC9C,SAAO,EAAE,UAAU,IAAI,IAAI,GAAG,EAAE,MAAM,GAAG,IAAI,CAAC,CAAC;AACjD;AAOA,SAAS,cAAc,cAA+B;AACpD,QAAM,SAAS,UAAU,OAAO,CAAC,UAAU,aAAa,GAAG;AAAA,IACzD,KAAK;AAAA,IACL,UAAU;AAAA,EACZ,CAAC;AACD,MAAI,OAAO,OAAO;AAChB,UAAM,IAAI;AAAA,MACR,mDAAmD,YAAY,KAAK,OAAO,MAAM,OAAO;AAAA,IAC1F;AAAA,EACF;AACA,MAAI,OAAO,WAAW,GAAG;AACvB,UAAM,IAAI;AAAA,MACR,uCAAuC,OAAO,MAAM,OAAO,YAAY,KAAK,OAAO,OAAO,KAAK,CAAC;AAAA,IAClG;AAAA,EACF;AACA,SAAO,OAAO,OAAO,KAAK,EAAE,SAAS;AACvC;;;ACvEO,SAAS,kBACd,MACmC;AACnC,QAAM,UAAU,KAAK,WAAW;AAEhC,SAAO;AAAA,IACL,MAAM,eAAe,KAAK,UAAU,IAAI;AAAA,IACxC,MAAM,QAAQ,KAAK;AACjB,YAAM,WAAW,gBAAgB,GAAG;AAEpC,UAAI,SAAS,WAAW,KAAK,IAAI,WAAW,OAAW,QAAO,CAAC;AAE/D,YAAM,WAA0B,CAAC;AACjC,eAAS,IAAI,GAAG,IAAI,IAAI,gBAAgB,KAAK;AAC3C,YAAI,IAAI,OAAO,QAAS;AACxB,cAAM,KAAK,MAAM,KAAK,SAAS,OAAO;AAAA,UACpC;AAAA,UACA,OAAO,GAAG,KAAK,UAAU,IAAI,OAAO,IAAI,UAAU,QAAQ,CAAC;AAAA,QAC7D,CAAC;AAID,YAAI;AACF,gBAAM,EAAE,SAAS,QAAQ,IAAI,MAAM,KAAK,UAAU,SAAS;AAAA,YACzD,cAAc,GAAG;AAAA,YACjB,QAAQ,IAAI;AAAA,YACZ;AAAA,YACA,SAAS,IAAI;AAAA,YACb,UAAU,IAAI,uBAAuB;AAAA,YACrC,QAAQ,IAAI;AAAA,UACd,CAAC;AACD,cAAI,CAAC,SAAS;AACZ,kBAAM,KAAK,SAAS,QAAQ,EAAE;AAC9B;AAAA,UACF;AACA,mBAAS,KAAK,MAAM,KAAK,SAAS,SAAS,IAAI,OAAO,CAAC;AAAA,QACzD,SAAS,KAAK;AAEZ,gBAAM,KAAK,SAAS,QAAQ,EAAE,EAAE,MAAM,MAAM;AAAA,UAAC,CAAC;AAC9C,gBAAM;AAAA,QACR;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAKA,SAAS,gBAAgB,KAAuD;AAC9E,QAAM,SAAS,IAAI;AACnB,MAAI,UAAU,OAAO,WAAW,YAAY,cAAc,QAAQ;AAChE,UAAM,IAAK,OAAiC;AAC5C,QAAI,MAAM,QAAQ,CAAC,KAAK,EAAE,SAAS,EAAG,QAAO;AAAA,EAC/C;AACA,SAAO,IAAI;AACb;;;ACrGA,SAAS,aAAAA,kBAAiB;AASnB,SAAS,oBAAoB,MAAsD;AACxF,SAAO;AAAA,IACL,MAAM;AAAA,IACN,MAAM,SAAS,EAAE,cAAc,SAAS,GAAG;AACzC,YAAM,QAAQ,MAAM,KAAK,mBAAmB,oBAAoB,QAAQ;AACxE,UAAI,MAAM,MAAM,WAAW,EAAG,QAAO,EAAE,SAAS,OAAO,SAAS,GAAG;AAEnE,UAAI,UAAU;AACd,iBAAW,QAAQ,MAAM,OAAO;AAC9B,YAAI,WAAW,KAAK,OAAO,YAAY,EAAG;AAAA,MAC5C;AACA,UAAI,YAAY,EAAG,QAAO,EAAE,SAAS,OAAO,SAAS,GAAG;AAExD,YAAM,UACJ,MAAM,MAAM,WAAW,IACnB,MAAM,MAAM,CAAC,EAAG,UAChB,YAAY,OAAO,gBAAgB,YAAY,IAAI,KAAK,GAAG;AACjE,aAAO,EAAE,SAAS,MAAM,QAAQ;AAAA,IAClC;AAAA,EACF;AACF;AAIA,SAAS,WAAW,OAAe,KAAsB;AACvD,QAAM,SAASA,WAAU,OAAO,CAAC,SAAS,oBAAoB,OAAO,GAAG,GAAG;AAAA,IACzE;AAAA,IACA,OAAO;AAAA,IACP,UAAU;AAAA,EACZ,CAAC;AACD,SAAO,OAAO,WAAW;AAC3B;","names":["spawnSync"]}

package/dist/index.d.ts CHANGED Viewed

@@ -2,10 +2,14 @@ import { AgentEvalError, KnowledgeReadinessReport, RunRecord, ControlEvalResult,
 export { AgentEvalError, AgentEvalErrorCode, ConfigError, ControlBudget, ControlDecision, ControlEvalResult, ControlRunResult, ControlStep, DataAcquisitionPlan, JudgeError, KnowledgeReadinessReport, KnowledgeRequirement, NotFoundError, RunRecord, ValidationError } from '@tangle-network/agent-eval';
 import { a as AgentBackendInput, b as AgentExecutionBackend, O as OpenAIChatTool, c as OpenAIChatToolChoice, d as AgentBackendContext, R as RuntimeStreamEvent, K as KnowledgeReadinessDecision, e as RunAgentTaskOptions, f as AgentTaskRunResult, g as RunAgentTaskStreamOptions, h as AgentRuntimeEvent, i as AgentTaskStatus, j as RuntimeSessionStore, k as RuntimeSession } from './types-CsCCryln.js';
 export { l as AgentAdapter, m as AgentKnowledgeProvider, n as AgentRuntimeEventSink, o as AgentTaskContext, A as AgentTaskSpec, B as BackendErrorDetail } from './types-CsCCryln.js';
-import { L as LoopSandboxClient } from './types-CmTjKLyB.js';
-export { R as RuntimeRunHandle, p as RuntimeRunPersistenceAdapter, q as RuntimeRunRow, s as startRuntimeRun } from './types-CmTjKLyB.js';
-import { d as DelegateCodeArgs, t as CoderReviewer, u as CoderWinnerSelection } from './otel-export-DgFMwsVy.js';
-export { L as EvalRunEvent, M as EvalRunGeneration, N as EvalRunsExportConfig, P as EvalRunsExportResult, Q as INTELLIGENCE_WIRE_VERSION, T as OtelAttribute, U as OtelExportConfig, O as OtelExporter, V as OtelSpan, W as buildLoopOtelSpans, X as createOtelExporter, Y as exportEvalRuns, Z as loopEventToOtelSpan, J as mcpToolsForRuntimeMcp, K as mcpToolsForRuntimeMcpSubset } from './otel-export-DgFMwsVy.js';
+import { Scenario } from '@tangle-network/agent-eval/campaign';
+import { R as RunAnalystLoopOpts, a as RunAnalystLoopResult } from './types-p8dWBIXL.js';
+import { O as OptimizePromptOptions, b as OptimizePromptResult } from './optimize-prompt-cmH9wZdH.js';
+import { T as TopologyPlanner, D as DynamicDecision } from './dynamic-DeOPeeAw.js';
+import { L as LoopSandboxClient, O as OutputAdapter, V as Validator, A as AgentRunSpec, b as LoopResult } from './types-CmkQl8qE.js';
+export { R as RuntimeRunHandle, p as RuntimeRunPersistenceAdapter, q as RuntimeRunRow, s as startRuntimeRun } from './types-CmkQl8qE.js';
+import { d as DelegateCodeArgs, t as CoderReviewer, u as CoderWinnerSelection, A as FactCandidate, w as CreateKbGateOptions } from './otel-export-CNmeg_7B.js';
+export { U as EvalRunEvent, V as EvalRunGeneration, W as EvalRunsExportConfig, X as EvalRunsExportResult, Y as INTELLIGENCE_WIRE_VERSION, Z as OtelAttribute, _ as OtelExportConfig, O as OtelExporter, $ as OtelSpan, a0 as buildLoopOtelSpans, a1 as createOtelExporter, a2 as exportEvalRuns, a3 as loopEventToOtelSpan, Q as mcpToolsForRuntimeMcp, T as mcpToolsForRuntimeMcpSubset } from './otel-export-CNmeg_7B.js';
 import { CoderOutput } from './profiles.js';
 import '@tangle-network/sandbox';
@@ -1089,6 +1093,64 @@ declare function coderLoopRunner(options: CoderLoopRunnerOptions): DelegatedLoop
 declare function reviewLoopRunner(options: CoderLoopRunnerOptions & {
     reviewer: CoderReviewer;
 }): DelegatedLoopRunner<CoderOutput>;
+/** @experimental Options for the default `dynamic` runner. */
+interface DynamicLoopRunnerOptions<Task, Output> {
+    sandboxClient: LoopSandboxClient;
+    /** The agent-authored topology planner (e.g. `createSandboxPlanner(...)`). */
+    planner: TopologyPlanner<Task, Output>;
+    task: Task;
+    output: OutputAdapter<Output>;
+    validator?: Validator<Output>;
+    /** Exactly one of `agentRun` / `agentRuns` (runLoop validates). */
+    agentRun?: AgentRunSpec<Task>;
+    agentRuns?: AgentRunSpec<Task>[];
+    maxIterations?: number;
+    maxFanout?: number;
+}
+/** @experimental `dynamic` mode — agent-authored topology over `runLoop`. */
+declare function dynamicLoopRunner<Task, Output>(o: DynamicLoopRunnerOptions<Task, Output>): DelegatedLoopRunner<LoopResult<Task, Output, DynamicDecision>>;
+/** @experimental A fact rejected at the KB gate — surfaced, never dropped. */
+interface VetoedFact {
+    candidate: FactCandidate;
+    vetoedBy?: string;
+    reason?: string;
+}
+/** @experimental */
+interface ResearchLoopResult {
+    /** Facts that passed the fail-closed gate — safe to write to the KB. */
+    accepted: FactCandidate[];
+    /** Facts the gate vetoed in the final round — escalate, do not silently drop. */
+    vetoed: VetoedFact[];
+    /** Research rounds actually run. */
+    rounds: number;
+}
+/** @experimental Options for the default `research` runner. */
+interface ResearchLoopRunnerOptions {
+    /**
+     * The research engine (the consumer's web/doc searcher + extractor). Called
+     * each round with the prior round's vetoes so it can re-research the gaps.
+     * Returns fact candidates carrying their grounding (`verbatimPassage` +
+     * `sourceText`).
+     */
+    research: (round: number, vetoed: VetoedFact[]) => Promise<FactCandidate[]>;
+    /** Gate config (extra judges, self-artifact kinds, …). The floor is always on. */
+    gate?: CreateKbGateOptions;
+    /** Max research rounds (correct-on-veto remediation). Default 1. */
+    maxRounds?: number;
+}
+/**
+ * @experimental `research` mode — research-in-a-loop with valid-only KB growth.
+ *
+ * Each round: research → gate every candidate (fail-closed; passage MUST be in
+ * the source) → accept the clean ones → re-research the vetoed ones next round,
+ * up to `maxRounds`. Vetoed facts in the final round are RETURNED (escalate,
+ * never silently dropped) so the caller audits vs retries.
+ */
+declare function researchLoopRunner(o: ResearchLoopRunnerOptions): DelegatedLoopRunner<ResearchLoopResult>;
+/** @experimental `self-improve` mode — identity-gated prompt optimization. */
+declare function selfImproveLoopRunner<TScenario extends Scenario, TArtifact>(options: OptimizePromptOptions<TScenario, TArtifact>): DelegatedLoopRunner<OptimizePromptResult<TArtifact, TScenario>>;
+/** @experimental `audit` mode — analyst loop over captured trace/run data. */
+declare function auditLoopRunner<TProposal = unknown, TEdit = unknown>(options: RunAnalystLoopOpts): DelegatedLoopRunner<RunAnalystLoopResult<TProposal, TEdit>>;
 /**
  * @stable
@@ -1371,4 +1433,4 @@ declare function readinessServerSentEvent(report: KnowledgeReadinessReport, opti
 /** @stable */
 declare function runtimeStreamServerSentEvent(event: RuntimeStreamEvent, options?: RuntimeTelemetryOptions & ServerSentEventOptions): string;
-export { AgentBackendContext, AgentBackendInput, AgentExecutionBackend, AgentRuntimeEvent, AgentTaskRunResult, AgentTaskStatus, type AuthSource, type BackendCallPolicy, BackendTransportError, type ChatStreamEvent, type ChatTurnHooks, type ChatTurnIdentity, type ChatTurnProducer, type ChatTurnResult, type CircuitBreakerConfig, CircuitBreakerState, CircuitOpenError, type CoderLoopRunnerOptions, type Conversation, type ConversationDriveState, type ConversationJournal, type ConversationJournalEntry, type ConversationParticipant, type ConversationPolicy, type ConversationResult, type ConversationStreamEvent, type ConversationTurn, type D1DatabaseLike, type D1StmtLike, DEFAULT_MAX_DEPTH, DEFAULT_ROUTER_BASE_URL, DeadlineExceededError, type DelegatedLoopMode, type DelegatedLoopRegistry, type DelegatedLoopResult, type DelegatedLoopRunner, FORWARD_HEADERS, FileConversationJournal, type ForwardHeaderName, type HaltContext, type HaltPredicate, type HaltReason, type HaltSignal, InMemoryConversationJournal, InMemoryRuntimeSessionStore, type ModelInfo, OpenAIChatTool, OpenAIChatToolChoice, PlannerError, type PropagatedHeaders, type ResolvedChatModel, type RetryBackoff, type RetryableErrorPredicate, type RouterEnv, type RunChatTurnInput, type RunConversationOptions, type RunDelegatedLoopOptions, type RuntimeEventCollector, RuntimeRunStateError, RuntimeSessionStore, RuntimeStreamEvent, type RuntimeStreamEventCollector, type RuntimeTelemetryOptions, type SanitizedKnowledgeReadinessReport, type SqlAdapter, SqlConversationJournal, type TurnOrder, applyRunRecordDefaults, buildForwardHeaders, cleanModelId, coderLoopRunner, computeBackoff, createConversationBackend, createIterableBackend, createOpenAICompatibleBackend, createRuntimeEventCollector, createRuntimeStreamEventCollector, createSandboxPromptBackend, d1ToSqlAdapter, decideKnowledgeReadiness, defaultIsRetryable, defineConversation, deriveExecutionId, getModels, handleChatTurn, isDepthExceeded, makePerAttemptSignal, readDepth, readinessServerSentEvent, resolveChatModel, resolveRouterBaseUrl, reviewLoopRunner, runAgentTask, runAgentTaskStream, runConversation, runConversationStream, runDelegatedLoop, runtimeStreamServerSentEvent, sanitizeAgentRuntimeEvent, sanitizeKnowledgeReadinessReport, sanitizeRuntimeStreamEvent, sleep, slugifySpeaker, turnId, validateChatModelId };
+export { AgentBackendContext, AgentBackendInput, AgentExecutionBackend, AgentRuntimeEvent, AgentTaskRunResult, AgentTaskStatus, type AuthSource, type BackendCallPolicy, BackendTransportError, type ChatStreamEvent, type ChatTurnHooks, type ChatTurnIdentity, type ChatTurnProducer, type ChatTurnResult, type CircuitBreakerConfig, CircuitBreakerState, CircuitOpenError, type CoderLoopRunnerOptions, type Conversation, type ConversationDriveState, type ConversationJournal, type ConversationJournalEntry, type ConversationParticipant, type ConversationPolicy, type ConversationResult, type ConversationStreamEvent, type ConversationTurn, type D1DatabaseLike, type D1StmtLike, DEFAULT_MAX_DEPTH, DEFAULT_ROUTER_BASE_URL, DeadlineExceededError, type DelegatedLoopMode, type DelegatedLoopRegistry, type DelegatedLoopResult, type DelegatedLoopRunner, type DynamicLoopRunnerOptions, FORWARD_HEADERS, FileConversationJournal, type ForwardHeaderName, type HaltContext, type HaltPredicate, type HaltReason, type HaltSignal, InMemoryConversationJournal, InMemoryRuntimeSessionStore, type ModelInfo, OpenAIChatTool, OpenAIChatToolChoice, PlannerError, type PropagatedHeaders, type ResearchLoopResult, type ResearchLoopRunnerOptions, type ResolvedChatModel, type RetryBackoff, type RetryableErrorPredicate, type RouterEnv, type RunChatTurnInput, type RunConversationOptions, type RunDelegatedLoopOptions, type RuntimeEventCollector, RuntimeRunStateError, RuntimeSessionStore, RuntimeStreamEvent, type RuntimeStreamEventCollector, type RuntimeTelemetryOptions, type SanitizedKnowledgeReadinessReport, type SqlAdapter, SqlConversationJournal, type TurnOrder, type VetoedFact, applyRunRecordDefaults, auditLoopRunner, buildForwardHeaders, cleanModelId, coderLoopRunner, computeBackoff, createConversationBackend, createIterableBackend, createOpenAICompatibleBackend, createRuntimeEventCollector, createRuntimeStreamEventCollector, createSandboxPromptBackend, d1ToSqlAdapter, decideKnowledgeReadiness, defaultIsRetryable, defineConversation, deriveExecutionId, dynamicLoopRunner, getModels, handleChatTurn, isDepthExceeded, makePerAttemptSignal, readDepth, readinessServerSentEvent, researchLoopRunner, resolveChatModel, resolveRouterBaseUrl, reviewLoopRunner, runAgentTask, runAgentTaskStream, runConversation, runConversationStream, runDelegatedLoop, runtimeStreamServerSentEvent, sanitizeAgentRuntimeEvent, sanitizeKnowledgeReadinessReport, sanitizeRuntimeStreamEvent, selfImproveLoopRunner, sleep, slugifySpeaker, turnId, validateChatModelId };

package/dist/index.js CHANGED Viewed

@@ -1,16 +1,26 @@
+import {
+  runAnalystLoop
+} from "./chunk-XBUG326M.js";
+import {
+  optimizePrompt
+} from "./chunk-VOX6Z3II.js";
 import {
   INTELLIGENCE_WIRE_VERSION,
   buildLoopOtelSpans,
+  createKbGate,
   createOtelExporter,
   exportEvalRuns,
   loopEventToOtelSpan,
   mcpToolsForRuntimeMcp,
   mcpToolsForRuntimeMcpSubset
-} from "./chunk-T3GJBKHA.js";
+} from "./chunk-Z523NPJK.js";
 import {
   createDefaultCoderDelegate
 } from "./chunk-V6GURW4W.js";
-import "./chunk-7JBDJQLO.js";
+import {
+  createDynamicDriver,
+  runLoop
+} from "./chunk-7JBDJQLO.js";
 import "./chunk-3HMHSN22.js";
 import "./chunk-PY6NMZYX.js";
 import {
@@ -1768,6 +1778,51 @@ function coderLoopRunner(options) {
 function reviewLoopRunner(options) {
   return coderLoopRunner(options);
 }
+function dynamicLoopRunner(o) {
+  return async (signal) => runLoop({
+    driver: createDynamicDriver({
+      planner: o.planner,
+      ...o.maxIterations !== void 0 ? { maxIterations: o.maxIterations } : {},
+      ...o.maxFanout !== void 0 ? { maxFanout: o.maxFanout } : {}
+    }),
+    ...o.agentRun ? { agentRun: o.agentRun } : {},
+    ...o.agentRuns ? { agentRuns: o.agentRuns } : {},
+    output: o.output,
+    ...o.validator ? { validator: o.validator } : {},
+    task: o.task,
+    ctx: { sandboxClient: o.sandboxClient, signal },
+    ...o.maxIterations !== void 0 ? { maxIterations: o.maxIterations } : {}
+  });
+}
+function researchLoopRunner(o) {
+  const gate = createKbGate(o.gate);
+  const maxRounds = Math.max(1, Math.trunc(o.maxRounds ?? 1));
+  return async (signal) => {
+    const accepted = [];
+    let vetoed = [];
+    let rounds = 0;
+    for (let round = 0; round < maxRounds; round += 1) {
+      if (signal.aborted) break;
+      rounds += 1;
+      const candidates = await o.research(round, vetoed);
+      if (candidates.length === 0) break;
+      vetoed = [];
+      for (const c of candidates) {
+        const v = await gate(c);
+        if (v.accepted) accepted.push(c);
+        else vetoed.push({ candidate: c, vetoedBy: v.vetoedBy, reason: v.reason });
+      }
+      if (vetoed.length === 0) break;
+    }
+    return { accepted, vetoed, rounds };
+  };
+}
+function selfImproveLoopRunner(options) {
+  return async () => optimizePrompt(options);
+}
+function auditLoopRunner(options) {
+  return async () => runAnalystLoop(options);
+}
 // src/model-resolution.ts
 var DEFAULT_ROUTER_BASE_URL = "https://router.tangle.tools";
@@ -2774,6 +2829,7 @@ export {
   SqlConversationJournal,
   ValidationError,
   applyRunRecordDefaults,
+  auditLoopRunner,
   buildForwardHeaders,
   buildLoopOtelSpans,
   cleanModelId,
@@ -2791,6 +2847,7 @@ export {
   defaultIsRetryable,
   defineConversation,
   deriveExecutionId,
+  dynamicLoopRunner,
   exportEvalRuns,
   getModels,
   handleChatTurn,
@@ -2801,6 +2858,7 @@ export {
   mcpToolsForRuntimeMcpSubset,
   readDepth,
   readinessServerSentEvent,
+  researchLoopRunner,
   resolveChatModel,
   resolveRouterBaseUrl,
   reviewLoopRunner,
@@ -2813,6 +2871,7 @@ export {
   sanitizeAgentRuntimeEvent,
   sanitizeKnowledgeReadinessReport,
   sanitizeRuntimeStreamEvent,
+  selfImproveLoopRunner,
   sleep2 as sleep,
   slugifySpeaker,
   startRuntimeRun,