@tangle-network/agent-eval 0.133.3 → 0.134.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/CHANGELOG.md +62 -0
  2. package/dist/analyst/index.d.ts +11 -35
  3. package/dist/analyst/index.d.ts.map +1 -1
  4. package/dist/analyst/index.js +4 -53
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/{analyze-runs-BClW9OSe.d.ts → analyze-runs-DMo3Lb_y.d.ts} +4 -4
  7. package/dist/{analyze-runs-BClW9OSe.d.ts.map → analyze-runs-DMo3Lb_y.d.ts.map} +1 -1
  8. package/dist/benchmarks/index.d.ts +1 -1
  9. package/dist/benchmarks/index.js +1 -1
  10. package/dist/{benchmarks-BP9sgMia.js → benchmarks-v5piCeDl.js} +3 -3
  11. package/dist/{benchmarks-BP9sgMia.js.map → benchmarks-v5piCeDl.js.map} +1 -1
  12. package/dist/campaign/index.d.ts +5 -4
  13. package/dist/campaign/index.js +4 -3
  14. package/dist/{campaign--V4ffEKR.js → campaign-DEC_7DLn.js} +2 -2
  15. package/dist/{campaign--V4ffEKR.js.map → campaign-DEC_7DLn.js.map} +1 -1
  16. package/dist/{client-Du7B81wW.d.ts → client-BIyh1RCr.d.ts} +3 -3
  17. package/dist/{client-Du7B81wW.d.ts.map → client-BIyh1RCr.d.ts.map} +1 -1
  18. package/dist/contract/index.d.ts +12 -11
  19. package/dist/contract/index.d.ts.map +1 -1
  20. package/dist/contract/index.js +4 -11
  21. package/dist/contract/index.js.map +1 -1
  22. package/dist/{default-registry-Cl3pHo4n.d.ts → default-registry-Brxr728w.d.ts} +4 -268
  23. package/dist/default-registry-Brxr728w.d.ts.map +1 -0
  24. package/dist/{default-registry-D3T9XbuY.js → default-registry-IjYs7T8l.js} +4 -61
  25. package/dist/default-registry-IjYs7T8l.js.map +1 -0
  26. package/dist/hosted/index.d.ts +2 -2
  27. package/dist/{index-DOqvIJ8I.d.ts → index-BoJNQR6n.d.ts} +4 -3
  28. package/dist/index-BoJNQR6n.d.ts.map +1 -0
  29. package/dist/{index-B5MNN1f1.d.ts → index-C21xKtxu.d.ts} +4 -4
  30. package/dist/{index-B5MNN1f1.d.ts.map → index-C21xKtxu.d.ts.map} +1 -1
  31. package/dist/index.d.ts +12 -11
  32. package/dist/index.d.ts.map +1 -1
  33. package/dist/index.js +109 -11
  34. package/dist/index.js.map +1 -1
  35. package/dist/multishot/index.d.ts +1 -1
  36. package/dist/openapi.json +1 -1
  37. package/dist/proposal-findings-DCawte-y.js +164 -0
  38. package/dist/proposal-findings-DCawte-y.js.map +1 -0
  39. package/dist/{release-report-DKBtegGt.d.ts → release-report-CuULWKyk.d.ts} +2 -2
  40. package/dist/{release-report-DKBtegGt.d.ts.map → release-report-CuULWKyk.d.ts.map} +1 -1
  41. package/dist/reporting.d.ts +2 -2
  42. package/dist/{researcher-BtD5U1Up.d.ts → researcher-DVtruQ9U.d.ts} +2 -2
  43. package/dist/{researcher-BtD5U1Up.d.ts.map → researcher-DVtruQ9U.d.ts.map} +1 -1
  44. package/dist/rl.d.ts +2 -2
  45. package/dist/rl.d.ts.map +1 -1
  46. package/dist/rl.js +14 -1
  47. package/dist/rl.js.map +1 -1
  48. package/dist/{semantic-concept-judge-BypLt6Fw.js → semantic-concept-judge-C0P1VTXD.js} +2 -3
  49. package/dist/{semantic-concept-judge-BypLt6Fw.js.map → semantic-concept-judge-C0P1VTXD.js.map} +1 -1
  50. package/dist/{skill-usage-BaaxFSJR.d.ts → skill-usage-BDQVPIG1.d.ts} +3 -2
  51. package/dist/skill-usage-BDQVPIG1.d.ts.map +1 -0
  52. package/dist/{skillopt-optimization-method-vvJ4bMNI.js → skillopt-optimization-method-BY6vKLJB.js} +51 -30
  53. package/dist/skillopt-optimization-method-BY6vKLJB.js.map +1 -0
  54. package/dist/{skillopt-optimization-method-Dxr8pdZd.d.ts → skillopt-optimization-method-DJ3l4w8W.d.ts} +10 -13
  55. package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +1 -0
  56. package/dist/{summary-report-DyOhItws.d.ts → summary-report-DGp0-_XO.d.ts} +59 -3
  57. package/dist/summary-report-DGp0-_XO.d.ts.map +1 -0
  58. package/dist/types-DVjczBM9.d.ts +276 -0
  59. package/dist/types-DVjczBM9.d.ts.map +1 -0
  60. package/dist/{types-BokuXvOG.d.ts → types-DiWLru6Z.d.ts} +20 -37
  61. package/dist/types-DiWLru6Z.d.ts.map +1 -0
  62. package/docs/campaign-proposers.md +5 -0
  63. package/package.json +1 -1
  64. package/dist/default-registry-Cl3pHo4n.d.ts.map +0 -1
  65. package/dist/default-registry-D3T9XbuY.js.map +0 -1
  66. package/dist/index-DOqvIJ8I.d.ts.map +0 -1
  67. package/dist/run-score-iEEAWiBY.js +0 -41
  68. package/dist/run-score-iEEAWiBY.js.map +0 -1
  69. package/dist/skill-usage-BaaxFSJR.d.ts.map +0 -1
  70. package/dist/skillopt-optimization-method-Dxr8pdZd.d.ts.map +0 -1
  71. package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +0 -1
  72. package/dist/summary-report-DyOhItws.d.ts.map +0 -1
  73. package/dist/types-BokuXvOG.d.ts.map +0 -1
@@ -1,8 +1,7 @@
1
1
  import { a as NotFoundError } from "./errors-8YnH8WlF.js";
2
- import { P as computeFindingId } from "./default-registry-D3T9XbuY.js";
3
2
  import { i as CostLedger } from "./cost-ledger-BrJxbrMy.js";
4
3
  import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, u as costReceiptFromLlmError } from "./llm-client-ClPW-dWB.js";
5
- import { n as aggregateRunScore, r as clamp01 } from "./run-score-iEEAWiBY.js";
4
+ import { a as clamp01, i as aggregateRunScore, s as computeFindingId } from "./proposal-findings-DCawte-y.js";
6
5
  import { t as Mutex } from "./concurrency-MUjT7VjM.js";
7
6
  import { appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync, statSync } from "node:fs";
8
7
  import { dirname, join } from "node:path";
@@ -722,4 +721,4 @@ function createSemanticConceptJudge(options = {}) {
722
721
  //#endregion
723
722
  export { RunCritic as a, buildSkillUsageReport as c, defaultIsMaterial as d, diffFindings as f, runSemanticConceptJudge as i, emitSkillUsageFindings as l, SEMANTIC_CONCEPT_JUDGE_VERSION as n, SKILL_USAGE_ANALYST as o, LockedJsonlAppender as p, createSemanticConceptJudge as r, SkillUsageAnalyst as s, DEFAULT_COMPLEXITY_WEIGHTS as t, FindingsStore as u };
724
723
 
725
- //# sourceMappingURL=semantic-concept-judge-BypLt6Fw.js.map
724
+ //# sourceMappingURL=semantic-concept-judge-C0P1VTXD.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"semantic-concept-judge-BypLt6Fw.js","names":[],"sources":["../src/locked-jsonl-appender.ts","../src/analyst/findings-store.ts","../src/analyst/kinds/skill-usage.ts","../src/run-critic.ts","../src/semantic-concept-judge.ts"],"sourcesContent":["/**\n * LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary\n * payloads. The reference-replay store does the same thing for typed\n * `ReferenceReplayRun` rows; this is the generic version used by\n * `MutationTelemetry`, `TrialTelemetry`, and any other consumer that wants\n * append-only durable telemetry without rolling its own lock.\n *\n * Locks are per absolute file path (process-local). Cross-process\n * concurrency is NOT addressed — that's an fcntl/flock problem.\n */\n\nimport { appendFileSync, existsSync, mkdirSync } from 'node:fs'\nimport { dirname } from 'node:path'\nimport { Mutex } from './concurrency'\n\nconst mutexes = new Map<string, Mutex>()\n\nfunction getMutex(path: string): Mutex {\n let m = mutexes.get(path)\n if (!m) {\n m = new Mutex()\n mutexes.set(path, m)\n }\n return m\n}\n\nexport class LockedJsonlAppender {\n private readonly mutex: Mutex\n constructor(public readonly path: string) {\n this.mutex = getMutex(path)\n if (!existsSync(dirname(path))) {\n mkdirSync(dirname(path), { recursive: true })\n }\n }\n\n async append(entry: unknown): Promise<void> {\n const line = `${JSON.stringify(entry)}\\n`\n await this.mutex.runExclusive(() => {\n appendFileSync(this.path, line)\n })\n }\n}\n\n/** Reset all internal mutex state — tests only. */\nexport function resetLockedAppendersForTesting(): void {\n mutexes.clear()\n}\n","/**\n * FindingsStore — durable persistence for AnalystFinding rows + a diff\n * helper so we can answer \"what changed since the last run?\" without\n * recomputing analysts.\n *\n * On-disk shape is JSONL: one finding per line, append-only, locked via\n * LockedJsonlAppender. Operators get crash-safety (no partial JSON),\n * cheap reads (sequential parse), and trivial backup (rsync the file).\n *\n * Reads are non-locking: a reader sees a consistent snapshot of all\n * fully-written lines and skips an incomplete trailing line if the\n * writer is mid-append. Cross-process locking is intentionally out of\n * scope (see locked-jsonl-appender.ts).\n *\n * The store is run-scoped: callers pass `runId` on append and on load,\n * which keeps multi-run files cleanly partitioned. The `diffFindings`\n * helper compares two run-id sets using stable `finding_id` semantics —\n * the diff is the cross-run signal the regression dashboard renders.\n */\n\nimport { existsSync, readFileSync } from 'node:fs'\n\nimport { LockedJsonlAppender } from '../locked-jsonl-appender'\nimport type { AnalystFinding } from './types'\n\n/**\n * One persisted row. We attach `run_id` on disk so a single file can\n * hold multiple runs and the diff helper can query without re-walking\n * separate files.\n */\nexport interface PersistedFinding extends AnalystFinding {\n run_id: string\n}\n\nexport class FindingsStore {\n private readonly appender: LockedJsonlAppender\n\n constructor(public readonly path: string) {\n this.appender = new LockedJsonlAppender(path)\n }\n\n async append(runId: string, findings: AnalystFinding[]): Promise<void> {\n for (const f of findings) {\n const row: PersistedFinding = { ...f, run_id: runId }\n await this.appender.append(row)\n }\n }\n\n /** Load every persisted finding. Discards malformed trailing lines silently. */\n loadAll(): PersistedFinding[] {\n if (!existsSync(this.path)) return []\n const raw = readFileSync(this.path, 'utf8')\n if (!raw) return []\n const out: PersistedFinding[] = []\n for (const line of raw.split('\\n')) {\n if (!line) continue\n try {\n out.push(JSON.parse(line) as PersistedFinding)\n } catch {\n // Skip torn trailing line — the lock guarantees no torn lines\n // mid-file, only at EOF when a writer is in-flight.\n }\n }\n return out\n }\n\n /** Filter to a single run. */\n loadRun(runId: string): PersistedFinding[] {\n return this.loadAll().filter((r) => r.run_id === runId)\n }\n}\n\n// ── Cross-run diff ──────────────────────────────────────────────────\n\nexport interface FindingsDiff {\n /** New finding ids in `current` that weren't in `previous`. */\n appeared: PersistedFinding[]\n /** Finding ids in `previous` that aren't in `current`. */\n disappeared: PersistedFinding[]\n /** Same finding id present in both runs and unchanged per the materiality test. */\n persisted: PersistedFinding[]\n /**\n * Same finding id in both runs but at least one non-identity field\n * shifted per `DiffPolicy.isMaterial`. Reported as [previous, current].\n */\n changed: Array<{ previous: PersistedFinding; current: PersistedFinding }>\n}\n\nexport interface DiffPolicy {\n /**\n * Predicate that decides whether two findings (same finding_id) count\n * as a material change. Defaults to {@link defaultIsMaterial}: severity\n * shift, confidence Δ > 0.05, or evidence count change. Compliance /\n * perf consumers MAY supply a stricter predicate (e.g. rationale text\n * diff, metric Δ thresholds).\n */\n isMaterial?: (previous: AnalystFinding, current: AnalystFinding) => boolean\n}\n\n/**\n * Default materiality test. Deliberately narrow so LLM-reword churn\n * doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.\n */\nexport function defaultIsMaterial(a: AnalystFinding, b: AnalystFinding): boolean {\n if (a.severity !== b.severity) return true\n if (Math.abs((a.confidence ?? 0) - (b.confidence ?? 0)) > 0.05) return true\n if (a.evidence_refs.length !== b.evidence_refs.length) return true\n return false\n}\n\n/**\n * Diff two findings sets by stable finding_id. Callers typically load\n * the two run-id slices from the same store and pass them in.\n */\nexport function diffFindings(\n previous: PersistedFinding[],\n current: PersistedFinding[],\n policy: DiffPolicy = {},\n): FindingsDiff {\n const isMaterial = policy.isMaterial ?? defaultIsMaterial\n const prevById = new Map(previous.map((f) => [f.finding_id, f]))\n const curById = new Map(current.map((f) => [f.finding_id, f]))\n\n const appeared: PersistedFinding[] = []\n const disappeared: PersistedFinding[] = []\n const persisted: PersistedFinding[] = []\n const changed: FindingsDiff['changed'] = []\n\n for (const [id, cur] of curById) {\n const prev = prevById.get(id)\n if (!prev) {\n appeared.push(cur)\n continue\n }\n if (isMaterial(prev, cur)) {\n changed.push({ previous: prev, current: cur })\n } else {\n persisted.push(cur)\n }\n }\n for (const [id, prev] of prevById) {\n if (!curById.has(id)) disappeared.push(prev)\n }\n return { appeared, disappeared, persisted, changed }\n}\n","/**\n * Skill-usage analyst — a DETERMINISTIC `Analyst` over a Claude/Codex skill\n * library + its trace corpus. Unlike the trace-store kinds (failure-mode,\n * improvement, ...) this kind calls no LLM: it mines real usage and skill\n * structure and emits findings by rule.\n *\n * It exists because the naive \"Skill-tool invocation count\" lies low — it\n * misses orchestrated sub-dispatch (a leaf skill run BY /pursue or /governor\n * logs under the parent), slash-command entry, local-script bypass, and\n * on-disk artifacts. The 2026-05-30 skill audit found 39/53 skills at zero\n * direct invocations, yet only one was a genuine cut: the rest were\n * measurement-invisible or discovery-limited. This analyst encodes that\n * lesson as a multi-signal usage model so a cheap repeatable pass can keep\n * the library honest, and so the expensive audit workflow's verdicts can\n * GEPA-distill it toward agreement (see `gold/skill-verdicts.gold.jsonl`).\n *\n * Report-building (`buildSkillUsageReport`, an fs scan) is separated from\n * finding emission (`SkillUsageAnalyst.analyze`, pure) so the slow scan runs\n * once at the registry boundary and the rule logic stays unit-testable.\n */\n\nimport { type Dirent, existsSync, readdirSync, readFileSync, statSync } from 'node:fs'\nimport { join } from 'node:path'\nimport type { Analyst, AnalystContext, AnalystFinding, AnalystSeverity } from '../types'\nimport { computeFindingId } from '../types'\n\n// ── Input model ──────────────────────────────────────────────────────\n\nexport type SkillKind = 'public' | 'private'\n\n/** One skill's multi-signal usage + structure. All counts are deterministic. */\nexport interface SkillUsageRecord {\n name: string\n kind: SkillKind\n /** Absolute path to the skill's SKILL.md. */\n path: string\n lines: number\n /** `\"skill\":\"<name>\"` Skill-tool invocations across the trace corpus. */\n directInvocations: number\n /** `<command-name>/<name>` slash invocations across the trace corpus. */\n slashInvocations: number\n /** Sibling skills whose SKILL.md dispatches to this one (`/<name>`). Proxy\n * for orchestrated sub-dispatch the per-skill counter cannot see. */\n inboundRefs: number\n /** On-disk artifacts attributable to the skill (e.g. `.evolve/<name>/**`). */\n artifactCount: number\n /** Tangle-private reference count in the body (leak signal for public skills). */\n tanglePrivateRefs: number\n hasReferencesDir: boolean\n hasEvalsDir: boolean\n /** Body mentions `skill-runs.jsonl` (visible to /reflect + /governor). */\n logsRuns: boolean\n /** Description carries an explicit `Triggers:` clause / trigger phrases. */\n hasTriggerPhrases: boolean\n}\n\nexport interface SkillUsageReport {\n generatedFromTraces: number\n records: SkillUsageRecord[]\n}\n\nexport interface SkillUsageScanConfig {\n /** Dirs holding `*.jsonl` transcripts (Claude `~/.claude/projects`, Codex sessions). */\n transcriptDirs: string[]\n /** Skill roots to scan; each dir directly under `root` with a `SKILL.md` is a skill. */\n skillRoots: { root: string; kind: SkillKind }[]\n /** Roots scanned for `<root>/.evolve/<skill>` artifact dirs. */\n artifactRoots?: string[]\n /** Token-prefixed mappings: skill name → extra artifact subpaths under an artifactRoot\n * (e.g. reflect → `.evolve/reflections`). Catches non-eponymous artifact dirs. */\n artifactAliases?: Record<string, string[]>\n /** Cap files read per transcript dir (bounds a huge corpus); 0 = unbounded. */\n maxTranscriptsPerDir?: number\n}\n\n// ── Deterministic thresholds ─────────────────────────────────────────\n\n/** Anthropic's authoring guidance keeps SKILL.md short; past this with no\n * `references/` split the body burns context budget every session. */\nconst BLOAT_LINE_THRESHOLD = 300\n\nconst TANGLE_PRIVATE_RE =\n /\\b(cli-bridge|tangletools|ops-board|drew-gtr-pro|@tangle-network\\/|~\\/company|tangle\\.tools|gtm-agent)\\b|\\bkimi\\b|\\btcloud\\b/gi\nconst TRIGGER_RE = /triggers?\\s*[:-]/i\n\n// ── Report builder (fs scan — slow, runs once at the registry boundary) ──\n\nfunction listSkillDirs(root: string): { name: string; path: string }[] {\n if (!existsSync(root)) return []\n const out: { name: string; path: string }[] = []\n for (const entry of readdirSync(root, { withFileTypes: true })) {\n if (!entry.isDirectory() && !entry.isSymbolicLink()) continue\n const skillMd = join(root, entry.name, 'SKILL.md')\n if (existsSync(skillMd)) out.push({ name: entry.name, path: skillMd })\n }\n return out\n}\n\nfunction walkJsonl(dir: string, cap: number): string[] {\n if (!existsSync(dir)) return []\n const files: string[] = []\n const stack = [dir]\n while (stack.length) {\n const cur = stack.pop()!\n let entries: Dirent[]\n try {\n entries = readdirSync(cur, { withFileTypes: true })\n } catch {\n continue\n }\n for (const e of entries) {\n const full = join(cur, e.name)\n if (e.isDirectory()) stack.push(full)\n else if (e.name.endsWith('.jsonl')) {\n files.push(full)\n if (cap > 0 && files.length >= cap) return files\n }\n }\n }\n return files\n}\n\nfunction frontmatterDescription(body: string): string {\n const fm = /^---\\n([\\s\\S]*?)\\n---/.exec(body)\n const block = fm?.[1] ?? ''\n const m = /description:\\s*(.+)/i.exec(block)\n return m?.[1] ?? ''\n}\n\nfunction countArtifacts(roots: string[], name: string, aliases: string[]): number {\n let n = 0\n for (const root of roots) {\n const candidates = [join(root, '.evolve', name), ...aliases.map((a) => join(root, a))]\n for (const dir of candidates) {\n if (!existsSync(dir)) continue\n try {\n if (statSync(dir).isDirectory()) n += readdirSync(dir).length\n else n += 1\n } catch {\n /* unreadable — skip */\n }\n }\n }\n return n\n}\n\n/** Scan the corpus + skill roots into a {@link SkillUsageReport}. Deterministic. */\nexport function buildSkillUsageReport(config: SkillUsageScanConfig): SkillUsageReport {\n const skills = config.skillRoots.flatMap(({ root, kind }) =>\n listSkillDirs(root).map((s) => ({ ...s, kind })),\n )\n const names = skills.map((s) => s.name)\n\n // One pass over the corpus accumulating direct + slash counts per skill.\n const direct = new Map<string, number>(names.map((n) => [n, 0]))\n const slash = new Map<string, number>(names.map((n) => [n, 0]))\n const skillRe = /\"skill\"\\s*:\\s*\"([a-z0-9_:-]+)\"/g\n const cmdRe = /<command-name>\\/?([a-z0-9_:-]+)<\\/command-name>/g\n let transcripts = 0\n for (const dir of config.transcriptDirs) {\n for (const file of walkJsonl(dir, config.maxTranscriptsPerDir ?? 0)) {\n transcripts += 1\n let data: string\n try {\n data = readFileSync(file, 'utf8')\n } catch {\n continue\n }\n for (const m of data.matchAll(skillRe)) {\n const g = m[1]\n if (!g) continue\n const n = g.split(':').pop() ?? g\n const prev = direct.get(n)\n if (prev !== undefined) direct.set(n, prev + 1)\n }\n for (const m of data.matchAll(cmdRe)) {\n const g = m[1]\n if (g === undefined) continue\n const prev = slash.get(g)\n if (prev !== undefined) slash.set(g, prev + 1)\n }\n }\n }\n\n // Read each skill body once; compute structure + inbound refs across siblings.\n const bodies = new Map<string, string>()\n for (const s of skills) {\n try {\n bodies.set(s.name, readFileSync(s.path, 'utf8'))\n } catch {\n bodies.set(s.name, '')\n }\n }\n const inbound = new Map<string, number>(names.map((n) => [n, 0]))\n for (const target of names) {\n const ref = new RegExp(`/${target}\\\\b|\\\\[\\\\[${target}\\\\]\\\\]`)\n for (const s of skills) {\n if (s.name === target) continue\n if (ref.test(bodies.get(s.name) ?? '')) inbound.set(target, inbound.get(target)! + 1)\n }\n }\n\n const records: SkillUsageRecord[] = skills.map((s) => {\n const body = bodies.get(s.name) ?? ''\n const dir = s.path.replace(/\\/SKILL\\.md$/, '')\n return {\n name: s.name,\n kind: s.kind,\n path: s.path,\n lines: body ? body.split('\\n').length : 0,\n directInvocations: direct.get(s.name) ?? 0,\n slashInvocations: slash.get(s.name) ?? 0,\n inboundRefs: inbound.get(s.name) ?? 0,\n artifactCount: countArtifacts(\n config.artifactRoots ?? [],\n s.name,\n config.artifactAliases?.[s.name] ?? [],\n ),\n tanglePrivateRefs: (body.match(TANGLE_PRIVATE_RE) ?? []).length,\n hasReferencesDir: existsSync(join(dir, 'references')),\n hasEvalsDir: existsSync(join(dir, 'evals')),\n logsRuns: body.includes('skill-runs.jsonl'),\n hasTriggerPhrases: TRIGGER_RE.test(frontmatterDescription(body) || body.slice(0, 600)),\n }\n })\n return { generatedFromTraces: transcripts, records }\n}\n\n// ── Finding emission (pure — unit-testable, no LLM, no fs) ────────────\n\nconst ANALYST_ID = 'skill-usage'\n\nfunction finding(\n area: string,\n subject: string,\n claim: string,\n severity: AnalystSeverity,\n confidence: number,\n producedAt: string,\n recommended: string,\n evidenceUri: string,\n rationale?: string,\n): AnalystFinding {\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({ analyst_id: ANALYST_ID, area, subject, claim }),\n analyst_id: ANALYST_ID,\n produced_at: producedAt,\n severity,\n area,\n claim,\n rationale,\n evidence_refs: [{ kind: 'artifact', uri: evidenceUri }],\n recommended_action: recommended,\n confidence,\n subject,\n }\n}\n\n/** Pure rule pass over a report → findings. Exported for direct/unit use. */\nexport function emitSkillUsageFindings(\n report: SkillUsageReport,\n producedAt: string,\n): AnalystFinding[] {\n const out: AnalystFinding[] = []\n for (const r of report.records) {\n const directTotal = r.directInvocations + r.slashInvocations\n const trueUsage = directTotal + r.inboundRefs + r.artifactCount\n\n // 1. Dead: no usage signal of ANY kind. The only real deprecation candidate.\n if (trueUsage === 0) {\n out.push(\n finding(\n 'skill-usage',\n r.name,\n `Skill '${r.name}' has zero usage across all signals (direct, slash, inbound-refs, artifacts)`,\n 'high',\n 0.6,\n producedAt,\n 'Confirm the skill covers a real recurring job; if not, deprecate. Zero true usage is the only deterministic deprecation candidate.',\n r.path,\n 'No Skill-tool call, no slash invocation, no sibling dispatches to it, and no on-disk artifacts.',\n ),\n )\n } else if (directTotal === 0 && r.inboundRefs + r.artifactCount > 0) {\n // 2. Measurement-invisible: real use via orchestration/artifacts, never invoked directly.\n out.push(\n finding(\n 'skill-usage',\n r.name,\n `Skill '${r.name}' shows 0 direct invocations but is used via orchestration/artifacts (inbound=${r.inboundRefs}, artifacts=${r.artifactCount})`,\n 'info',\n 0.8,\n producedAt,\n 'Do NOT treat as unused — usage is real but logged under parent skills or on disk. Strengthen direct-invocation discovery only if direct use is desired.',\n r.path,\n 'The Skill-tool counter undercounts orchestrated/chained leaf skills.',\n ),\n )\n }\n\n // 3. Discovery gap: low direct use AND weak trigger surface.\n if (directTotal <= 2 && !r.hasTriggerPhrases) {\n out.push(\n finding(\n 'discoverability',\n r.name,\n `Skill '${r.name}' is rarely invoked directly and its description has no explicit trigger phrases`,\n 'medium',\n 0.7,\n producedAt,\n 'Add a `Triggers:` clause with verbatim user phrases to the frontmatter description so the model auto-invokes it.',\n r.path,\n ),\n )\n }\n\n // 4. Public-repo leak.\n if (r.kind === 'public' && r.tanglePrivateRefs > 0) {\n out.push(\n finding(\n 'safety',\n r.name,\n `Public skill '${r.name}' carries ${r.tanglePrivateRefs} Tangle-private reference(s)`,\n 'high',\n 0.75,\n producedAt,\n 'Sanitize incidental internal refs (cli-bridge/kimi/tcloud/~company/private repos) or relocate to a private repo. Verify @tangle-network/* refs are to PUBLISHED packages before treating as a leak.',\n r.path,\n ),\n )\n }\n\n // 5. Bloat / no progressive disclosure.\n if (r.lines > BLOAT_LINE_THRESHOLD && !r.hasReferencesDir) {\n out.push(\n finding(\n 'maintainability',\n r.name,\n `Skill '${r.name}' is ${r.lines} lines with no references/ split (progressive disclosure)`,\n 'medium',\n 0.8,\n producedAt,\n `Split detail into references/ loaded on demand; keep SKILL.md a short overview. ${r.lines} lines load into every session's context budget.`,\n r.path,\n ),\n )\n }\n\n // 6. No evals (Anthropic's \">=3 evals before docs\" rule).\n if (!r.hasEvalsDir) {\n out.push(\n finding(\n 'data-quality',\n r.name,\n `Skill '${r.name}' ships no evals/`,\n 'low',\n 0.6,\n producedAt,\n 'Add evals/evals.json with >=3 scenarios proving the skill beats baseline; gives regression coverage.',\n r.path,\n ),\n )\n }\n\n // 7. No run logging → invisible to /reflect and /governor.\n if (!r.logsRuns) {\n out.push(\n finding(\n 'observability',\n r.name,\n `Skill '${r.name}' never appends to .evolve/skill-runs.jsonl`,\n 'low',\n 0.55,\n producedAt,\n 'Append one run line to .evolve/skill-runs.jsonl on completion, or declare it a non-logging leaf, so the self-improvement loop can see it ran.',\n r.path,\n ),\n )\n }\n }\n return out\n}\n\n// ── The Analyst ──────────────────────────────────────────────────────\n\nexport class SkillUsageAnalyst implements Analyst<SkillUsageReport> {\n readonly id = ANALYST_ID\n readonly description =\n 'Deterministic multi-signal skill-usage analysis: flags dead skills, measurement-invisible (orchestrated) usage, discovery gaps, public-repo leaks, bloat, missing evals, and missing run-logging.'\n readonly inputKind = 'custom' as const\n readonly cost = { kind: 'deterministic' as const, est_usd_per_run: 0 }\n readonly version = '1.0.0'\n\n async analyze(input: SkillUsageReport, ctx: AnalystContext): Promise<AnalystFinding[]> {\n const producedAt = ctx.tags?.producedAt ?? new Date().toISOString()\n ctx.log?.(\n `skill-usage: ${input.records.length} skills over ${input.generatedFromTraces} transcripts`,\n )\n return emitSkillUsageFindings(input, producedAt)\n }\n}\n\nexport const SKILL_USAGE_ANALYST = new SkillUsageAnalyst()\n","import { NotFoundError } from './errors'\nimport { aggregateRunScore, clamp01, type RunScore, type RunScoreWeights } from './run-score'\nimport type { Artifact, BudgetLedgerEntry, Run, Span, TraceEvent, TraceStore } from './trace'\n\nexport interface RunTrace {\n run: Run\n spans: Span[]\n events: TraceEvent[]\n artifacts: Artifact[]\n budget: BudgetLedgerEntry[]\n}\n\nexport interface RunCriticOptions {\n weights?: Partial<RunScoreWeights>\n driftPatterns?: RegExp[]\n}\n\nconst DEFAULT_DRIFT_PATTERNS = [\n /https?:\\/\\//i,\n /\\btitle:\\s/i,\n /\\bsummary:\\s/i,\n /\\burl:\\s/i,\n /\\bnpm package usage\\b/i,\n /\\bnews\\b/i,\n]\n\nexport class RunCritic {\n private readonly weights?: Partial<RunScoreWeights>\n private readonly driftPatterns: RegExp[]\n\n constructor(options: RunCriticOptions = {}) {\n this.weights = options.weights\n this.driftPatterns = options.driftPatterns ?? DEFAULT_DRIFT_PATTERNS\n }\n\n async score(store: TraceStore, runId: string): Promise<RunScore> {\n const run = await store.getRun(runId)\n if (!run) throw new NotFoundError(`run ${runId} not found`)\n const [spans, events, artifacts, budget] = await Promise.all([\n store.spans({ runId }),\n store.events({ runId }),\n store.artifacts(runId),\n store.budget(runId),\n ])\n return this.scoreTrace({ run, spans, events, artifacts, budget })\n }\n\n scoreTrace(trace: RunTrace): RunScore {\n const notes: string[] = []\n const llmSpans = trace.spans.filter(\n (s): s is Extract<Span, { kind: 'llm' }> => s.kind === 'llm',\n )\n const toolSpans = trace.spans.filter(\n (s): s is Extract<Span, { kind: 'tool' }> => s.kind === 'tool',\n )\n const judgeSpans = trace.spans.filter(\n (s): s is Extract<Span, { kind: 'judge' }> => s.kind === 'judge',\n )\n const sandboxSpans = trace.spans.filter(\n (s): s is Extract<Span, { kind: 'sandbox' }> => s.kind === 'sandbox',\n )\n const finalGateSpans = judgeSpans.filter(\n (span) => span.dimension === 'final_gate' || span.attributes?.finalGate === true,\n )\n\n const success =\n trace.run.outcome?.pass === true ? 1 : trace.run.status === 'completed' ? 0.5 : 0\n if (!success) notes.push('run did not complete with pass=true')\n\n const judgeAverage = judgeSpans.length\n ? judgeSpans.reduce((sum, span) => sum + normalizeJudgeScore(span.score), 0) /\n judgeSpans.length\n : undefined\n const outcomeScore =\n typeof trace.run.outcome?.score === 'number'\n ? clamp01(\n trace.run.outcome.score > 1 ? trace.run.outcome.score / 100 : trace.run.outcome.score,\n )\n : undefined\n const goalProgress = outcomeScore ?? judgeAverage ?? success\n\n const successfulTools = toolSpans.filter((span) => span.status !== 'error').length\n const toolUseQuality = toolSpans.length === 0 ? 0 : successfulTools / toolSpans.length\n if (toolSpans.length === 0) notes.push('no tool spans recorded')\n\n const patchEvidence =\n trace.artifacts.length +\n toolSpans.filter((span) => /write|edit|patch|apply/i.test(span.toolName)).length\n const patchQuality = patchEvidence > 0 ? clamp01(patchEvidence / 4) : 0\n if (!patchQuality) notes.push('no artifact or edit evidence recorded')\n\n const sandboxTests = sandboxSpans.filter(\n (span) => typeof span.testsTotal === 'number' && span.testsTotal > 0,\n )\n const testReality = sandboxTests.length\n ? sandboxTests.reduce(\n (sum, span) => sum + (span.testsPassed ?? 0) / Math.max(1, span.testsTotal ?? 1),\n 0,\n ) / sandboxTests.length\n : toolSpans.some((span) =>\n /\\btest|vitest|pytest|jest|build|tsc\\b/i.test(JSON.stringify(span.args)),\n )\n ? 0.4\n : 0\n if (!testReality) notes.push('no real test/build evidence recorded')\n\n const blockerSpans = judgeSpans.filter((span) => isBlockingJudge(span))\n const finalGateBlockers = finalGateSpans.filter((span) => isBlockingJudge(span))\n const finalGate = finalGateSpans.length ? (finalGateBlockers.length ? 0 : 1) : success\n if (finalGateBlockers.length)\n notes.push(`final gate blocked by ${finalGateBlockers.length} reviewer(s)`)\n else if (!finalGateSpans.length) notes.push('no final gate judgment recorded')\n\n const reviewerBlockers = judgeSpans.length ? blockerSpans.length / judgeSpans.length : 0\n if (reviewerBlockers) notes.push(`detected ${blockerSpans.length} blocking reviewer signal(s)`)\n\n const positiveGroundingSignals =\n patchEvidence +\n sandboxSpans.length +\n llmSpans.filter((span) => looksRepoGrounded(span.output ?? '')).length\n const driftSignals =\n llmSpans.filter((span) => this.isDrift(span.output ?? '')).length +\n trace.events.filter((event) => this.isDrift(JSON.stringify(event.payload))).length\n const repoGroundedness =\n positiveGroundingSignals + driftSignals === 0\n ? 0\n : positiveGroundingSignals / (positiveGroundingSignals + driftSignals)\n const driftPenalty =\n positiveGroundingSignals + driftSignals === 0\n ? 0\n : driftSignals / (positiveGroundingSignals + driftSignals)\n if (driftSignals > 0) notes.push(`detected ${driftSignals} drift signal(s)`)\n\n const costUsd = trace.budget.length\n ? Math.max(\n ...trace.budget\n .filter((entry: BudgetLedgerEntry) => entry.dimension === 'usd')\n .map((entry: BudgetLedgerEntry) => entry.consumed),\n 0,\n )\n : llmSpans.reduce((sum, span) => sum + (span.costUsd ?? 0), 0)\n const wallSeconds =\n trace.run.endedAt && trace.run.startedAt\n ? Math.max(0, (trace.run.endedAt - trace.run.startedAt) / 1000)\n : 0\n\n return {\n success,\n goalProgress,\n repoGroundedness,\n driftPenalty,\n toolUseQuality,\n patchQuality,\n testReality,\n finalGate,\n reviewerBlockers,\n costUsd,\n wallSeconds,\n notes,\n }\n }\n\n rank(score: RunScore): number {\n return aggregateRunScore(score, this.weights)\n }\n\n private isDrift(text: string): boolean {\n return this.driftPatterns.some((pattern) => pattern.test(text))\n }\n}\n\nfunction normalizeJudgeScore(score: number): number {\n return score > 1 ? clamp01(score / 10) : clamp01(score)\n}\n\nfunction looksRepoGrounded(text: string): boolean {\n return /(?:src\\/|tests?\\/|package\\.json|tsconfig|\\.ts\\b|\\.tsx\\b|git status|pnpm |npm |vitest|pytest|jest)/i.test(\n text,\n )\n}\n\nfunction isBlockingJudge(span: Extract<Span, { kind: 'judge' }>): boolean {\n return (\n span.attributes?.blocking === true ||\n span.attributes?.verdict === 'BLOCKING' ||\n positiveNumber(span.attributes?.blockingFindings) ||\n positiveNumber(span.attributes?.highFindings) ||\n span.score <= 2\n )\n}\n\nfunction positiveNumber(value: unknown): boolean {\n return typeof value === 'number' && value > 0\n}\n","/**\n * Semantic concept judge — \"does the built artifact actually implement\n * the features the user asked for?\"\n *\n * Distinct from the domain/code/coherence judges in `judges.ts`:\n * - those judges score free-form conversational agent outputs along\n * quality dimensions (accuracy, depth, etc.)\n * - this judge scores a *built artifact* (served HTML + source files)\n * against an explicit list of expected concepts, returning per-concept\n * {present, score 0-10, evidence, severity}.\n *\n * The judge is strict about distinguishing (a) a working implementation\n * from (b) a keyword-present stub. \"// TODO: mint button\" is NOT present.\n * Only real, functional, wired-up code counts.\n *\n * Use via {@link createSemanticConceptJudge} or directly via\n * {@link runSemanticConceptJudge}. Soft-fails (available=false) on LLM\n * or JSON-parse errors so the caller can treat that as \"layer skipped\"\n * rather than \"layer failed\" in a multi-layer pipeline.\n */\n\nimport { CostLedger, type CostLedgerHandle, type CostReceipt } from './cost-ledger'\nimport {\n callLlmJson,\n costReceiptFromLlm,\n costReceiptFromLlmError,\n type LlmCallRequest,\n type LlmClientOptions,\n maximumChargeForLlmRequest,\n} from './llm-client'\nimport type { Severity } from './multi-layer-verifier'\n\n// ─── Types ──────────────────────────────────────────────────────────────\n\n/**\n * Implementation complexity class for weighted scoring.\n *\n * - `render` (default): the concept is a UI surface that displays static\n * data — render a list, show a counter, lay out a button. Single-file\n * work, no external integration.\n * - `integrate`: the concept requires wiring a real external system —\n * wallet connect (wagmi + RainbowKit + chain config), payment provider\n * (Stripe Elements + intent + webhook), an API client with auth.\n * Multi-file, library-knowledge, runtime correctness matters.\n * - `compute`: the concept requires algorithmic work — solver, simulator,\n * constraint propagation, ML inference. Correctness > UI polish.\n *\n * Default weights (when applied via `weightConcepts: 'complexity'`):\n * render=1.0, integrate=2.0, compute=2.5\n *\n * Cross-vertical scoring without complexity weighting silently inflates\n * the rate of UI-heavy verticals (healthcare, fintech dashboards) vs\n * integration-heavy verticals (DeFi, wallets) — all concepts treated\n * equally even though the agent does 2-3x the work for `integrate`.\n */\nexport type ConceptComplexity = 'render' | 'integrate' | 'compute'\n\nexport interface ConceptSpec {\n name: string\n /** Short hints that help the judge; not used for matching. */\n keywords?: string[]\n /** Optional explicit weight; default 1.0. Overrides complexity-derived weight. */\n weight?: number\n /** Implementation complexity class. Default `render`. */\n complexity?: ConceptComplexity\n}\n\nexport interface ConceptFinding {\n concept: string\n present: boolean\n /** 0..10. 10 = production-ready; 7 = functional thin; 4 = partial; 0 = absent. */\n score: number\n evidence: string\n severity: Severity\n}\n\nexport interface SemanticConceptJudgeInput {\n /** Full natural-language prompt the agent was handed. */\n userRequest: string\n /** Rendered HTML the preview returns (UI artifacts). Optional. */\n servedHtml?: string\n /** Top-level source files from the agent's workdir. */\n sourceFiles: Array<{ path: string; content: string }>\n /** The expected concept list. */\n expectedConcepts: ConceptSpec[]\n /** Free-form metadata (id, difficulty) to inject into the prompt. */\n artifactLabel?: string\n artifactDescription?: string\n}\n\nexport interface SemanticConceptJudgeResult {\n kind: 'semantic-concept'\n version: string\n /** Normalized 0..1 score — mean of per-concept scores / 10. */\n score: number\n presentCount: number\n totalCount: number\n findings: ConceptFinding[]\n summary: string\n durationMs: number\n costUsd: number | null\n /** False on LLM/JSON error — treat as \"skipped / unable to judge\" in pipelines. */\n available: boolean\n error?: string\n}\n\n/**\n * Score-aggregation strategy. `mean` averages 0-10 scores uniformly.\n * `complexity` applies the default weight table (render=1, integrate=2,\n * compute=2.5) unless a concept has an explicit `weight`. `explicit`\n * honors only `weight` (defaulting to 1 for unspecified).\n */\nexport type ConceptWeightStrategy = 'mean' | 'complexity' | 'explicit'\n\nexport const DEFAULT_COMPLEXITY_WEIGHTS: Record<ConceptComplexity, number> = {\n render: 1.0,\n integrate: 2.0,\n compute: 2.5,\n}\n\nexport interface SemanticConceptJudgeOptions {\n /** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */\n model?: string\n /** Per-call timeout. Default 300s. */\n timeoutMs?: number\n /** Provider-enforced output limit. Default 16000. */\n maxTokens?: number\n /** Pipeline budget for the prompt (source blob truncation). Default 45000. */\n maxSourceChars?: number\n /** Per-file cap before inclusion. Default 20000. */\n maxPerFileChars?: number\n /** HTML cap. Default 30000. */\n maxHtmlChars?: number\n /** LlmClient config (baseUrl, apiKey, authHeader, …). */\n llm?: LlmClientOptions\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /**\n * Score aggregation strategy. Default `mean` — uniform average across\n * concepts. Cross-vertical comparisons should use `complexity` to\n * neutralize the integrate-vs-render asymmetry.\n */\n weightConcepts?: ConceptWeightStrategy\n /** Override the default complexity → weight table. */\n complexityWeights?: Partial<Record<ConceptComplexity, number>>\n}\n\n// ─── Prompt assembly ────────────────────────────────────────────────────\n\nexport const SEMANTIC_CONCEPT_JUDGE_VERSION = 'semantic-concept-judge-v1-2026-04-24'\n\nconst DEFAULT_MAX_SOURCE = 45_000\nconst DEFAULT_MAX_HTML = 30_000\nconst DEFAULT_MAX_PER_FILE = 20_000\nconst DEFAULT_TIMEOUT = 300_000\nconst DEFAULT_MAX_TOKENS = 16_000\nconst DEFAULT_MODEL = 'claude-sonnet-4-6'\n\nconst SEMANTIC_SCHEMA = {\n type: 'object',\n additionalProperties: false,\n required: ['summary', 'concepts'],\n properties: {\n summary: { type: 'string', minLength: 20, maxLength: 600 },\n concepts: {\n type: 'array',\n minItems: 1,\n items: {\n type: 'object',\n additionalProperties: false,\n required: ['concept', 'present', 'score', 'evidence', 'severity'],\n properties: {\n concept: { type: 'string', minLength: 1, maxLength: 120 },\n present: { type: 'boolean' },\n score: { type: 'number', minimum: 0, maximum: 10 },\n evidence: { type: 'string', minLength: 5, maxLength: 400 },\n severity: { type: 'string', enum: ['critical', 'major', 'minor', 'info'] },\n },\n },\n },\n },\n}\n\nfunction truncate(body: string, cap: number, label: string): string {\n if (body.length <= cap) return body\n return `${body.slice(0, cap)}\\n… [truncated ${body.length - cap} chars of ${label}]`\n}\n\nfunction buildPrompt(\n input: SemanticConceptJudgeInput,\n opts: Required<SemanticConceptJudgeOptions>,\n): string {\n const sourceBlob = input.sourceFiles\n .filter((f) => f.content.length <= opts.maxPerFileChars)\n .map((f) => `--- FILE: ${f.path} ---\\n${f.content}`)\n .join('\\n\\n')\n\n const html = input.servedHtml ?? ''\n\n return `You are a strict code-review judge evaluating whether an agent's 0-to-1 build actually implements the features the user asked for.\n\nYou MUST distinguish:\n (a) WORKING code that implements the concept (rendered UI, wired handler, real API call),\n (b) KEYWORD-PRESENT stub (comments mentioning the concept, variable names, TODOs),\n (c) ABSENT (concept nowhere).\n\nA comment like \"// TODO: add mint button\" is NOT present — score 2-3. Only count a concept as present if there is real functional code: a rendered component, a call handler wired to state or a network call, a computed value actually used.\n\nUSER REQUEST (what the agent was asked to build):\n${input.userRequest}\n\n${input.artifactLabel ? `ARTIFACT METADATA:\\n name: ${input.artifactLabel}\\n description: ${input.artifactDescription ?? ''}\\n\\n` : ''}EXPECTED CONCEPTS (each must be graded independently):\n${input.expectedConcepts\n .map(\n (c, i) =>\n ` ${i + 1}. \"${c.name}\"${c.keywords?.length ? ` — hints: [${c.keywords.slice(0, 6).join(' | ')}]` : ''}`,\n )\n .join('\\n')}\n\n${html ? `SERVED HTML (what the preview returns when hit):\\n${truncate(html, opts.maxHtmlChars, 'HTML')}\\n\\n` : ''}SOURCE FILES (the agent's workdir):\n${truncate(sourceBlob, opts.maxSourceChars, 'source')}\n\nFor EACH concept, return:\n - concept: the concept name as given (match exactly)\n - present: boolean — does a working implementation exist?\n - score: 0-10 — 10 = production-ready; 7 = functional but thin; 4 = partial/stubbed; 2 = keyword-only comment; 0 = absent\n - evidence: cite \"<file>:<line>\" or \"served-html:<selector>\" pointing at the strongest supporting code. If the concept is absent or stubbed, explain what's missing.\n - severity:\n \"info\" when present: true AND score >= 7\n \"minor\" when present: true AND 4 <= score < 7\n \"major\" when present: false OR score < 4\n \"critical\" when the concept is not only absent but a core user flow depends on it\n\nAlso produce a \"summary\" (one sentence, 20-600 chars): overall verdict on whether this is a shippable implementation of the user request vs a keyword-dense placeholder.\n\nBE SKEPTICAL. Keyword matching already passed — your job is to catch what keyword matching misses. If the agent shipped a working build, say so. If it shipped a stub, say so. Don't grade on effort.\n\nReturn STRICT JSON. No prose outside the JSON.`\n}\n\n// ─── Runner ─────────────────────────────────────────────────────────────\n\n/**\n * Run the semantic concept judge. Soft-fails to available=false on\n * LLM/JSON errors — callers in a MultiLayerVerifier pipeline can treat\n * that as \"skip\" rather than \"fail.\"\n */\nexport async function runSemanticConceptJudge(\n input: SemanticConceptJudgeInput,\n options: SemanticConceptJudgeOptions = {},\n): Promise<SemanticConceptJudgeResult> {\n const start = Date.now()\n const totalCount = input.expectedConcepts.length\n\n if (totalCount === 0) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount: 0,\n findings: [],\n summary: 'no expected concepts declared',\n durationMs: 0,\n costUsd: null,\n available: false,\n error: 'no expected concepts declared',\n }\n }\n\n const opts: Required<SemanticConceptJudgeOptions> = {\n model: options.model ?? DEFAULT_MODEL,\n timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,\n maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,\n maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,\n maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,\n maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,\n llm: options.llm ?? {},\n costLedger: options.costLedger ?? new CostLedger(),\n costPhase: options.costPhase ?? 'judge.semantic-concept',\n costTags: options.costTags ?? {},\n signal: options.signal ?? new AbortController().signal,\n weightConcepts: options.weightConcepts ?? 'mean',\n complexityWeights: { ...DEFAULT_COMPLEXITY_WEIGHTS, ...(options.complexityWeights ?? {}) },\n }\n\n // Build a name → weight map for aggregation. Mean strategy keeps every\n // weight at 1 (uniform average). Complexity strategy reads the table\n // and lets an explicit `weight` override. Explicit strategy uses ONLY\n // the spec's `weight` (defaulting to 1).\n const weightForConcept = (spec: ConceptSpec): number => {\n if (opts.weightConcepts === 'mean') return 1\n if (spec.weight != null) return spec.weight\n if (opts.weightConcepts === 'complexity') {\n return opts.complexityWeights[spec.complexity ?? 'render'] ?? 1\n }\n return 1\n }\n const weightByName = new Map<string, number>(\n input.expectedConcepts.map((c) => [c.name, weightForConcept(c)]),\n )\n\n let receipt: CostReceipt | undefined\n try {\n const request = {\n model: opts.model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You are a strict code-review judge. Return strict JSON only. No prose outside the JSON. A keyword in a comment is NOT a working implementation.',\n },\n { role: 'user' as const, content: buildPrompt(input, opts) },\n ],\n jsonSchema: { name: 'semantic_concept_judge', schema: SEMANTIC_SCHEMA },\n temperature: 0,\n maxTokens: opts.maxTokens,\n timeoutMs: opts.timeoutMs,\n } satisfies LlmCallRequest\n const paid = await opts.costLedger.runPaidCall({\n channel: 'judge',\n phase: opts.costPhase,\n actor: 'semantic-concept',\n model: opts.model,\n ...(Object.keys(opts.costTags).length > 0 ? { tags: opts.costTags } : {}),\n maximumCharge: maximumChargeForLlmRequest(request, opts.llm),\n signal: opts.signal,\n execute: (signal, callId) =>\n callLlmJson<{ summary: string; concepts: ConceptFinding[] }>(request, {\n ...opts.llm,\n signal,\n idempotencyKey: callId,\n }),\n receipt: ({ result }) => costReceiptFromLlm(result),\n receiptFromError: costReceiptFromLlmError,\n })\n receipt = paid.receipt\n if (!paid.succeeded) throw paid.error\n const { value } = paid.value\n\n if (!value?.concepts || !Array.isArray(value.concepts)) {\n throw new Error('judge returned malformed response — expected array under \"concepts\"')\n }\n\n const findings: ConceptFinding[] = value.concepts.map((c) => ({\n concept: String(c.concept),\n present: Boolean(c.present),\n score: Math.max(0, Math.min(10, Number(c.score ?? 0))),\n evidence: String(c.evidence ?? ''),\n severity: (['critical', 'major', 'minor', 'info'] as const).includes(c.severity)\n ? c.severity\n : 'info',\n }))\n\n const presentCount = findings.filter((f) => f.present && f.score >= 7).length\n let weightSum = 0\n let weightedScoreSum = 0\n for (const f of findings) {\n const w = weightByName.get(f.concept) ?? 1\n weightSum += w\n weightedScoreSum += w * f.score\n }\n const scoreAvg =\n weightSum > 0\n ? weightedScoreSum / weightSum\n : findings.reduce((a, f) => a + f.score, 0) / Math.max(1, findings.length)\n\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: Number((scoreAvg / 10).toFixed(3)),\n presentCount,\n totalCount,\n findings,\n summary: String(value.summary ?? ''),\n durationMs: Date.now() - start,\n costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,\n available: true,\n }\n } catch (err) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount,\n findings: [],\n summary: '',\n durationMs: Date.now() - start,\n costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,\n available: false,\n error: err instanceof Error ? err.message : String(err),\n }\n }\n}\n\n/**\n * Factory: pin LLM options once, return a closure that accepts inputs.\n * Convenient for pipelines that want to share a single LlmClient config.\n */\nexport function createSemanticConceptJudge(\n options: SemanticConceptJudgeOptions = {},\n): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult> {\n return (input) => runSemanticConceptJudge(input, options)\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;AAeA,MAAM,0BAAU,IAAI,IAAmB;AAEvC,SAAS,SAAS,MAAqB;CACrC,IAAI,IAAI,QAAQ,IAAI,IAAI;CACxB,IAAI,CAAC,GAAG;EACN,IAAI,IAAI,MAAM;EACd,QAAQ,IAAI,MAAM,CAAC;CACrB;CACA,OAAO;AACT;AAEA,IAAa,sBAAb,MAAiC;CAEH;CAD5B;CACA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,QAAQ,SAAS,IAAI;EAC1B,IAAI,CAAC,WAAW,QAAQ,IAAI,CAAC,GAC3B,UAAU,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAEhD;CAEA,MAAM,OAAO,OAA+B;EAC1C,MAAM,OAAO,GAAG,KAAK,UAAU,KAAK,EAAE;EACtC,MAAM,KAAK,MAAM,mBAAmB;GAClC,eAAe,KAAK,MAAM,IAAI;EAChC,CAAC;CACH;AACF;;;;;;;;;;;;;;;;;;;;;;ACPA,IAAa,gBAAb,MAA2B;CAGG;CAF5B;CAEA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,WAAW,IAAI,oBAAoB,IAAI;CAC9C;CAEA,MAAM,OAAO,OAAe,UAA2C;EACrE,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,MAAwB;IAAE,GAAG;IAAG,QAAQ;GAAM;GACpD,MAAM,KAAK,SAAS,OAAO,GAAG;EAChC;CACF;;CAGA,UAA8B;EAC5B,IAAI,CAAC,WAAW,KAAK,IAAI,GAAG,OAAO,CAAC;EACpC,MAAM,MAAM,aAAa,KAAK,MAAM,MAAM;EAC1C,IAAI,CAAC,KAAK,OAAO,CAAC;EAClB,MAAM,MAA0B,CAAC;EACjC,KAAK,MAAM,QAAQ,IAAI,MAAM,IAAI,GAAG;GAClC,IAAI,CAAC,MAAM;GACX,IAAI;IACF,IAAI,KAAK,KAAK,MAAM,IAAI,CAAqB;GAC/C,QAAQ,CAGR;EACF;EACA,OAAO;CACT;;CAGA,QAAQ,OAAmC;EACzC,OAAO,KAAK,QAAQ,CAAC,CAAC,QAAQ,MAAM,EAAE,WAAW,KAAK;CACxD;AACF;;;;;AAiCA,SAAgB,kBAAkB,GAAmB,GAA4B;CAC/E,IAAI,EAAE,aAAa,EAAE,UAAU,OAAO;CACtC,IAAI,KAAK,KAAK,EAAE,cAAc,MAAM,EAAE,cAAc,EAAE,IAAI,KAAM,OAAO;CACvE,IAAI,EAAE,cAAc,WAAW,EAAE,cAAc,QAAQ,OAAO;CAC9D,OAAO;AACT;;;;;AAMA,SAAgB,aACd,UACA,SACA,SAAqB,CAAC,GACR;CACd,MAAM,aAAa,OAAO,cAAc;CACxC,MAAM,WAAW,IAAI,IAAI,SAAS,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAC/D,MAAM,UAAU,IAAI,IAAI,QAAQ,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAE7D,MAAM,WAA+B,CAAC;CACtC,MAAM,cAAkC,CAAC;CACzC,MAAM,YAAgC,CAAC;CACvC,MAAM,UAAmC,CAAC;CAE1C,KAAK,MAAM,CAAC,IAAI,QAAQ,SAAS;EAC/B,MAAM,OAAO,SAAS,IAAI,EAAE;EAC5B,IAAI,CAAC,MAAM;GACT,SAAS,KAAK,GAAG;GACjB;EACF;EACA,IAAI,WAAW,MAAM,GAAG,GACtB,QAAQ,KAAK;GAAE,UAAU;GAAM,SAAS;EAAI,CAAC;OAE7C,UAAU,KAAK,GAAG;CAEtB;CACA,KAAK,MAAM,CAAC,IAAI,SAAS,UACvB,IAAI,CAAC,QAAQ,IAAI,EAAE,GAAG,YAAY,KAAK,IAAI;CAE7C,OAAO;EAAE;EAAU;EAAa;EAAW;CAAQ;AACrD;;;;;;;;;;;;;;;;;;;;;;;;;ACjEA,MAAM,uBAAuB;AAE7B,MAAM,oBACJ;AACF,MAAM,aAAa;AAInB,SAAS,cAAc,MAAgD;CACrE,IAAI,CAAC,WAAW,IAAI,GAAG,OAAO,CAAC;CAC/B,MAAM,MAAwC,CAAC;CAC/C,KAAK,MAAM,SAAS,YAAY,MAAM,EAAE,eAAe,KAAK,CAAC,GAAG;EAC9D,IAAI,CAAC,MAAM,YAAY,KAAK,CAAC,MAAM,eAAe,GAAG;EACrD,MAAM,UAAU,KAAK,MAAM,MAAM,MAAM,UAAU;EACjD,IAAI,WAAW,OAAO,GAAG,IAAI,KAAK;GAAE,MAAM,MAAM;GAAM,MAAM;EAAQ,CAAC;CACvE;CACA,OAAO;AACT;AAEA,SAAS,UAAU,KAAa,KAAuB;CACrD,IAAI,CAAC,WAAW,GAAG,GAAG,OAAO,CAAC;CAC9B,MAAM,QAAkB,CAAC;CACzB,MAAM,QAAQ,CAAC,GAAG;CAClB,OAAO,MAAM,QAAQ;EACnB,MAAM,MAAM,MAAM,IAAI;EACtB,IAAI;EACJ,IAAI;GACF,UAAU,YAAY,KAAK,EAAE,eAAe,KAAK,CAAC;EACpD,QAAQ;GACN;EACF;EACA,KAAK,MAAM,KAAK,SAAS;GACvB,MAAM,OAAO,KAAK,KAAK,EAAE,IAAI;GAC7B,IAAI,EAAE,YAAY,GAAG,MAAM,KAAK,IAAI;QAC/B,IAAI,EAAE,KAAK,SAAS,QAAQ,GAAG;IAClC,MAAM,KAAK,IAAI;IACf,IAAI,MAAM,KAAK,MAAM,UAAU,KAAK,OAAO;GAC7C;EACF;CACF;CACA,OAAO;AACT;AAEA,SAAS,uBAAuB,MAAsB;CAEpD,MAAM,QADK,wBAAwB,KAAK,IACzB,CAAC,GAAG,MAAM;CAEzB,OADU,uBAAuB,KAAK,KAC/B,CAAC,GAAG,MAAM;AACnB;AAEA,SAAS,eAAe,OAAiB,MAAc,SAA2B;CAChF,IAAI,IAAI;CACR,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,aAAa,CAAC,KAAK,MAAM,WAAW,IAAI,GAAG,GAAG,QAAQ,KAAK,MAAM,KAAK,MAAM,CAAC,CAAC,CAAC;EACrF,KAAK,MAAM,OAAO,YAAY;GAC5B,IAAI,CAAC,WAAW,GAAG,GAAG;GACtB,IAAI;IACF,IAAI,SAAS,GAAG,CAAC,CAAC,YAAY,GAAG,KAAK,YAAY,GAAG,CAAC,CAAC;SAClD,KAAK;GACZ,QAAQ,CAER;EACF;CACF;CACA,OAAO;AACT;;AAGA,SAAgB,sBAAsB,QAAgD;CACpF,MAAM,SAAS,OAAO,WAAW,SAAS,EAAE,MAAM,WAChD,cAAc,IAAI,CAAC,CAAC,KAAK,OAAO;EAAE,GAAG;EAAG;CAAK,EAAE,CACjD;CACA,MAAM,QAAQ,OAAO,KAAK,MAAM,EAAE,IAAI;CAGtC,MAAM,SAAS,IAAI,IAAoB,MAAM,KAAK,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;CAC/D,MAAM,QAAQ,IAAI,IAAoB,MAAM,KAAK,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;CAC9D,MAAM,UAAU;CAChB,MAAM,QAAQ;CACd,IAAI,cAAc;CAClB,KAAK,MAAM,OAAO,OAAO,gBACvB,KAAK,MAAM,QAAQ,UAAU,KAAK,OAAO,wBAAwB,CAAC,GAAG;EACnE,eAAe;EACf,IAAI;EACJ,IAAI;GACF,OAAO,aAAa,MAAM,MAAM;EAClC,QAAQ;GACN;EACF;EACA,KAAK,MAAM,KAAK,KAAK,SAAS,OAAO,GAAG;GACtC,MAAM,IAAI,EAAE;GACZ,IAAI,CAAC,GAAG;GACR,MAAM,IAAI,EAAE,MAAM,GAAG,CAAC,CAAC,IAAI,KAAK;GAChC,MAAM,OAAO,OAAO,IAAI,CAAC;GACzB,IAAI,SAAS,KAAA,GAAW,OAAO,IAAI,GAAG,OAAO,CAAC;EAChD;EACA,KAAK,MAAM,KAAK,KAAK,SAAS,KAAK,GAAG;GACpC,MAAM,IAAI,EAAE;GACZ,IAAI,MAAM,KAAA,GAAW;GACrB,MAAM,OAAO,MAAM,IAAI,CAAC;GACxB,IAAI,SAAS,KAAA,GAAW,MAAM,IAAI,GAAG,OAAO,CAAC;EAC/C;CACF;CAIF,MAAM,yBAAS,IAAI,IAAoB;CACvC,KAAK,MAAM,KAAK,QACd,IAAI;EACF,OAAO,IAAI,EAAE,MAAM,aAAa,EAAE,MAAM,MAAM,CAAC;CACjD,QAAQ;EACN,OAAO,IAAI,EAAE,MAAM,EAAE;CACvB;CAEF,MAAM,UAAU,IAAI,IAAoB,MAAM,KAAK,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;CAChE,KAAK,MAAM,UAAU,OAAO;EAC1B,MAAM,MAAM,IAAI,OAAO,IAAI,OAAO,YAAY,OAAO,OAAO;EAC5D,KAAK,MAAM,KAAK,QAAQ;GACtB,IAAI,EAAE,SAAS,QAAQ;GACvB,IAAI,IAAI,KAAK,OAAO,IAAI,EAAE,IAAI,KAAK,EAAE,GAAG,QAAQ,IAAI,QAAQ,QAAQ,IAAI,MAAM,IAAK,CAAC;EACtF;CACF;CAEA,MAAM,UAA8B,OAAO,KAAK,MAAM;EACpD,MAAM,OAAO,OAAO,IAAI,EAAE,IAAI,KAAK;EACnC,MAAM,MAAM,EAAE,KAAK,QAAQ,gBAAgB,EAAE;EAC7C,OAAO;GACL,MAAM,EAAE;GACR,MAAM,EAAE;GACR,MAAM,EAAE;GACR,OAAO,OAAO,KAAK,MAAM,IAAI,CAAC,CAAC,SAAS;GACxC,mBAAmB,OAAO,IAAI,EAAE,IAAI,KAAK;GACzC,kBAAkB,MAAM,IAAI,EAAE,IAAI,KAAK;GACvC,aAAa,QAAQ,IAAI,EAAE,IAAI,KAAK;GACpC,eAAe,eACb,OAAO,iBAAiB,CAAC,GACzB,EAAE,MACF,OAAO,kBAAkB,EAAE,SAAS,CAAC,CACvC;GACA,oBAAoB,KAAK,MAAM,iBAAiB,KAAK,CAAC,EAAA,CAAG;GACzD,kBAAkB,WAAW,KAAK,KAAK,YAAY,CAAC;GACpD,aAAa,WAAW,KAAK,KAAK,OAAO,CAAC;GAC1C,UAAU,KAAK,SAAS,kBAAkB;GAC1C,mBAAmB,WAAW,KAAK,uBAAuB,IAAI,KAAK,KAAK,MAAM,GAAG,GAAG,CAAC;EACvF;CACF,CAAC;CACD,OAAO;EAAE,qBAAqB;EAAa;CAAQ;AACrD;AAIA,MAAM,aAAa;AAEnB,SAAS,QACP,MACA,SACA,OACA,UACA,YACA,YACA,aACA,aACA,WACgB;CAChB,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAAE,YAAY;GAAY;GAAM;GAAS;EAAM,CAAC;EAC7E,YAAY;EACZ,aAAa;EACb;EACA;EACA;EACA;EACA,eAAe,CAAC;GAAE,MAAM;GAAY,KAAK;EAAY,CAAC;EACtD,oBAAoB;EACpB;EACA;CACF;AACF;;AAGA,SAAgB,uBACd,QACA,YACkB;CAClB,MAAM,MAAwB,CAAC;CAC/B,KAAK,MAAM,KAAK,OAAO,SAAS;EAC9B,MAAM,cAAc,EAAE,oBAAoB,EAAE;EAI5C,IAHkB,cAAc,EAAE,cAAc,EAAE,kBAGhC,GAChB,IAAI,KACF,QACE,eACA,EAAE,MACF,UAAU,EAAE,KAAK,+EACjB,QACA,IACA,YACA,sIACA,EAAE,MACF,iGACF,CACF;OACK,IAAI,gBAAgB,KAAK,EAAE,cAAc,EAAE,gBAAgB,GAEhE,IAAI,KACF,QACE,eACA,EAAE,MACF,UAAU,EAAE,KAAK,gFAAgF,EAAE,YAAY,cAAc,EAAE,cAAc,IAC7I,QACA,IACA,YACA,2JACA,EAAE,MACF,sEACF,CACF;EAIF,IAAI,eAAe,KAAK,CAAC,EAAE,mBACzB,IAAI,KACF,QACE,mBACA,EAAE,MACF,UAAU,EAAE,KAAK,mFACjB,UACA,IACA,YACA,oHACA,EAAE,IACJ,CACF;EAIF,IAAI,EAAE,SAAS,YAAY,EAAE,oBAAoB,GAC/C,IAAI,KACF,QACE,UACA,EAAE,MACF,iBAAiB,EAAE,KAAK,YAAY,EAAE,kBAAkB,+BACxD,QACA,KACA,YACA,uMACA,EAAE,IACJ,CACF;EAIF,IAAI,EAAE,QAAQ,wBAAwB,CAAC,EAAE,kBACvC,IAAI,KACF,QACE,mBACA,EAAE,MACF,UAAU,EAAE,KAAK,OAAO,EAAE,MAAM,4DAChC,UACA,IACA,YACA,mFAAmF,EAAE,MAAM,mDAC3F,EAAE,IACJ,CACF;EAIF,IAAI,CAAC,EAAE,aACL,IAAI,KACF,QACE,gBACA,EAAE,MACF,UAAU,EAAE,KAAK,oBACjB,OACA,IACA,YACA,wGACA,EAAE,IACJ,CACF;EAIF,IAAI,CAAC,EAAE,UACL,IAAI,KACF,QACE,iBACA,EAAE,MACF,UAAU,EAAE,KAAK,8CACjB,OACA,KACA,YACA,iJACA,EAAE,IACJ,CACF;CAEJ;CACA,OAAO;AACT;AAIA,IAAa,oBAAb,MAAoE;CAClE,KAAc;CACd,cACE;CACF,YAAqB;CACrB,OAAgB;EAAE,MAAM;EAA0B,iBAAiB;CAAE;CACrE,UAAmB;CAEnB,MAAM,QAAQ,OAAyB,KAAgD;EACrF,MAAM,aAAa,IAAI,MAAM,+BAAc,IAAI,KAAK,EAAA,CAAE,YAAY;EAClE,IAAI,MACF,gBAAgB,MAAM,QAAQ,OAAO,eAAe,MAAM,oBAAoB,aAChF;EACA,OAAO,uBAAuB,OAAO,UAAU;CACjD;AACF;AAEA,MAAa,sBAAsB,IAAI,kBAAkB;;;AClYzD,MAAM,yBAAyB;CAC7B;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,IAAa,YAAb,MAAuB;CACrB;CACA;CAEA,YAAY,UAA4B,CAAC,GAAG;EAC1C,KAAK,UAAU,QAAQ;EACvB,KAAK,gBAAgB,QAAQ,iBAAiB;CAChD;CAEA,MAAM,MAAM,OAAmB,OAAkC;EAC/D,MAAM,MAAM,MAAM,MAAM,OAAO,KAAK;EACpC,IAAI,CAAC,KAAK,MAAM,IAAI,cAAc,OAAO,MAAM,WAAW;EAC1D,MAAM,CAAC,OAAO,QAAQ,WAAW,UAAU,MAAM,QAAQ,IAAI;GAC3D,MAAM,MAAM,EAAE,MAAM,CAAC;GACrB,MAAM,OAAO,EAAE,MAAM,CAAC;GACtB,MAAM,UAAU,KAAK;GACrB,MAAM,OAAO,KAAK;EACpB,CAAC;EACD,OAAO,KAAK,WAAW;GAAE;GAAK;GAAO;GAAQ;GAAW;EAAO,CAAC;CAClE;CAEA,WAAW,OAA2B;EACpC,MAAM,QAAkB,CAAC;EACzB,MAAM,WAAW,MAAM,MAAM,QAC1B,MAA2C,EAAE,SAAS,KACzD;EACA,MAAM,YAAY,MAAM,MAAM,QAC3B,MAA4C,EAAE,SAAS,MAC1D;EACA,MAAM,aAAa,MAAM,MAAM,QAC5B,MAA6C,EAAE,SAAS,OAC3D;EACA,MAAM,eAAe,MAAM,MAAM,QAC9B,MAA+C,EAAE,SAAS,SAC7D;EACA,MAAM,iBAAiB,WAAW,QAC/B,SAAS,KAAK,cAAc,gBAAgB,KAAK,YAAY,cAAc,IAC9E;EAEA,MAAM,UACJ,MAAM,IAAI,SAAS,SAAS,OAAO,IAAI,MAAM,IAAI,WAAW,cAAc,KAAM;EAClF,IAAI,CAAC,SAAS,MAAM,KAAK,qCAAqC;EAE9D,MAAM,eAAe,WAAW,SAC5B,WAAW,QAAQ,KAAK,SAAS,MAAM,oBAAoB,KAAK,KAAK,GAAG,CAAC,IACzE,WAAW,SACX,KAAA;EAOJ,MAAM,gBALJ,OAAO,MAAM,IAAI,SAAS,UAAU,WAChC,QACE,MAAM,IAAI,QAAQ,QAAQ,IAAI,MAAM,IAAI,QAAQ,QAAQ,MAAM,MAAM,IAAI,QAAQ,KAClF,IACA,KAAA,MAC+B,gBAAgB;EAErD,MAAM,kBAAkB,UAAU,QAAQ,SAAS,KAAK,WAAW,OAAO,CAAC,CAAC;EAC5E,MAAM,iBAAiB,UAAU,WAAW,IAAI,IAAI,kBAAkB,UAAU;EAChF,IAAI,UAAU,WAAW,GAAG,MAAM,KAAK,wBAAwB;EAE/D,MAAM,gBACJ,MAAM,UAAU,SAChB,UAAU,QAAQ,SAAS,0BAA0B,KAAK,KAAK,QAAQ,CAAC,CAAC,CAAC;EAC5E,MAAM,eAAe,gBAAgB,IAAI,QAAQ,gBAAgB,CAAC,IAAI;EACtE,IAAI,CAAC,cAAc,MAAM,KAAK,uCAAuC;EAErE,MAAM,eAAe,aAAa,QAC/B,SAAS,OAAO,KAAK,eAAe,YAAY,KAAK,aAAa,CACrE;EACA,MAAM,cAAc,aAAa,SAC7B,aAAa,QACV,KAAK,SAAS,OAAO,KAAK,eAAe,KAAK,KAAK,IAAI,GAAG,KAAK,cAAc,CAAC,GAC/E,CACF,IAAI,aAAa,SACjB,UAAU,MAAM,SACZ,yCAAyC,KAAK,KAAK,UAAU,KAAK,IAAI,CAAC,CACzE,IACA,KACA;EACN,IAAI,CAAC,aAAa,MAAM,KAAK,sCAAsC;EAEnE,MAAM,eAAe,WAAW,QAAQ,SAAS,gBAAgB,IAAI,CAAC;EACtE,MAAM,oBAAoB,eAAe,QAAQ,SAAS,gBAAgB,IAAI,CAAC;EAC/E,MAAM,YAAY,eAAe,SAAU,kBAAkB,SAAS,IAAI,IAAK;EAC/E,IAAI,kBAAkB,QACpB,MAAM,KAAK,yBAAyB,kBAAkB,OAAO,aAAa;OACvE,IAAI,CAAC,eAAe,QAAQ,MAAM,KAAK,iCAAiC;EAE7E,MAAM,mBAAmB,WAAW,SAAS,aAAa,SAAS,WAAW,SAAS;EACvF,IAAI,kBAAkB,MAAM,KAAK,YAAY,aAAa,OAAO,6BAA6B;EAE9F,MAAM,2BACJ,gBACA,aAAa,SACb,SAAS,QAAQ,SAAS,kBAAkB,KAAK,UAAU,EAAE,CAAC,CAAC,CAAC;EAClE,MAAM,eACJ,SAAS,QAAQ,SAAS,KAAK,QAAQ,KAAK,UAAU,EAAE,CAAC,CAAC,CAAC,SAC3D,MAAM,OAAO,QAAQ,UAAU,KAAK,QAAQ,KAAK,UAAU,MAAM,OAAO,CAAC,CAAC,CAAC,CAAC;EAC9E,MAAM,mBACJ,2BAA2B,iBAAiB,IACxC,IACA,4BAA4B,2BAA2B;EAC7D,MAAM,eACJ,2BAA2B,iBAAiB,IACxC,IACA,gBAAgB,2BAA2B;EACjD,IAAI,eAAe,GAAG,MAAM,KAAK,YAAY,aAAa,iBAAiB;EAe3E,OAAO;GACL;GACA;GACA;GACA;GACA;GACA;GACA;GACA;GACA;GACA,SAvBc,MAAM,OAAO,SACzB,KAAK,IACH,GAAG,MAAM,OACN,QAAQ,UAA6B,MAAM,cAAc,KAAK,CAAC,CAC/D,KAAK,UAA6B,MAAM,QAAQ,GACnD,CACF,IACA,SAAS,QAAQ,KAAK,SAAS,OAAO,KAAK,WAAW,IAAI,CAAC;GAiB7D,aAfA,MAAM,IAAI,WAAW,MAAM,IAAI,YAC3B,KAAK,IAAI,IAAI,MAAM,IAAI,UAAU,MAAM,IAAI,aAAa,GAAI,IAC5D;GAcJ;EACF;CACF;CAEA,KAAK,OAAyB;EAC5B,OAAO,kBAAkB,OAAO,KAAK,OAAO;CAC9C;CAEA,QAAgB,MAAuB;EACrC,OAAO,KAAK,cAAc,MAAM,YAAY,QAAQ,KAAK,IAAI,CAAC;CAChE;AACF;AAEA,SAAS,oBAAoB,OAAuB;CAClD,OAAO,QAAQ,IAAI,QAAQ,QAAQ,EAAE,IAAI,QAAQ,KAAK;AACxD;AAEA,SAAS,kBAAkB,MAAuB;CAChD,OAAO,qGAAqG,KAC1G,IACF;AACF;AAEA,SAAS,gBAAgB,MAAiD;CACxE,OACE,KAAK,YAAY,aAAa,QAC9B,KAAK,YAAY,YAAY,cAC7B,eAAe,KAAK,YAAY,gBAAgB,KAChD,eAAe,KAAK,YAAY,YAAY,KAC5C,KAAK,SAAS;AAElB;AAEA,SAAS,eAAe,OAAyB;CAC/C,OAAO,OAAO,UAAU,YAAY,QAAQ;AAC9C;;;;;;;;;;;;;;;;;;;;;;;AC/EA,MAAa,6BAAgE;CAC3E,QAAQ;CACR,WAAW;CACX,SAAS;AACX;AAiCA,MAAa,iCAAiC;AAE9C,MAAM,qBAAqB;AAC3B,MAAM,mBAAmB;AACzB,MAAM,uBAAuB;AAC7B,MAAM,kBAAkB;AACxB,MAAM,qBAAqB;AAC3B,MAAM,gBAAgB;AAEtB,MAAM,kBAAkB;CACtB,MAAM;CACN,sBAAsB;CACtB,UAAU,CAAC,WAAW,UAAU;CAChC,YAAY;EACV,SAAS;GAAE,MAAM;GAAU,WAAW;GAAI,WAAW;EAAI;EACzD,UAAU;GACR,MAAM;GACN,UAAU;GACV,OAAO;IACL,MAAM;IACN,sBAAsB;IACtB,UAAU;KAAC;KAAW;KAAW;KAAS;KAAY;IAAU;IAChE,YAAY;KACV,SAAS;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACxD,SAAS,EAAE,MAAM,UAAU;KAC3B,OAAO;MAAE,MAAM;MAAU,SAAS;MAAG,SAAS;KAAG;KACjD,UAAU;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACzD,UAAU;MAAE,MAAM;MAAU,MAAM;OAAC;OAAY;OAAS;OAAS;MAAM;KAAE;IAC3E;GACF;EACF;CACF;AACF;AAEA,SAAS,SAAS,MAAc,KAAa,OAAuB;CAClE,IAAI,KAAK,UAAU,KAAK,OAAO;CAC/B,OAAO,GAAG,KAAK,MAAM,GAAG,GAAG,EAAE,iBAAiB,KAAK,SAAS,IAAI,YAAY,MAAM;AACpF;AAEA,SAAS,YACP,OACA,MACQ;CACR,MAAM,aAAa,MAAM,YACtB,QAAQ,MAAM,EAAE,QAAQ,UAAU,KAAK,eAAe,CAAC,CACvD,KAAK,MAAM,aAAa,EAAE,KAAK,QAAQ,EAAE,SAAS,CAAC,CACnD,KAAK,MAAM;CAEd,MAAM,OAAO,MAAM,cAAc;CAEjC,OAAO;;;;;;;;;;EAUP,MAAM,YAAY;;EAElB,MAAM,gBAAgB,+BAA+B,MAAM,cAAc,mBAAmB,MAAM,uBAAuB,GAAG,QAAQ,GAAG;EACvI,MAAM,iBACL,KACE,GAAG,MACF,KAAK,IAAI,EAAE,KAAK,EAAE,KAAK,GAAG,EAAE,UAAU,SAAS,cAAc,EAAE,SAAS,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,KAAK,EAAE,KAAK,IACzG,CAAC,CACA,KAAK,IAAI,EAAE;;EAEZ,OAAO,qDAAqD,SAAS,MAAM,KAAK,cAAc,MAAM,EAAE,QAAQ,GAAG;EACjH,SAAS,YAAY,KAAK,gBAAgB,QAAQ,EAAE;;;;;;;;;;;;;;;;;;AAkBtD;;;;;;AASA,eAAsB,wBACpB,OACA,UAAuC,CAAC,GACH;CACrC,MAAM,QAAQ,KAAK,IAAI;CACvB,MAAM,aAAa,MAAM,iBAAiB;CAE1C,IAAI,eAAe,GACjB,OAAO;EACL,MAAM;EACN,SAAS;EACT,OAAO;EACP,cAAc;EACd,YAAY;EACZ,UAAU,CAAC;EACX,SAAS;EACT,YAAY;EACZ,SAAS;EACT,WAAW;EACX,OAAO;CACT;CAGF,MAAM,OAA8C;EAClD,OAAO,QAAQ,SAAS;EACxB,WAAW,QAAQ,aAAa;EAChC,WAAW,QAAQ,aAAa;EAChC,gBAAgB,QAAQ,kBAAkB;EAC1C,iBAAiB,QAAQ,mBAAmB;EAC5C,cAAc,QAAQ,gBAAgB;EACtC,KAAK,QAAQ,OAAO,CAAC;EACrB,YAAY,QAAQ,cAAc,IAAI,WAAW;EACjD,WAAW,QAAQ,aAAa;EAChC,UAAU,QAAQ,YAAY,CAAC;EAC/B,QAAQ,QAAQ,UAAU,IAAI,gBAAgB,CAAC,CAAC;EAChD,gBAAgB,QAAQ,kBAAkB;EAC1C,mBAAmB;GAAE,GAAG;GAA4B,GAAI,QAAQ,qBAAqB,CAAC;EAAG;CAC3F;CAMA,MAAM,oBAAoB,SAA8B;EACtD,IAAI,KAAK,mBAAmB,QAAQ,OAAO;EAC3C,IAAI,KAAK,UAAU,MAAM,OAAO,KAAK;EACrC,IAAI,KAAK,mBAAmB,cAC1B,OAAO,KAAK,kBAAkB,KAAK,cAAc,aAAa;EAEhE,OAAO;CACT;CACA,MAAM,eAAe,IAAI,IACvB,MAAM,iBAAiB,KAAK,MAAM,CAAC,EAAE,MAAM,iBAAiB,CAAC,CAAC,CAAC,CACjE;CAEA,IAAI;CACJ,IAAI;EACF,MAAM,UAAU;GACd,OAAO,KAAK;GACZ,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IAAE,MAAM;IAAiB,SAAS,YAAY,OAAO,IAAI;GAAE,CAC7D;GACA,YAAY;IAAE,MAAM;IAA0B,QAAQ;GAAgB;GACtE,aAAa;GACb,WAAW,KAAK;GAChB,WAAW,KAAK;EAClB;EACA,MAAM,OAAO,MAAM,KAAK,WAAW,YAAY;GAC7C,SAAS;GACT,OAAO,KAAK;GACZ,OAAO;GACP,OAAO,KAAK;GACZ,GAAI,OAAO,KAAK,KAAK,QAAQ,CAAC,CAAC,SAAS,IAAI,EAAE,MAAM,KAAK,SAAS,IAAI,CAAC;GACvE,eAAe,2BAA2B,SAAS,KAAK,GAAG;GAC3D,QAAQ,KAAK;GACb,UAAU,QAAQ,WAChB,YAA6D,SAAS;IACpE,GAAG,KAAK;IACR;IACA,gBAAgB;GAClB,CAAC;GACH,UAAU,EAAE,aAAa,mBAAmB,MAAM;GAClD,kBAAkB;EACpB,CAAC;EACD,UAAU,KAAK;EACf,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;EAChC,MAAM,EAAE,UAAU,KAAK;EAEvB,IAAI,CAAC,OAAO,YAAY,CAAC,MAAM,QAAQ,MAAM,QAAQ,GACnD,MAAM,IAAI,MAAM,uEAAqE;EAGvF,MAAM,WAA6B,MAAM,SAAS,KAAK,OAAO;GAC5D,SAAS,OAAO,EAAE,OAAO;GACzB,SAAS,QAAQ,EAAE,OAAO;GAC1B,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,IAAI,OAAO,EAAE,SAAS,CAAC,CAAC,CAAC;GACrD,UAAU,OAAO,EAAE,YAAY,EAAE;GACjC,UAAW;IAAC;IAAY;IAAS;IAAS;GAAM,CAAC,CAAW,SAAS,EAAE,QAAQ,IAC3E,EAAE,WACF;EACN,EAAE;EAEF,MAAM,eAAe,SAAS,QAAQ,MAAM,EAAE,WAAW,EAAE,SAAS,CAAC,CAAC,CAAC;EACvE,IAAI,YAAY;EAChB,IAAI,mBAAmB;EACvB,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,IAAI,aAAa,IAAI,EAAE,OAAO,KAAK;GACzC,aAAa;GACb,oBAAoB,IAAI,EAAE;EAC5B;EACA,MAAM,WACJ,YAAY,IACR,mBAAmB,YACnB,SAAS,QAAQ,GAAG,MAAM,IAAI,EAAE,OAAO,CAAC,IAAI,KAAK,IAAI,GAAG,SAAS,MAAM;EAE7E,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO,QAAQ,WAAW,GAAA,CAAI,QAAQ,CAAC,CAAC;GACxC;GACA;GACA;GACA,SAAS,OAAO,MAAM,WAAW,EAAE;GACnC,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,KAAK,QAAQ,cAAc,OAAO,KAAK,QAAQ;GACxD,WAAW;EACb;CACF,SAAS,KAAK;EACZ,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO;GACP,cAAc;GACd;GACA,UAAU,CAAC;GACX,SAAS;GACT,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,WAAW,CAAC,QAAQ,cAAc,QAAQ,UAAU;GAC7D,WAAW;GACX,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;EACxD;CACF;AACF;;;;;AAMA,SAAgB,2BACd,UAAuC,CAAC,GACmC;CAC3E,QAAQ,UAAU,wBAAwB,OAAO,OAAO;AAC1D"}
1
+ {"version":3,"file":"semantic-concept-judge-C0P1VTXD.js","names":[],"sources":["../src/locked-jsonl-appender.ts","../src/analyst/findings-store.ts","../src/analyst/kinds/skill-usage.ts","../src/run-critic.ts","../src/semantic-concept-judge.ts"],"sourcesContent":["/**\n * LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary\n * payloads. The reference-replay store does the same thing for typed\n * `ReferenceReplayRun` rows; this is the generic version used by\n * `MutationTelemetry`, `TrialTelemetry`, and any other consumer that wants\n * append-only durable telemetry without rolling its own lock.\n *\n * Locks are per absolute file path (process-local). Cross-process\n * concurrency is NOT addressed — that's an fcntl/flock problem.\n */\n\nimport { appendFileSync, existsSync, mkdirSync } from 'node:fs'\nimport { dirname } from 'node:path'\nimport { Mutex } from './concurrency'\n\nconst mutexes = new Map<string, Mutex>()\n\nfunction getMutex(path: string): Mutex {\n let m = mutexes.get(path)\n if (!m) {\n m = new Mutex()\n mutexes.set(path, m)\n }\n return m\n}\n\nexport class LockedJsonlAppender {\n private readonly mutex: Mutex\n constructor(public readonly path: string) {\n this.mutex = getMutex(path)\n if (!existsSync(dirname(path))) {\n mkdirSync(dirname(path), { recursive: true })\n }\n }\n\n async append(entry: unknown): Promise<void> {\n const line = `${JSON.stringify(entry)}\\n`\n await this.mutex.runExclusive(() => {\n appendFileSync(this.path, line)\n })\n }\n}\n\n/** Reset all internal mutex state — tests only. */\nexport function resetLockedAppendersForTesting(): void {\n mutexes.clear()\n}\n","/**\n * FindingsStore — durable persistence for AnalystFinding rows + a diff\n * helper so we can answer \"what changed since the last run?\" without\n * recomputing analysts.\n *\n * On-disk shape is JSONL: one finding per line, append-only, locked via\n * LockedJsonlAppender. Operators get crash-safety (no partial JSON),\n * cheap reads (sequential parse), and trivial backup (rsync the file).\n *\n * Reads are non-locking: a reader sees a consistent snapshot of all\n * fully-written lines and skips an incomplete trailing line if the\n * writer is mid-append. Cross-process locking is intentionally out of\n * scope (see locked-jsonl-appender.ts).\n *\n * The store is run-scoped: callers pass `runId` on append and on load,\n * which keeps multi-run files cleanly partitioned. The `diffFindings`\n * helper compares two run-id sets using stable `finding_id` semantics —\n * the diff is the cross-run signal the regression dashboard renders.\n */\n\nimport { existsSync, readFileSync } from 'node:fs'\n\nimport { LockedJsonlAppender } from '../locked-jsonl-appender'\nimport type { AnalystFinding } from './types'\n\n/**\n * One persisted row. We attach `run_id` on disk so a single file can\n * hold multiple runs and the diff helper can query without re-walking\n * separate files.\n */\nexport interface PersistedFinding extends AnalystFinding {\n run_id: string\n}\n\nexport class FindingsStore {\n private readonly appender: LockedJsonlAppender\n\n constructor(public readonly path: string) {\n this.appender = new LockedJsonlAppender(path)\n }\n\n async append(runId: string, findings: AnalystFinding[]): Promise<void> {\n for (const f of findings) {\n const row: PersistedFinding = { ...f, run_id: runId }\n await this.appender.append(row)\n }\n }\n\n /** Load every persisted finding. Discards malformed trailing lines silently. */\n loadAll(): PersistedFinding[] {\n if (!existsSync(this.path)) return []\n const raw = readFileSync(this.path, 'utf8')\n if (!raw) return []\n const out: PersistedFinding[] = []\n for (const line of raw.split('\\n')) {\n if (!line) continue\n try {\n out.push(JSON.parse(line) as PersistedFinding)\n } catch {\n // Skip torn trailing line — the lock guarantees no torn lines\n // mid-file, only at EOF when a writer is in-flight.\n }\n }\n return out\n }\n\n /** Filter to a single run. */\n loadRun(runId: string): PersistedFinding[] {\n return this.loadAll().filter((r) => r.run_id === runId)\n }\n}\n\n// ── Cross-run diff ──────────────────────────────────────────────────\n\nexport interface FindingsDiff {\n /** New finding ids in `current` that weren't in `previous`. */\n appeared: PersistedFinding[]\n /** Finding ids in `previous` that aren't in `current`. */\n disappeared: PersistedFinding[]\n /** Same finding id present in both runs and unchanged per the materiality test. */\n persisted: PersistedFinding[]\n /**\n * Same finding id in both runs but at least one non-identity field\n * shifted per `DiffPolicy.isMaterial`. Reported as [previous, current].\n */\n changed: Array<{ previous: PersistedFinding; current: PersistedFinding }>\n}\n\nexport interface DiffPolicy {\n /**\n * Predicate that decides whether two findings (same finding_id) count\n * as a material change. Defaults to {@link defaultIsMaterial}: severity\n * shift, confidence Δ > 0.05, or evidence count change. Compliance /\n * perf consumers MAY supply a stricter predicate (e.g. rationale text\n * diff, metric Δ thresholds).\n */\n isMaterial?: (previous: AnalystFinding, current: AnalystFinding) => boolean\n}\n\n/**\n * Default materiality test. Deliberately narrow so LLM-reword churn\n * doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.\n */\nexport function defaultIsMaterial(a: AnalystFinding, b: AnalystFinding): boolean {\n if (a.severity !== b.severity) return true\n if (Math.abs((a.confidence ?? 0) - (b.confidence ?? 0)) > 0.05) return true\n if (a.evidence_refs.length !== b.evidence_refs.length) return true\n return false\n}\n\n/**\n * Diff two findings sets by stable finding_id. Callers typically load\n * the two run-id slices from the same store and pass them in.\n */\nexport function diffFindings(\n previous: PersistedFinding[],\n current: PersistedFinding[],\n policy: DiffPolicy = {},\n): FindingsDiff {\n const isMaterial = policy.isMaterial ?? defaultIsMaterial\n const prevById = new Map(previous.map((f) => [f.finding_id, f]))\n const curById = new Map(current.map((f) => [f.finding_id, f]))\n\n const appeared: PersistedFinding[] = []\n const disappeared: PersistedFinding[] = []\n const persisted: PersistedFinding[] = []\n const changed: FindingsDiff['changed'] = []\n\n for (const [id, cur] of curById) {\n const prev = prevById.get(id)\n if (!prev) {\n appeared.push(cur)\n continue\n }\n if (isMaterial(prev, cur)) {\n changed.push({ previous: prev, current: cur })\n } else {\n persisted.push(cur)\n }\n }\n for (const [id, prev] of prevById) {\n if (!curById.has(id)) disappeared.push(prev)\n }\n return { appeared, disappeared, persisted, changed }\n}\n","/**\n * Skill-usage analyst — a DETERMINISTIC `Analyst` over a Claude/Codex skill\n * library + its trace corpus. Unlike the trace-store kinds (failure-mode,\n * improvement, ...) this kind calls no LLM: it mines real usage and skill\n * structure and emits findings by rule.\n *\n * It exists because the naive \"Skill-tool invocation count\" lies low — it\n * misses orchestrated sub-dispatch (a leaf skill run BY /pursue or /governor\n * logs under the parent), slash-command entry, local-script bypass, and\n * on-disk artifacts. The 2026-05-30 skill audit found 39/53 skills at zero\n * direct invocations, yet only one was a genuine cut: the rest were\n * measurement-invisible or discovery-limited. This analyst encodes that\n * lesson as a multi-signal usage model so a cheap repeatable pass can keep\n * the library honest, and so the expensive audit workflow's verdicts can\n * GEPA-distill it toward agreement (see `gold/skill-verdicts.gold.jsonl`).\n *\n * Report-building (`buildSkillUsageReport`, an fs scan) is separated from\n * finding emission (`SkillUsageAnalyst.analyze`, pure) so the slow scan runs\n * once at the registry boundary and the rule logic stays unit-testable.\n */\n\nimport { type Dirent, existsSync, readdirSync, readFileSync, statSync } from 'node:fs'\nimport { join } from 'node:path'\nimport type { Analyst, AnalystContext, AnalystFinding, AnalystSeverity } from '../types'\nimport { computeFindingId } from '../types'\n\n// ── Input model ──────────────────────────────────────────────────────\n\nexport type SkillKind = 'public' | 'private'\n\n/** One skill's multi-signal usage + structure. All counts are deterministic. */\nexport interface SkillUsageRecord {\n name: string\n kind: SkillKind\n /** Absolute path to the skill's SKILL.md. */\n path: string\n lines: number\n /** `\"skill\":\"<name>\"` Skill-tool invocations across the trace corpus. */\n directInvocations: number\n /** `<command-name>/<name>` slash invocations across the trace corpus. */\n slashInvocations: number\n /** Sibling skills whose SKILL.md dispatches to this one (`/<name>`). Proxy\n * for orchestrated sub-dispatch the per-skill counter cannot see. */\n inboundRefs: number\n /** On-disk artifacts attributable to the skill (e.g. `.evolve/<name>/**`). */\n artifactCount: number\n /** Tangle-private reference count in the body (leak signal for public skills). */\n tanglePrivateRefs: number\n hasReferencesDir: boolean\n hasEvalsDir: boolean\n /** Body mentions `skill-runs.jsonl` (visible to /reflect + /governor). */\n logsRuns: boolean\n /** Description carries an explicit `Triggers:` clause / trigger phrases. */\n hasTriggerPhrases: boolean\n}\n\nexport interface SkillUsageReport {\n generatedFromTraces: number\n records: SkillUsageRecord[]\n}\n\nexport interface SkillUsageScanConfig {\n /** Dirs holding `*.jsonl` transcripts (Claude `~/.claude/projects`, Codex sessions). */\n transcriptDirs: string[]\n /** Skill roots to scan; each dir directly under `root` with a `SKILL.md` is a skill. */\n skillRoots: { root: string; kind: SkillKind }[]\n /** Roots scanned for `<root>/.evolve/<skill>` artifact dirs. */\n artifactRoots?: string[]\n /** Token-prefixed mappings: skill name → extra artifact subpaths under an artifactRoot\n * (e.g. reflect → `.evolve/reflections`). Catches non-eponymous artifact dirs. */\n artifactAliases?: Record<string, string[]>\n /** Cap files read per transcript dir (bounds a huge corpus); 0 = unbounded. */\n maxTranscriptsPerDir?: number\n}\n\n// ── Deterministic thresholds ─────────────────────────────────────────\n\n/** Anthropic's authoring guidance keeps SKILL.md short; past this with no\n * `references/` split the body burns context budget every session. */\nconst BLOAT_LINE_THRESHOLD = 300\n\nconst TANGLE_PRIVATE_RE =\n /\\b(cli-bridge|tangletools|ops-board|drew-gtr-pro|@tangle-network\\/|~\\/company|tangle\\.tools|gtm-agent)\\b|\\bkimi\\b|\\btcloud\\b/gi\nconst TRIGGER_RE = /triggers?\\s*[:-]/i\n\n// ── Report builder (fs scan — slow, runs once at the registry boundary) ──\n\nfunction listSkillDirs(root: string): { name: string; path: string }[] {\n if (!existsSync(root)) return []\n const out: { name: string; path: string }[] = []\n for (const entry of readdirSync(root, { withFileTypes: true })) {\n if (!entry.isDirectory() && !entry.isSymbolicLink()) continue\n const skillMd = join(root, entry.name, 'SKILL.md')\n if (existsSync(skillMd)) out.push({ name: entry.name, path: skillMd })\n }\n return out\n}\n\nfunction walkJsonl(dir: string, cap: number): string[] {\n if (!existsSync(dir)) return []\n const files: string[] = []\n const stack = [dir]\n while (stack.length) {\n const cur = stack.pop()!\n let entries: Dirent[]\n try {\n entries = readdirSync(cur, { withFileTypes: true })\n } catch {\n continue\n }\n for (const e of entries) {\n const full = join(cur, e.name)\n if (e.isDirectory()) stack.push(full)\n else if (e.name.endsWith('.jsonl')) {\n files.push(full)\n if (cap > 0 && files.length >= cap) return files\n }\n }\n }\n return files\n}\n\nfunction frontmatterDescription(body: string): string {\n const fm = /^---\\n([\\s\\S]*?)\\n---/.exec(body)\n const block = fm?.[1] ?? ''\n const m = /description:\\s*(.+)/i.exec(block)\n return m?.[1] ?? ''\n}\n\nfunction countArtifacts(roots: string[], name: string, aliases: string[]): number {\n let n = 0\n for (const root of roots) {\n const candidates = [join(root, '.evolve', name), ...aliases.map((a) => join(root, a))]\n for (const dir of candidates) {\n if (!existsSync(dir)) continue\n try {\n if (statSync(dir).isDirectory()) n += readdirSync(dir).length\n else n += 1\n } catch {\n /* unreadable — skip */\n }\n }\n }\n return n\n}\n\n/** Scan the corpus + skill roots into a {@link SkillUsageReport}. Deterministic. */\nexport function buildSkillUsageReport(config: SkillUsageScanConfig): SkillUsageReport {\n const skills = config.skillRoots.flatMap(({ root, kind }) =>\n listSkillDirs(root).map((s) => ({ ...s, kind })),\n )\n const names = skills.map((s) => s.name)\n\n // One pass over the corpus accumulating direct + slash counts per skill.\n const direct = new Map<string, number>(names.map((n) => [n, 0]))\n const slash = new Map<string, number>(names.map((n) => [n, 0]))\n const skillRe = /\"skill\"\\s*:\\s*\"([a-z0-9_:-]+)\"/g\n const cmdRe = /<command-name>\\/?([a-z0-9_:-]+)<\\/command-name>/g\n let transcripts = 0\n for (const dir of config.transcriptDirs) {\n for (const file of walkJsonl(dir, config.maxTranscriptsPerDir ?? 0)) {\n transcripts += 1\n let data: string\n try {\n data = readFileSync(file, 'utf8')\n } catch {\n continue\n }\n for (const m of data.matchAll(skillRe)) {\n const g = m[1]\n if (!g) continue\n const n = g.split(':').pop() ?? g\n const prev = direct.get(n)\n if (prev !== undefined) direct.set(n, prev + 1)\n }\n for (const m of data.matchAll(cmdRe)) {\n const g = m[1]\n if (g === undefined) continue\n const prev = slash.get(g)\n if (prev !== undefined) slash.set(g, prev + 1)\n }\n }\n }\n\n // Read each skill body once; compute structure + inbound refs across siblings.\n const bodies = new Map<string, string>()\n for (const s of skills) {\n try {\n bodies.set(s.name, readFileSync(s.path, 'utf8'))\n } catch {\n bodies.set(s.name, '')\n }\n }\n const inbound = new Map<string, number>(names.map((n) => [n, 0]))\n for (const target of names) {\n const ref = new RegExp(`/${target}\\\\b|\\\\[\\\\[${target}\\\\]\\\\]`)\n for (const s of skills) {\n if (s.name === target) continue\n if (ref.test(bodies.get(s.name) ?? '')) inbound.set(target, inbound.get(target)! + 1)\n }\n }\n\n const records: SkillUsageRecord[] = skills.map((s) => {\n const body = bodies.get(s.name) ?? ''\n const dir = s.path.replace(/\\/SKILL\\.md$/, '')\n return {\n name: s.name,\n kind: s.kind,\n path: s.path,\n lines: body ? body.split('\\n').length : 0,\n directInvocations: direct.get(s.name) ?? 0,\n slashInvocations: slash.get(s.name) ?? 0,\n inboundRefs: inbound.get(s.name) ?? 0,\n artifactCount: countArtifacts(\n config.artifactRoots ?? [],\n s.name,\n config.artifactAliases?.[s.name] ?? [],\n ),\n tanglePrivateRefs: (body.match(TANGLE_PRIVATE_RE) ?? []).length,\n hasReferencesDir: existsSync(join(dir, 'references')),\n hasEvalsDir: existsSync(join(dir, 'evals')),\n logsRuns: body.includes('skill-runs.jsonl'),\n hasTriggerPhrases: TRIGGER_RE.test(frontmatterDescription(body) || body.slice(0, 600)),\n }\n })\n return { generatedFromTraces: transcripts, records }\n}\n\n// ── Finding emission (pure — unit-testable, no LLM, no fs) ────────────\n\nconst ANALYST_ID = 'skill-usage'\n\nfunction finding(\n area: string,\n subject: string,\n claim: string,\n severity: AnalystSeverity,\n confidence: number,\n producedAt: string,\n recommended: string,\n evidenceUri: string,\n rationale?: string,\n): AnalystFinding {\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({ analyst_id: ANALYST_ID, area, subject, claim }),\n analyst_id: ANALYST_ID,\n produced_at: producedAt,\n severity,\n area,\n claim,\n rationale,\n evidence_refs: [{ kind: 'artifact', uri: evidenceUri }],\n recommended_action: recommended,\n confidence,\n subject,\n }\n}\n\n/** Pure rule pass over a report → findings. Exported for direct/unit use. */\nexport function emitSkillUsageFindings(\n report: SkillUsageReport,\n producedAt: string,\n): AnalystFinding[] {\n const out: AnalystFinding[] = []\n for (const r of report.records) {\n const directTotal = r.directInvocations + r.slashInvocations\n const trueUsage = directTotal + r.inboundRefs + r.artifactCount\n\n // 1. Dead: no usage signal of ANY kind. The only real deprecation candidate.\n if (trueUsage === 0) {\n out.push(\n finding(\n 'skill-usage',\n r.name,\n `Skill '${r.name}' has zero usage across all signals (direct, slash, inbound-refs, artifacts)`,\n 'high',\n 0.6,\n producedAt,\n 'Confirm the skill covers a real recurring job; if not, deprecate. Zero true usage is the only deterministic deprecation candidate.',\n r.path,\n 'No Skill-tool call, no slash invocation, no sibling dispatches to it, and no on-disk artifacts.',\n ),\n )\n } else if (directTotal === 0 && r.inboundRefs + r.artifactCount > 0) {\n // 2. Measurement-invisible: real use via orchestration/artifacts, never invoked directly.\n out.push(\n finding(\n 'skill-usage',\n r.name,\n `Skill '${r.name}' shows 0 direct invocations but is used via orchestration/artifacts (inbound=${r.inboundRefs}, artifacts=${r.artifactCount})`,\n 'info',\n 0.8,\n producedAt,\n 'Do NOT treat as unused — usage is real but logged under parent skills or on disk. Strengthen direct-invocation discovery only if direct use is desired.',\n r.path,\n 'The Skill-tool counter undercounts orchestrated/chained leaf skills.',\n ),\n )\n }\n\n // 3. Discovery gap: low direct use AND weak trigger surface.\n if (directTotal <= 2 && !r.hasTriggerPhrases) {\n out.push(\n finding(\n 'discoverability',\n r.name,\n `Skill '${r.name}' is rarely invoked directly and its description has no explicit trigger phrases`,\n 'medium',\n 0.7,\n producedAt,\n 'Add a `Triggers:` clause with verbatim user phrases to the frontmatter description so the model auto-invokes it.',\n r.path,\n ),\n )\n }\n\n // 4. Public-repo leak.\n if (r.kind === 'public' && r.tanglePrivateRefs > 0) {\n out.push(\n finding(\n 'safety',\n r.name,\n `Public skill '${r.name}' carries ${r.tanglePrivateRefs} Tangle-private reference(s)`,\n 'high',\n 0.75,\n producedAt,\n 'Sanitize incidental internal refs (cli-bridge/kimi/tcloud/~company/private repos) or relocate to a private repo. Verify @tangle-network/* refs are to PUBLISHED packages before treating as a leak.',\n r.path,\n ),\n )\n }\n\n // 5. Bloat / no progressive disclosure.\n if (r.lines > BLOAT_LINE_THRESHOLD && !r.hasReferencesDir) {\n out.push(\n finding(\n 'maintainability',\n r.name,\n `Skill '${r.name}' is ${r.lines} lines with no references/ split (progressive disclosure)`,\n 'medium',\n 0.8,\n producedAt,\n `Split detail into references/ loaded on demand; keep SKILL.md a short overview. ${r.lines} lines load into every session's context budget.`,\n r.path,\n ),\n )\n }\n\n // 6. No evals (Anthropic's \">=3 evals before docs\" rule).\n if (!r.hasEvalsDir) {\n out.push(\n finding(\n 'data-quality',\n r.name,\n `Skill '${r.name}' ships no evals/`,\n 'low',\n 0.6,\n producedAt,\n 'Add evals/evals.json with >=3 scenarios proving the skill beats baseline; gives regression coverage.',\n r.path,\n ),\n )\n }\n\n // 7. No run logging → invisible to /reflect and /governor.\n if (!r.logsRuns) {\n out.push(\n finding(\n 'observability',\n r.name,\n `Skill '${r.name}' never appends to .evolve/skill-runs.jsonl`,\n 'low',\n 0.55,\n producedAt,\n 'Append one run line to .evolve/skill-runs.jsonl on completion, or declare it a non-logging leaf, so the self-improvement loop can see it ran.',\n r.path,\n ),\n )\n }\n }\n return out\n}\n\n// ── The Analyst ──────────────────────────────────────────────────────\n\nexport class SkillUsageAnalyst implements Analyst<SkillUsageReport> {\n readonly id = ANALYST_ID\n readonly description =\n 'Deterministic multi-signal skill-usage analysis: flags dead skills, measurement-invisible (orchestrated) usage, discovery gaps, public-repo leaks, bloat, missing evals, and missing run-logging.'\n readonly inputKind = 'custom' as const\n readonly cost = { kind: 'deterministic' as const, est_usd_per_run: 0 }\n readonly version = '1.0.0'\n\n async analyze(input: SkillUsageReport, ctx: AnalystContext): Promise<AnalystFinding[]> {\n const producedAt = ctx.tags?.producedAt ?? new Date().toISOString()\n ctx.log?.(\n `skill-usage: ${input.records.length} skills over ${input.generatedFromTraces} transcripts`,\n )\n return emitSkillUsageFindings(input, producedAt)\n }\n}\n\nexport const SKILL_USAGE_ANALYST = new SkillUsageAnalyst()\n","import { NotFoundError } from './errors'\nimport { aggregateRunScore, clamp01, type RunScore, type RunScoreWeights } from './run-score'\nimport type { Artifact, BudgetLedgerEntry, Run, Span, TraceEvent, TraceStore } from './trace'\n\nexport interface RunTrace {\n run: Run\n spans: Span[]\n events: TraceEvent[]\n artifacts: Artifact[]\n budget: BudgetLedgerEntry[]\n}\n\nexport interface RunCriticOptions {\n weights?: Partial<RunScoreWeights>\n driftPatterns?: RegExp[]\n}\n\nconst DEFAULT_DRIFT_PATTERNS = [\n /https?:\\/\\//i,\n /\\btitle:\\s/i,\n /\\bsummary:\\s/i,\n /\\burl:\\s/i,\n /\\bnpm package usage\\b/i,\n /\\bnews\\b/i,\n]\n\nexport class RunCritic {\n private readonly weights?: Partial<RunScoreWeights>\n private readonly driftPatterns: RegExp[]\n\n constructor(options: RunCriticOptions = {}) {\n this.weights = options.weights\n this.driftPatterns = options.driftPatterns ?? DEFAULT_DRIFT_PATTERNS\n }\n\n async score(store: TraceStore, runId: string): Promise<RunScore> {\n const run = await store.getRun(runId)\n if (!run) throw new NotFoundError(`run ${runId} not found`)\n const [spans, events, artifacts, budget] = await Promise.all([\n store.spans({ runId }),\n store.events({ runId }),\n store.artifacts(runId),\n store.budget(runId),\n ])\n return this.scoreTrace({ run, spans, events, artifacts, budget })\n }\n\n scoreTrace(trace: RunTrace): RunScore {\n const notes: string[] = []\n const llmSpans = trace.spans.filter(\n (s): s is Extract<Span, { kind: 'llm' }> => s.kind === 'llm',\n )\n const toolSpans = trace.spans.filter(\n (s): s is Extract<Span, { kind: 'tool' }> => s.kind === 'tool',\n )\n const judgeSpans = trace.spans.filter(\n (s): s is Extract<Span, { kind: 'judge' }> => s.kind === 'judge',\n )\n const sandboxSpans = trace.spans.filter(\n (s): s is Extract<Span, { kind: 'sandbox' }> => s.kind === 'sandbox',\n )\n const finalGateSpans = judgeSpans.filter(\n (span) => span.dimension === 'final_gate' || span.attributes?.finalGate === true,\n )\n\n const success =\n trace.run.outcome?.pass === true ? 1 : trace.run.status === 'completed' ? 0.5 : 0\n if (!success) notes.push('run did not complete with pass=true')\n\n const judgeAverage = judgeSpans.length\n ? judgeSpans.reduce((sum, span) => sum + normalizeJudgeScore(span.score), 0) /\n judgeSpans.length\n : undefined\n const outcomeScore =\n typeof trace.run.outcome?.score === 'number'\n ? clamp01(\n trace.run.outcome.score > 1 ? trace.run.outcome.score / 100 : trace.run.outcome.score,\n )\n : undefined\n const goalProgress = outcomeScore ?? judgeAverage ?? success\n\n const successfulTools = toolSpans.filter((span) => span.status !== 'error').length\n const toolUseQuality = toolSpans.length === 0 ? 0 : successfulTools / toolSpans.length\n if (toolSpans.length === 0) notes.push('no tool spans recorded')\n\n const patchEvidence =\n trace.artifacts.length +\n toolSpans.filter((span) => /write|edit|patch|apply/i.test(span.toolName)).length\n const patchQuality = patchEvidence > 0 ? clamp01(patchEvidence / 4) : 0\n if (!patchQuality) notes.push('no artifact or edit evidence recorded')\n\n const sandboxTests = sandboxSpans.filter(\n (span) => typeof span.testsTotal === 'number' && span.testsTotal > 0,\n )\n const testReality = sandboxTests.length\n ? sandboxTests.reduce(\n (sum, span) => sum + (span.testsPassed ?? 0) / Math.max(1, span.testsTotal ?? 1),\n 0,\n ) / sandboxTests.length\n : toolSpans.some((span) =>\n /\\btest|vitest|pytest|jest|build|tsc\\b/i.test(JSON.stringify(span.args)),\n )\n ? 0.4\n : 0\n if (!testReality) notes.push('no real test/build evidence recorded')\n\n const blockerSpans = judgeSpans.filter((span) => isBlockingJudge(span))\n const finalGateBlockers = finalGateSpans.filter((span) => isBlockingJudge(span))\n const finalGate = finalGateSpans.length ? (finalGateBlockers.length ? 0 : 1) : success\n if (finalGateBlockers.length)\n notes.push(`final gate blocked by ${finalGateBlockers.length} reviewer(s)`)\n else if (!finalGateSpans.length) notes.push('no final gate judgment recorded')\n\n const reviewerBlockers = judgeSpans.length ? blockerSpans.length / judgeSpans.length : 0\n if (reviewerBlockers) notes.push(`detected ${blockerSpans.length} blocking reviewer signal(s)`)\n\n const positiveGroundingSignals =\n patchEvidence +\n sandboxSpans.length +\n llmSpans.filter((span) => looksRepoGrounded(span.output ?? '')).length\n const driftSignals =\n llmSpans.filter((span) => this.isDrift(span.output ?? '')).length +\n trace.events.filter((event) => this.isDrift(JSON.stringify(event.payload))).length\n const repoGroundedness =\n positiveGroundingSignals + driftSignals === 0\n ? 0\n : positiveGroundingSignals / (positiveGroundingSignals + driftSignals)\n const driftPenalty =\n positiveGroundingSignals + driftSignals === 0\n ? 0\n : driftSignals / (positiveGroundingSignals + driftSignals)\n if (driftSignals > 0) notes.push(`detected ${driftSignals} drift signal(s)`)\n\n const costUsd = trace.budget.length\n ? Math.max(\n ...trace.budget\n .filter((entry: BudgetLedgerEntry) => entry.dimension === 'usd')\n .map((entry: BudgetLedgerEntry) => entry.consumed),\n 0,\n )\n : llmSpans.reduce((sum, span) => sum + (span.costUsd ?? 0), 0)\n const wallSeconds =\n trace.run.endedAt && trace.run.startedAt\n ? Math.max(0, (trace.run.endedAt - trace.run.startedAt) / 1000)\n : 0\n\n return {\n success,\n goalProgress,\n repoGroundedness,\n driftPenalty,\n toolUseQuality,\n patchQuality,\n testReality,\n finalGate,\n reviewerBlockers,\n costUsd,\n wallSeconds,\n notes,\n }\n }\n\n rank(score: RunScore): number {\n return aggregateRunScore(score, this.weights)\n }\n\n private isDrift(text: string): boolean {\n return this.driftPatterns.some((pattern) => pattern.test(text))\n }\n}\n\nfunction normalizeJudgeScore(score: number): number {\n return score > 1 ? clamp01(score / 10) : clamp01(score)\n}\n\nfunction looksRepoGrounded(text: string): boolean {\n return /(?:src\\/|tests?\\/|package\\.json|tsconfig|\\.ts\\b|\\.tsx\\b|git status|pnpm |npm |vitest|pytest|jest)/i.test(\n text,\n )\n}\n\nfunction isBlockingJudge(span: Extract<Span, { kind: 'judge' }>): boolean {\n return (\n span.attributes?.blocking === true ||\n span.attributes?.verdict === 'BLOCKING' ||\n positiveNumber(span.attributes?.blockingFindings) ||\n positiveNumber(span.attributes?.highFindings) ||\n span.score <= 2\n )\n}\n\nfunction positiveNumber(value: unknown): boolean {\n return typeof value === 'number' && value > 0\n}\n","/**\n * Semantic concept judge — \"does the built artifact actually implement\n * the features the user asked for?\"\n *\n * Distinct from the domain/code/coherence judges in `judges.ts`:\n * - those judges score free-form conversational agent outputs along\n * quality dimensions (accuracy, depth, etc.)\n * - this judge scores a *built artifact* (served HTML + source files)\n * against an explicit list of expected concepts, returning per-concept\n * {present, score 0-10, evidence, severity}.\n *\n * The judge is strict about distinguishing (a) a working implementation\n * from (b) a keyword-present stub. \"// TODO: mint button\" is NOT present.\n * Only real, functional, wired-up code counts.\n *\n * Use via {@link createSemanticConceptJudge} or directly via\n * {@link runSemanticConceptJudge}. Soft-fails (available=false) on LLM\n * or JSON-parse errors so the caller can treat that as \"layer skipped\"\n * rather than \"layer failed\" in a multi-layer pipeline.\n */\n\nimport { CostLedger, type CostLedgerHandle, type CostReceipt } from './cost-ledger'\nimport {\n callLlmJson,\n costReceiptFromLlm,\n costReceiptFromLlmError,\n type LlmCallRequest,\n type LlmClientOptions,\n maximumChargeForLlmRequest,\n} from './llm-client'\nimport type { Severity } from './multi-layer-verifier'\n\n// ─── Types ──────────────────────────────────────────────────────────────\n\n/**\n * Implementation complexity class for weighted scoring.\n *\n * - `render` (default): the concept is a UI surface that displays static\n * data — render a list, show a counter, lay out a button. Single-file\n * work, no external integration.\n * - `integrate`: the concept requires wiring a real external system —\n * wallet connect (wagmi + RainbowKit + chain config), payment provider\n * (Stripe Elements + intent + webhook), an API client with auth.\n * Multi-file, library-knowledge, runtime correctness matters.\n * - `compute`: the concept requires algorithmic work — solver, simulator,\n * constraint propagation, ML inference. Correctness > UI polish.\n *\n * Default weights (when applied via `weightConcepts: 'complexity'`):\n * render=1.0, integrate=2.0, compute=2.5\n *\n * Cross-vertical scoring without complexity weighting silently inflates\n * the rate of UI-heavy verticals (healthcare, fintech dashboards) vs\n * integration-heavy verticals (DeFi, wallets) — all concepts treated\n * equally even though the agent does 2-3x the work for `integrate`.\n */\nexport type ConceptComplexity = 'render' | 'integrate' | 'compute'\n\nexport interface ConceptSpec {\n name: string\n /** Short hints that help the judge; not used for matching. */\n keywords?: string[]\n /** Optional explicit weight; default 1.0. Overrides complexity-derived weight. */\n weight?: number\n /** Implementation complexity class. Default `render`. */\n complexity?: ConceptComplexity\n}\n\nexport interface ConceptFinding {\n concept: string\n present: boolean\n /** 0..10. 10 = production-ready; 7 = functional thin; 4 = partial; 0 = absent. */\n score: number\n evidence: string\n severity: Severity\n}\n\nexport interface SemanticConceptJudgeInput {\n /** Full natural-language prompt the agent was handed. */\n userRequest: string\n /** Rendered HTML the preview returns (UI artifacts). Optional. */\n servedHtml?: string\n /** Top-level source files from the agent's workdir. */\n sourceFiles: Array<{ path: string; content: string }>\n /** The expected concept list. */\n expectedConcepts: ConceptSpec[]\n /** Free-form metadata (id, difficulty) to inject into the prompt. */\n artifactLabel?: string\n artifactDescription?: string\n}\n\nexport interface SemanticConceptJudgeResult {\n kind: 'semantic-concept'\n version: string\n /** Normalized 0..1 score — mean of per-concept scores / 10. */\n score: number\n presentCount: number\n totalCount: number\n findings: ConceptFinding[]\n summary: string\n durationMs: number\n costUsd: number | null\n /** False on LLM/JSON error — treat as \"skipped / unable to judge\" in pipelines. */\n available: boolean\n error?: string\n}\n\n/**\n * Score-aggregation strategy. `mean` averages 0-10 scores uniformly.\n * `complexity` applies the default weight table (render=1, integrate=2,\n * compute=2.5) unless a concept has an explicit `weight`. `explicit`\n * honors only `weight` (defaulting to 1 for unspecified).\n */\nexport type ConceptWeightStrategy = 'mean' | 'complexity' | 'explicit'\n\nexport const DEFAULT_COMPLEXITY_WEIGHTS: Record<ConceptComplexity, number> = {\n render: 1.0,\n integrate: 2.0,\n compute: 2.5,\n}\n\nexport interface SemanticConceptJudgeOptions {\n /** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */\n model?: string\n /** Per-call timeout. Default 300s. */\n timeoutMs?: number\n /** Provider-enforced output limit. Default 16000. */\n maxTokens?: number\n /** Pipeline budget for the prompt (source blob truncation). Default 45000. */\n maxSourceChars?: number\n /** Per-file cap before inclusion. Default 20000. */\n maxPerFileChars?: number\n /** HTML cap. Default 30000. */\n maxHtmlChars?: number\n /** LlmClient config (baseUrl, apiKey, authHeader, …). */\n llm?: LlmClientOptions\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /**\n * Score aggregation strategy. Default `mean` — uniform average across\n * concepts. Cross-vertical comparisons should use `complexity` to\n * neutralize the integrate-vs-render asymmetry.\n */\n weightConcepts?: ConceptWeightStrategy\n /** Override the default complexity → weight table. */\n complexityWeights?: Partial<Record<ConceptComplexity, number>>\n}\n\n// ─── Prompt assembly ────────────────────────────────────────────────────\n\nexport const SEMANTIC_CONCEPT_JUDGE_VERSION = 'semantic-concept-judge-v1-2026-04-24'\n\nconst DEFAULT_MAX_SOURCE = 45_000\nconst DEFAULT_MAX_HTML = 30_000\nconst DEFAULT_MAX_PER_FILE = 20_000\nconst DEFAULT_TIMEOUT = 300_000\nconst DEFAULT_MAX_TOKENS = 16_000\nconst DEFAULT_MODEL = 'claude-sonnet-4-6'\n\nconst SEMANTIC_SCHEMA = {\n type: 'object',\n additionalProperties: false,\n required: ['summary', 'concepts'],\n properties: {\n summary: { type: 'string', minLength: 20, maxLength: 600 },\n concepts: {\n type: 'array',\n minItems: 1,\n items: {\n type: 'object',\n additionalProperties: false,\n required: ['concept', 'present', 'score', 'evidence', 'severity'],\n properties: {\n concept: { type: 'string', minLength: 1, maxLength: 120 },\n present: { type: 'boolean' },\n score: { type: 'number', minimum: 0, maximum: 10 },\n evidence: { type: 'string', minLength: 5, maxLength: 400 },\n severity: { type: 'string', enum: ['critical', 'major', 'minor', 'info'] },\n },\n },\n },\n },\n}\n\nfunction truncate(body: string, cap: number, label: string): string {\n if (body.length <= cap) return body\n return `${body.slice(0, cap)}\\n… [truncated ${body.length - cap} chars of ${label}]`\n}\n\nfunction buildPrompt(\n input: SemanticConceptJudgeInput,\n opts: Required<SemanticConceptJudgeOptions>,\n): string {\n const sourceBlob = input.sourceFiles\n .filter((f) => f.content.length <= opts.maxPerFileChars)\n .map((f) => `--- FILE: ${f.path} ---\\n${f.content}`)\n .join('\\n\\n')\n\n const html = input.servedHtml ?? ''\n\n return `You are a strict code-review judge evaluating whether an agent's 0-to-1 build actually implements the features the user asked for.\n\nYou MUST distinguish:\n (a) WORKING code that implements the concept (rendered UI, wired handler, real API call),\n (b) KEYWORD-PRESENT stub (comments mentioning the concept, variable names, TODOs),\n (c) ABSENT (concept nowhere).\n\nA comment like \"// TODO: add mint button\" is NOT present — score 2-3. Only count a concept as present if there is real functional code: a rendered component, a call handler wired to state or a network call, a computed value actually used.\n\nUSER REQUEST (what the agent was asked to build):\n${input.userRequest}\n\n${input.artifactLabel ? `ARTIFACT METADATA:\\n name: ${input.artifactLabel}\\n description: ${input.artifactDescription ?? ''}\\n\\n` : ''}EXPECTED CONCEPTS (each must be graded independently):\n${input.expectedConcepts\n .map(\n (c, i) =>\n ` ${i + 1}. \"${c.name}\"${c.keywords?.length ? ` — hints: [${c.keywords.slice(0, 6).join(' | ')}]` : ''}`,\n )\n .join('\\n')}\n\n${html ? `SERVED HTML (what the preview returns when hit):\\n${truncate(html, opts.maxHtmlChars, 'HTML')}\\n\\n` : ''}SOURCE FILES (the agent's workdir):\n${truncate(sourceBlob, opts.maxSourceChars, 'source')}\n\nFor EACH concept, return:\n - concept: the concept name as given (match exactly)\n - present: boolean — does a working implementation exist?\n - score: 0-10 — 10 = production-ready; 7 = functional but thin; 4 = partial/stubbed; 2 = keyword-only comment; 0 = absent\n - evidence: cite \"<file>:<line>\" or \"served-html:<selector>\" pointing at the strongest supporting code. If the concept is absent or stubbed, explain what's missing.\n - severity:\n \"info\" when present: true AND score >= 7\n \"minor\" when present: true AND 4 <= score < 7\n \"major\" when present: false OR score < 4\n \"critical\" when the concept is not only absent but a core user flow depends on it\n\nAlso produce a \"summary\" (one sentence, 20-600 chars): overall verdict on whether this is a shippable implementation of the user request vs a keyword-dense placeholder.\n\nBE SKEPTICAL. Keyword matching already passed — your job is to catch what keyword matching misses. If the agent shipped a working build, say so. If it shipped a stub, say so. Don't grade on effort.\n\nReturn STRICT JSON. No prose outside the JSON.`\n}\n\n// ─── Runner ─────────────────────────────────────────────────────────────\n\n/**\n * Run the semantic concept judge. Soft-fails to available=false on\n * LLM/JSON errors — callers in a MultiLayerVerifier pipeline can treat\n * that as \"skip\" rather than \"fail.\"\n */\nexport async function runSemanticConceptJudge(\n input: SemanticConceptJudgeInput,\n options: SemanticConceptJudgeOptions = {},\n): Promise<SemanticConceptJudgeResult> {\n const start = Date.now()\n const totalCount = input.expectedConcepts.length\n\n if (totalCount === 0) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount: 0,\n findings: [],\n summary: 'no expected concepts declared',\n durationMs: 0,\n costUsd: null,\n available: false,\n error: 'no expected concepts declared',\n }\n }\n\n const opts: Required<SemanticConceptJudgeOptions> = {\n model: options.model ?? DEFAULT_MODEL,\n timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,\n maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,\n maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,\n maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,\n maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,\n llm: options.llm ?? {},\n costLedger: options.costLedger ?? new CostLedger(),\n costPhase: options.costPhase ?? 'judge.semantic-concept',\n costTags: options.costTags ?? {},\n signal: options.signal ?? new AbortController().signal,\n weightConcepts: options.weightConcepts ?? 'mean',\n complexityWeights: { ...DEFAULT_COMPLEXITY_WEIGHTS, ...(options.complexityWeights ?? {}) },\n }\n\n // Build a name → weight map for aggregation. Mean strategy keeps every\n // weight at 1 (uniform average). Complexity strategy reads the table\n // and lets an explicit `weight` override. Explicit strategy uses ONLY\n // the spec's `weight` (defaulting to 1).\n const weightForConcept = (spec: ConceptSpec): number => {\n if (opts.weightConcepts === 'mean') return 1\n if (spec.weight != null) return spec.weight\n if (opts.weightConcepts === 'complexity') {\n return opts.complexityWeights[spec.complexity ?? 'render'] ?? 1\n }\n return 1\n }\n const weightByName = new Map<string, number>(\n input.expectedConcepts.map((c) => [c.name, weightForConcept(c)]),\n )\n\n let receipt: CostReceipt | undefined\n try {\n const request = {\n model: opts.model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You are a strict code-review judge. Return strict JSON only. No prose outside the JSON. A keyword in a comment is NOT a working implementation.',\n },\n { role: 'user' as const, content: buildPrompt(input, opts) },\n ],\n jsonSchema: { name: 'semantic_concept_judge', schema: SEMANTIC_SCHEMA },\n temperature: 0,\n maxTokens: opts.maxTokens,\n timeoutMs: opts.timeoutMs,\n } satisfies LlmCallRequest\n const paid = await opts.costLedger.runPaidCall({\n channel: 'judge',\n phase: opts.costPhase,\n actor: 'semantic-concept',\n model: opts.model,\n ...(Object.keys(opts.costTags).length > 0 ? { tags: opts.costTags } : {}),\n maximumCharge: maximumChargeForLlmRequest(request, opts.llm),\n signal: opts.signal,\n execute: (signal, callId) =>\n callLlmJson<{ summary: string; concepts: ConceptFinding[] }>(request, {\n ...opts.llm,\n signal,\n idempotencyKey: callId,\n }),\n receipt: ({ result }) => costReceiptFromLlm(result),\n receiptFromError: costReceiptFromLlmError,\n })\n receipt = paid.receipt\n if (!paid.succeeded) throw paid.error\n const { value } = paid.value\n\n if (!value?.concepts || !Array.isArray(value.concepts)) {\n throw new Error('judge returned malformed response — expected array under \"concepts\"')\n }\n\n const findings: ConceptFinding[] = value.concepts.map((c) => ({\n concept: String(c.concept),\n present: Boolean(c.present),\n score: Math.max(0, Math.min(10, Number(c.score ?? 0))),\n evidence: String(c.evidence ?? ''),\n severity: (['critical', 'major', 'minor', 'info'] as const).includes(c.severity)\n ? c.severity\n : 'info',\n }))\n\n const presentCount = findings.filter((f) => f.present && f.score >= 7).length\n let weightSum = 0\n let weightedScoreSum = 0\n for (const f of findings) {\n const w = weightByName.get(f.concept) ?? 1\n weightSum += w\n weightedScoreSum += w * f.score\n }\n const scoreAvg =\n weightSum > 0\n ? weightedScoreSum / weightSum\n : findings.reduce((a, f) => a + f.score, 0) / Math.max(1, findings.length)\n\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: Number((scoreAvg / 10).toFixed(3)),\n presentCount,\n totalCount,\n findings,\n summary: String(value.summary ?? ''),\n durationMs: Date.now() - start,\n costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,\n available: true,\n }\n } catch (err) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount,\n findings: [],\n summary: '',\n durationMs: Date.now() - start,\n costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,\n available: false,\n error: err instanceof Error ? err.message : String(err),\n }\n }\n}\n\n/**\n * Factory: pin LLM options once, return a closure that accepts inputs.\n * Convenient for pipelines that want to share a single LlmClient config.\n */\nexport function createSemanticConceptJudge(\n options: SemanticConceptJudgeOptions = {},\n): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult> {\n return (input) => runSemanticConceptJudge(input, options)\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;AAeA,MAAM,0BAAU,IAAI,IAAmB;AAEvC,SAAS,SAAS,MAAqB;CACrC,IAAI,IAAI,QAAQ,IAAI,IAAI;CACxB,IAAI,CAAC,GAAG;EACN,IAAI,IAAI,MAAM;EACd,QAAQ,IAAI,MAAM,CAAC;CACrB;CACA,OAAO;AACT;AAEA,IAAa,sBAAb,MAAiC;CAEH;CAD5B;CACA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,QAAQ,SAAS,IAAI;EAC1B,IAAI,CAAC,WAAW,QAAQ,IAAI,CAAC,GAC3B,UAAU,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAEhD;CAEA,MAAM,OAAO,OAA+B;EAC1C,MAAM,OAAO,GAAG,KAAK,UAAU,KAAK,EAAE;EACtC,MAAM,KAAK,MAAM,mBAAmB;GAClC,eAAe,KAAK,MAAM,IAAI;EAChC,CAAC;CACH;AACF;;;;;;;;;;;;;;;;;;;;;;ACPA,IAAa,gBAAb,MAA2B;CAGG;CAF5B;CAEA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,WAAW,IAAI,oBAAoB,IAAI;CAC9C;CAEA,MAAM,OAAO,OAAe,UAA2C;EACrE,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,MAAwB;IAAE,GAAG;IAAG,QAAQ;GAAM;GACpD,MAAM,KAAK,SAAS,OAAO,GAAG;EAChC;CACF;;CAGA,UAA8B;EAC5B,IAAI,CAAC,WAAW,KAAK,IAAI,GAAG,OAAO,CAAC;EACpC,MAAM,MAAM,aAAa,KAAK,MAAM,MAAM;EAC1C,IAAI,CAAC,KAAK,OAAO,CAAC;EAClB,MAAM,MAA0B,CAAC;EACjC,KAAK,MAAM,QAAQ,IAAI,MAAM,IAAI,GAAG;GAClC,IAAI,CAAC,MAAM;GACX,IAAI;IACF,IAAI,KAAK,KAAK,MAAM,IAAI,CAAqB;GAC/C,QAAQ,CAGR;EACF;EACA,OAAO;CACT;;CAGA,QAAQ,OAAmC;EACzC,OAAO,KAAK,QAAQ,CAAC,CAAC,QAAQ,MAAM,EAAE,WAAW,KAAK;CACxD;AACF;;;;;AAiCA,SAAgB,kBAAkB,GAAmB,GAA4B;CAC/E,IAAI,EAAE,aAAa,EAAE,UAAU,OAAO;CACtC,IAAI,KAAK,KAAK,EAAE,cAAc,MAAM,EAAE,cAAc,EAAE,IAAI,KAAM,OAAO;CACvE,IAAI,EAAE,cAAc,WAAW,EAAE,cAAc,QAAQ,OAAO;CAC9D,OAAO;AACT;;;;;AAMA,SAAgB,aACd,UACA,SACA,SAAqB,CAAC,GACR;CACd,MAAM,aAAa,OAAO,cAAc;CACxC,MAAM,WAAW,IAAI,IAAI,SAAS,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAC/D,MAAM,UAAU,IAAI,IAAI,QAAQ,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAE7D,MAAM,WAA+B,CAAC;CACtC,MAAM,cAAkC,CAAC;CACzC,MAAM,YAAgC,CAAC;CACvC,MAAM,UAAmC,CAAC;CAE1C,KAAK,MAAM,CAAC,IAAI,QAAQ,SAAS;EAC/B,MAAM,OAAO,SAAS,IAAI,EAAE;EAC5B,IAAI,CAAC,MAAM;GACT,SAAS,KAAK,GAAG;GACjB;EACF;EACA,IAAI,WAAW,MAAM,GAAG,GACtB,QAAQ,KAAK;GAAE,UAAU;GAAM,SAAS;EAAI,CAAC;OAE7C,UAAU,KAAK,GAAG;CAEtB;CACA,KAAK,MAAM,CAAC,IAAI,SAAS,UACvB,IAAI,CAAC,QAAQ,IAAI,EAAE,GAAG,YAAY,KAAK,IAAI;CAE7C,OAAO;EAAE;EAAU;EAAa;EAAW;CAAQ;AACrD;;;;;;;;;;;;;;;;;;;;;;;;;ACjEA,MAAM,uBAAuB;AAE7B,MAAM,oBACJ;AACF,MAAM,aAAa;AAInB,SAAS,cAAc,MAAgD;CACrE,IAAI,CAAC,WAAW,IAAI,GAAG,OAAO,CAAC;CAC/B,MAAM,MAAwC,CAAC;CAC/C,KAAK,MAAM,SAAS,YAAY,MAAM,EAAE,eAAe,KAAK,CAAC,GAAG;EAC9D,IAAI,CAAC,MAAM,YAAY,KAAK,CAAC,MAAM,eAAe,GAAG;EACrD,MAAM,UAAU,KAAK,MAAM,MAAM,MAAM,UAAU;EACjD,IAAI,WAAW,OAAO,GAAG,IAAI,KAAK;GAAE,MAAM,MAAM;GAAM,MAAM;EAAQ,CAAC;CACvE;CACA,OAAO;AACT;AAEA,SAAS,UAAU,KAAa,KAAuB;CACrD,IAAI,CAAC,WAAW,GAAG,GAAG,OAAO,CAAC;CAC9B,MAAM,QAAkB,CAAC;CACzB,MAAM,QAAQ,CAAC,GAAG;CAClB,OAAO,MAAM,QAAQ;EACnB,MAAM,MAAM,MAAM,IAAI;EACtB,IAAI;EACJ,IAAI;GACF,UAAU,YAAY,KAAK,EAAE,eAAe,KAAK,CAAC;EACpD,QAAQ;GACN;EACF;EACA,KAAK,MAAM,KAAK,SAAS;GACvB,MAAM,OAAO,KAAK,KAAK,EAAE,IAAI;GAC7B,IAAI,EAAE,YAAY,GAAG,MAAM,KAAK,IAAI;QAC/B,IAAI,EAAE,KAAK,SAAS,QAAQ,GAAG;IAClC,MAAM,KAAK,IAAI;IACf,IAAI,MAAM,KAAK,MAAM,UAAU,KAAK,OAAO;GAC7C;EACF;CACF;CACA,OAAO;AACT;AAEA,SAAS,uBAAuB,MAAsB;CAEpD,MAAM,QADK,wBAAwB,KAAK,IACzB,CAAC,GAAG,MAAM;CAEzB,OADU,uBAAuB,KAAK,KAC/B,CAAC,GAAG,MAAM;AACnB;AAEA,SAAS,eAAe,OAAiB,MAAc,SAA2B;CAChF,IAAI,IAAI;CACR,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,aAAa,CAAC,KAAK,MAAM,WAAW,IAAI,GAAG,GAAG,QAAQ,KAAK,MAAM,KAAK,MAAM,CAAC,CAAC,CAAC;EACrF,KAAK,MAAM,OAAO,YAAY;GAC5B,IAAI,CAAC,WAAW,GAAG,GAAG;GACtB,IAAI;IACF,IAAI,SAAS,GAAG,CAAC,CAAC,YAAY,GAAG,KAAK,YAAY,GAAG,CAAC,CAAC;SAClD,KAAK;GACZ,QAAQ,CAER;EACF;CACF;CACA,OAAO;AACT;;AAGA,SAAgB,sBAAsB,QAAgD;CACpF,MAAM,SAAS,OAAO,WAAW,SAAS,EAAE,MAAM,WAChD,cAAc,IAAI,CAAC,CAAC,KAAK,OAAO;EAAE,GAAG;EAAG;CAAK,EAAE,CACjD;CACA,MAAM,QAAQ,OAAO,KAAK,MAAM,EAAE,IAAI;CAGtC,MAAM,SAAS,IAAI,IAAoB,MAAM,KAAK,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;CAC/D,MAAM,QAAQ,IAAI,IAAoB,MAAM,KAAK,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;CAC9D,MAAM,UAAU;CAChB,MAAM,QAAQ;CACd,IAAI,cAAc;CAClB,KAAK,MAAM,OAAO,OAAO,gBACvB,KAAK,MAAM,QAAQ,UAAU,KAAK,OAAO,wBAAwB,CAAC,GAAG;EACnE,eAAe;EACf,IAAI;EACJ,IAAI;GACF,OAAO,aAAa,MAAM,MAAM;EAClC,QAAQ;GACN;EACF;EACA,KAAK,MAAM,KAAK,KAAK,SAAS,OAAO,GAAG;GACtC,MAAM,IAAI,EAAE;GACZ,IAAI,CAAC,GAAG;GACR,MAAM,IAAI,EAAE,MAAM,GAAG,CAAC,CAAC,IAAI,KAAK;GAChC,MAAM,OAAO,OAAO,IAAI,CAAC;GACzB,IAAI,SAAS,KAAA,GAAW,OAAO,IAAI,GAAG,OAAO,CAAC;EAChD;EACA,KAAK,MAAM,KAAK,KAAK,SAAS,KAAK,GAAG;GACpC,MAAM,IAAI,EAAE;GACZ,IAAI,MAAM,KAAA,GAAW;GACrB,MAAM,OAAO,MAAM,IAAI,CAAC;GACxB,IAAI,SAAS,KAAA,GAAW,MAAM,IAAI,GAAG,OAAO,CAAC;EAC/C;CACF;CAIF,MAAM,yBAAS,IAAI,IAAoB;CACvC,KAAK,MAAM,KAAK,QACd,IAAI;EACF,OAAO,IAAI,EAAE,MAAM,aAAa,EAAE,MAAM,MAAM,CAAC;CACjD,QAAQ;EACN,OAAO,IAAI,EAAE,MAAM,EAAE;CACvB;CAEF,MAAM,UAAU,IAAI,IAAoB,MAAM,KAAK,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;CAChE,KAAK,MAAM,UAAU,OAAO;EAC1B,MAAM,MAAM,IAAI,OAAO,IAAI,OAAO,YAAY,OAAO,OAAO;EAC5D,KAAK,MAAM,KAAK,QAAQ;GACtB,IAAI,EAAE,SAAS,QAAQ;GACvB,IAAI,IAAI,KAAK,OAAO,IAAI,EAAE,IAAI,KAAK,EAAE,GAAG,QAAQ,IAAI,QAAQ,QAAQ,IAAI,MAAM,IAAK,CAAC;EACtF;CACF;CAEA,MAAM,UAA8B,OAAO,KAAK,MAAM;EACpD,MAAM,OAAO,OAAO,IAAI,EAAE,IAAI,KAAK;EACnC,MAAM,MAAM,EAAE,KAAK,QAAQ,gBAAgB,EAAE;EAC7C,OAAO;GACL,MAAM,EAAE;GACR,MAAM,EAAE;GACR,MAAM,EAAE;GACR,OAAO,OAAO,KAAK,MAAM,IAAI,CAAC,CAAC,SAAS;GACxC,mBAAmB,OAAO,IAAI,EAAE,IAAI,KAAK;GACzC,kBAAkB,MAAM,IAAI,EAAE,IAAI,KAAK;GACvC,aAAa,QAAQ,IAAI,EAAE,IAAI,KAAK;GACpC,eAAe,eACb,OAAO,iBAAiB,CAAC,GACzB,EAAE,MACF,OAAO,kBAAkB,EAAE,SAAS,CAAC,CACvC;GACA,oBAAoB,KAAK,MAAM,iBAAiB,KAAK,CAAC,EAAA,CAAG;GACzD,kBAAkB,WAAW,KAAK,KAAK,YAAY,CAAC;GACpD,aAAa,WAAW,KAAK,KAAK,OAAO,CAAC;GAC1C,UAAU,KAAK,SAAS,kBAAkB;GAC1C,mBAAmB,WAAW,KAAK,uBAAuB,IAAI,KAAK,KAAK,MAAM,GAAG,GAAG,CAAC;EACvF;CACF,CAAC;CACD,OAAO;EAAE,qBAAqB;EAAa;CAAQ;AACrD;AAIA,MAAM,aAAa;AAEnB,SAAS,QACP,MACA,SACA,OACA,UACA,YACA,YACA,aACA,aACA,WACgB;CAChB,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAAE,YAAY;GAAY;GAAM;GAAS;EAAM,CAAC;EAC7E,YAAY;EACZ,aAAa;EACb;EACA;EACA;EACA;EACA,eAAe,CAAC;GAAE,MAAM;GAAY,KAAK;EAAY,CAAC;EACtD,oBAAoB;EACpB;EACA;CACF;AACF;;AAGA,SAAgB,uBACd,QACA,YACkB;CAClB,MAAM,MAAwB,CAAC;CAC/B,KAAK,MAAM,KAAK,OAAO,SAAS;EAC9B,MAAM,cAAc,EAAE,oBAAoB,EAAE;EAI5C,IAHkB,cAAc,EAAE,cAAc,EAAE,kBAGhC,GAChB,IAAI,KACF,QACE,eACA,EAAE,MACF,UAAU,EAAE,KAAK,+EACjB,QACA,IACA,YACA,sIACA,EAAE,MACF,iGACF,CACF;OACK,IAAI,gBAAgB,KAAK,EAAE,cAAc,EAAE,gBAAgB,GAEhE,IAAI,KACF,QACE,eACA,EAAE,MACF,UAAU,EAAE,KAAK,gFAAgF,EAAE,YAAY,cAAc,EAAE,cAAc,IAC7I,QACA,IACA,YACA,2JACA,EAAE,MACF,sEACF,CACF;EAIF,IAAI,eAAe,KAAK,CAAC,EAAE,mBACzB,IAAI,KACF,QACE,mBACA,EAAE,MACF,UAAU,EAAE,KAAK,mFACjB,UACA,IACA,YACA,oHACA,EAAE,IACJ,CACF;EAIF,IAAI,EAAE,SAAS,YAAY,EAAE,oBAAoB,GAC/C,IAAI,KACF,QACE,UACA,EAAE,MACF,iBAAiB,EAAE,KAAK,YAAY,EAAE,kBAAkB,+BACxD,QACA,KACA,YACA,uMACA,EAAE,IACJ,CACF;EAIF,IAAI,EAAE,QAAQ,wBAAwB,CAAC,EAAE,kBACvC,IAAI,KACF,QACE,mBACA,EAAE,MACF,UAAU,EAAE,KAAK,OAAO,EAAE,MAAM,4DAChC,UACA,IACA,YACA,mFAAmF,EAAE,MAAM,mDAC3F,EAAE,IACJ,CACF;EAIF,IAAI,CAAC,EAAE,aACL,IAAI,KACF,QACE,gBACA,EAAE,MACF,UAAU,EAAE,KAAK,oBACjB,OACA,IACA,YACA,wGACA,EAAE,IACJ,CACF;EAIF,IAAI,CAAC,EAAE,UACL,IAAI,KACF,QACE,iBACA,EAAE,MACF,UAAU,EAAE,KAAK,8CACjB,OACA,KACA,YACA,iJACA,EAAE,IACJ,CACF;CAEJ;CACA,OAAO;AACT;AAIA,IAAa,oBAAb,MAAoE;CAClE,KAAc;CACd,cACE;CACF,YAAqB;CACrB,OAAgB;EAAE,MAAM;EAA0B,iBAAiB;CAAE;CACrE,UAAmB;CAEnB,MAAM,QAAQ,OAAyB,KAAgD;EACrF,MAAM,aAAa,IAAI,MAAM,+BAAc,IAAI,KAAK,EAAA,CAAE,YAAY;EAClE,IAAI,MACF,gBAAgB,MAAM,QAAQ,OAAO,eAAe,MAAM,oBAAoB,aAChF;EACA,OAAO,uBAAuB,OAAO,UAAU;CACjD;AACF;AAEA,MAAa,sBAAsB,IAAI,kBAAkB;;;AClYzD,MAAM,yBAAyB;CAC7B;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,IAAa,YAAb,MAAuB;CACrB;CACA;CAEA,YAAY,UAA4B,CAAC,GAAG;EAC1C,KAAK,UAAU,QAAQ;EACvB,KAAK,gBAAgB,QAAQ,iBAAiB;CAChD;CAEA,MAAM,MAAM,OAAmB,OAAkC;EAC/D,MAAM,MAAM,MAAM,MAAM,OAAO,KAAK;EACpC,IAAI,CAAC,KAAK,MAAM,IAAI,cAAc,OAAO,MAAM,WAAW;EAC1D,MAAM,CAAC,OAAO,QAAQ,WAAW,UAAU,MAAM,QAAQ,IAAI;GAC3D,MAAM,MAAM,EAAE,MAAM,CAAC;GACrB,MAAM,OAAO,EAAE,MAAM,CAAC;GACtB,MAAM,UAAU,KAAK;GACrB,MAAM,OAAO,KAAK;EACpB,CAAC;EACD,OAAO,KAAK,WAAW;GAAE;GAAK;GAAO;GAAQ;GAAW;EAAO,CAAC;CAClE;CAEA,WAAW,OAA2B;EACpC,MAAM,QAAkB,CAAC;EACzB,MAAM,WAAW,MAAM,MAAM,QAC1B,MAA2C,EAAE,SAAS,KACzD;EACA,MAAM,YAAY,MAAM,MAAM,QAC3B,MAA4C,EAAE,SAAS,MAC1D;EACA,MAAM,aAAa,MAAM,MAAM,QAC5B,MAA6C,EAAE,SAAS,OAC3D;EACA,MAAM,eAAe,MAAM,MAAM,QAC9B,MAA+C,EAAE,SAAS,SAC7D;EACA,MAAM,iBAAiB,WAAW,QAC/B,SAAS,KAAK,cAAc,gBAAgB,KAAK,YAAY,cAAc,IAC9E;EAEA,MAAM,UACJ,MAAM,IAAI,SAAS,SAAS,OAAO,IAAI,MAAM,IAAI,WAAW,cAAc,KAAM;EAClF,IAAI,CAAC,SAAS,MAAM,KAAK,qCAAqC;EAE9D,MAAM,eAAe,WAAW,SAC5B,WAAW,QAAQ,KAAK,SAAS,MAAM,oBAAoB,KAAK,KAAK,GAAG,CAAC,IACzE,WAAW,SACX,KAAA;EAOJ,MAAM,gBALJ,OAAO,MAAM,IAAI,SAAS,UAAU,WAChC,QACE,MAAM,IAAI,QAAQ,QAAQ,IAAI,MAAM,IAAI,QAAQ,QAAQ,MAAM,MAAM,IAAI,QAAQ,KAClF,IACA,KAAA,MAC+B,gBAAgB;EAErD,MAAM,kBAAkB,UAAU,QAAQ,SAAS,KAAK,WAAW,OAAO,CAAC,CAAC;EAC5E,MAAM,iBAAiB,UAAU,WAAW,IAAI,IAAI,kBAAkB,UAAU;EAChF,IAAI,UAAU,WAAW,GAAG,MAAM,KAAK,wBAAwB;EAE/D,MAAM,gBACJ,MAAM,UAAU,SAChB,UAAU,QAAQ,SAAS,0BAA0B,KAAK,KAAK,QAAQ,CAAC,CAAC,CAAC;EAC5E,MAAM,eAAe,gBAAgB,IAAI,QAAQ,gBAAgB,CAAC,IAAI;EACtE,IAAI,CAAC,cAAc,MAAM,KAAK,uCAAuC;EAErE,MAAM,eAAe,aAAa,QAC/B,SAAS,OAAO,KAAK,eAAe,YAAY,KAAK,aAAa,CACrE;EACA,MAAM,cAAc,aAAa,SAC7B,aAAa,QACV,KAAK,SAAS,OAAO,KAAK,eAAe,KAAK,KAAK,IAAI,GAAG,KAAK,cAAc,CAAC,GAC/E,CACF,IAAI,aAAa,SACjB,UAAU,MAAM,SACZ,yCAAyC,KAAK,KAAK,UAAU,KAAK,IAAI,CAAC,CACzE,IACA,KACA;EACN,IAAI,CAAC,aAAa,MAAM,KAAK,sCAAsC;EAEnE,MAAM,eAAe,WAAW,QAAQ,SAAS,gBAAgB,IAAI,CAAC;EACtE,MAAM,oBAAoB,eAAe,QAAQ,SAAS,gBAAgB,IAAI,CAAC;EAC/E,MAAM,YAAY,eAAe,SAAU,kBAAkB,SAAS,IAAI,IAAK;EAC/E,IAAI,kBAAkB,QACpB,MAAM,KAAK,yBAAyB,kBAAkB,OAAO,aAAa;OACvE,IAAI,CAAC,eAAe,QAAQ,MAAM,KAAK,iCAAiC;EAE7E,MAAM,mBAAmB,WAAW,SAAS,aAAa,SAAS,WAAW,SAAS;EACvF,IAAI,kBAAkB,MAAM,KAAK,YAAY,aAAa,OAAO,6BAA6B;EAE9F,MAAM,2BACJ,gBACA,aAAa,SACb,SAAS,QAAQ,SAAS,kBAAkB,KAAK,UAAU,EAAE,CAAC,CAAC,CAAC;EAClE,MAAM,eACJ,SAAS,QAAQ,SAAS,KAAK,QAAQ,KAAK,UAAU,EAAE,CAAC,CAAC,CAAC,SAC3D,MAAM,OAAO,QAAQ,UAAU,KAAK,QAAQ,KAAK,UAAU,MAAM,OAAO,CAAC,CAAC,CAAC,CAAC;EAC9E,MAAM,mBACJ,2BAA2B,iBAAiB,IACxC,IACA,4BAA4B,2BAA2B;EAC7D,MAAM,eACJ,2BAA2B,iBAAiB,IACxC,IACA,gBAAgB,2BAA2B;EACjD,IAAI,eAAe,GAAG,MAAM,KAAK,YAAY,aAAa,iBAAiB;EAe3E,OAAO;GACL;GACA;GACA;GACA;GACA;GACA;GACA;GACA;GACA;GACA,SAvBc,MAAM,OAAO,SACzB,KAAK,IACH,GAAG,MAAM,OACN,QAAQ,UAA6B,MAAM,cAAc,KAAK,CAAC,CAC/D,KAAK,UAA6B,MAAM,QAAQ,GACnD,CACF,IACA,SAAS,QAAQ,KAAK,SAAS,OAAO,KAAK,WAAW,IAAI,CAAC;GAiB7D,aAfA,MAAM,IAAI,WAAW,MAAM,IAAI,YAC3B,KAAK,IAAI,IAAI,MAAM,IAAI,UAAU,MAAM,IAAI,aAAa,GAAI,IAC5D;GAcJ;EACF;CACF;CAEA,KAAK,OAAyB;EAC5B,OAAO,kBAAkB,OAAO,KAAK,OAAO;CAC9C;CAEA,QAAgB,MAAuB;EACrC,OAAO,KAAK,cAAc,MAAM,YAAY,QAAQ,KAAK,IAAI,CAAC;CAChE;AACF;AAEA,SAAS,oBAAoB,OAAuB;CAClD,OAAO,QAAQ,IAAI,QAAQ,QAAQ,EAAE,IAAI,QAAQ,KAAK;AACxD;AAEA,SAAS,kBAAkB,MAAuB;CAChD,OAAO,qGAAqG,KAC1G,IACF;AACF;AAEA,SAAS,gBAAgB,MAAiD;CACxE,OACE,KAAK,YAAY,aAAa,QAC9B,KAAK,YAAY,YAAY,cAC7B,eAAe,KAAK,YAAY,gBAAgB,KAChD,eAAe,KAAK,YAAY,YAAY,KAC5C,KAAK,SAAS;AAElB;AAEA,SAAS,eAAe,OAAyB;CAC/C,OAAO,OAAO,UAAU,YAAY,QAAQ;AAC9C;;;;;;;;;;;;;;;;;;;;;;;AC/EA,MAAa,6BAAgE;CAC3E,QAAQ;CACR,WAAW;CACX,SAAS;AACX;AAiCA,MAAa,iCAAiC;AAE9C,MAAM,qBAAqB;AAC3B,MAAM,mBAAmB;AACzB,MAAM,uBAAuB;AAC7B,MAAM,kBAAkB;AACxB,MAAM,qBAAqB;AAC3B,MAAM,gBAAgB;AAEtB,MAAM,kBAAkB;CACtB,MAAM;CACN,sBAAsB;CACtB,UAAU,CAAC,WAAW,UAAU;CAChC,YAAY;EACV,SAAS;GAAE,MAAM;GAAU,WAAW;GAAI,WAAW;EAAI;EACzD,UAAU;GACR,MAAM;GACN,UAAU;GACV,OAAO;IACL,MAAM;IACN,sBAAsB;IACtB,UAAU;KAAC;KAAW;KAAW;KAAS;KAAY;IAAU;IAChE,YAAY;KACV,SAAS;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACxD,SAAS,EAAE,MAAM,UAAU;KAC3B,OAAO;MAAE,MAAM;MAAU,SAAS;MAAG,SAAS;KAAG;KACjD,UAAU;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACzD,UAAU;MAAE,MAAM;MAAU,MAAM;OAAC;OAAY;OAAS;OAAS;MAAM;KAAE;IAC3E;GACF;EACF;CACF;AACF;AAEA,SAAS,SAAS,MAAc,KAAa,OAAuB;CAClE,IAAI,KAAK,UAAU,KAAK,OAAO;CAC/B,OAAO,GAAG,KAAK,MAAM,GAAG,GAAG,EAAE,iBAAiB,KAAK,SAAS,IAAI,YAAY,MAAM;AACpF;AAEA,SAAS,YACP,OACA,MACQ;CACR,MAAM,aAAa,MAAM,YACtB,QAAQ,MAAM,EAAE,QAAQ,UAAU,KAAK,eAAe,CAAC,CACvD,KAAK,MAAM,aAAa,EAAE,KAAK,QAAQ,EAAE,SAAS,CAAC,CACnD,KAAK,MAAM;CAEd,MAAM,OAAO,MAAM,cAAc;CAEjC,OAAO;;;;;;;;;;EAUP,MAAM,YAAY;;EAElB,MAAM,gBAAgB,+BAA+B,MAAM,cAAc,mBAAmB,MAAM,uBAAuB,GAAG,QAAQ,GAAG;EACvI,MAAM,iBACL,KACE,GAAG,MACF,KAAK,IAAI,EAAE,KAAK,EAAE,KAAK,GAAG,EAAE,UAAU,SAAS,cAAc,EAAE,SAAS,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,KAAK,EAAE,KAAK,IACzG,CAAC,CACA,KAAK,IAAI,EAAE;;EAEZ,OAAO,qDAAqD,SAAS,MAAM,KAAK,cAAc,MAAM,EAAE,QAAQ,GAAG;EACjH,SAAS,YAAY,KAAK,gBAAgB,QAAQ,EAAE;;;;;;;;;;;;;;;;;;AAkBtD;;;;;;AASA,eAAsB,wBACpB,OACA,UAAuC,CAAC,GACH;CACrC,MAAM,QAAQ,KAAK,IAAI;CACvB,MAAM,aAAa,MAAM,iBAAiB;CAE1C,IAAI,eAAe,GACjB,OAAO;EACL,MAAM;EACN,SAAS;EACT,OAAO;EACP,cAAc;EACd,YAAY;EACZ,UAAU,CAAC;EACX,SAAS;EACT,YAAY;EACZ,SAAS;EACT,WAAW;EACX,OAAO;CACT;CAGF,MAAM,OAA8C;EAClD,OAAO,QAAQ,SAAS;EACxB,WAAW,QAAQ,aAAa;EAChC,WAAW,QAAQ,aAAa;EAChC,gBAAgB,QAAQ,kBAAkB;EAC1C,iBAAiB,QAAQ,mBAAmB;EAC5C,cAAc,QAAQ,gBAAgB;EACtC,KAAK,QAAQ,OAAO,CAAC;EACrB,YAAY,QAAQ,cAAc,IAAI,WAAW;EACjD,WAAW,QAAQ,aAAa;EAChC,UAAU,QAAQ,YAAY,CAAC;EAC/B,QAAQ,QAAQ,UAAU,IAAI,gBAAgB,CAAC,CAAC;EAChD,gBAAgB,QAAQ,kBAAkB;EAC1C,mBAAmB;GAAE,GAAG;GAA4B,GAAI,QAAQ,qBAAqB,CAAC;EAAG;CAC3F;CAMA,MAAM,oBAAoB,SAA8B;EACtD,IAAI,KAAK,mBAAmB,QAAQ,OAAO;EAC3C,IAAI,KAAK,UAAU,MAAM,OAAO,KAAK;EACrC,IAAI,KAAK,mBAAmB,cAC1B,OAAO,KAAK,kBAAkB,KAAK,cAAc,aAAa;EAEhE,OAAO;CACT;CACA,MAAM,eAAe,IAAI,IACvB,MAAM,iBAAiB,KAAK,MAAM,CAAC,EAAE,MAAM,iBAAiB,CAAC,CAAC,CAAC,CACjE;CAEA,IAAI;CACJ,IAAI;EACF,MAAM,UAAU;GACd,OAAO,KAAK;GACZ,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IAAE,MAAM;IAAiB,SAAS,YAAY,OAAO,IAAI;GAAE,CAC7D;GACA,YAAY;IAAE,MAAM;IAA0B,QAAQ;GAAgB;GACtE,aAAa;GACb,WAAW,KAAK;GAChB,WAAW,KAAK;EAClB;EACA,MAAM,OAAO,MAAM,KAAK,WAAW,YAAY;GAC7C,SAAS;GACT,OAAO,KAAK;GACZ,OAAO;GACP,OAAO,KAAK;GACZ,GAAI,OAAO,KAAK,KAAK,QAAQ,CAAC,CAAC,SAAS,IAAI,EAAE,MAAM,KAAK,SAAS,IAAI,CAAC;GACvE,eAAe,2BAA2B,SAAS,KAAK,GAAG;GAC3D,QAAQ,KAAK;GACb,UAAU,QAAQ,WAChB,YAA6D,SAAS;IACpE,GAAG,KAAK;IACR;IACA,gBAAgB;GAClB,CAAC;GACH,UAAU,EAAE,aAAa,mBAAmB,MAAM;GAClD,kBAAkB;EACpB,CAAC;EACD,UAAU,KAAK;EACf,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;EAChC,MAAM,EAAE,UAAU,KAAK;EAEvB,IAAI,CAAC,OAAO,YAAY,CAAC,MAAM,QAAQ,MAAM,QAAQ,GACnD,MAAM,IAAI,MAAM,uEAAqE;EAGvF,MAAM,WAA6B,MAAM,SAAS,KAAK,OAAO;GAC5D,SAAS,OAAO,EAAE,OAAO;GACzB,SAAS,QAAQ,EAAE,OAAO;GAC1B,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,IAAI,OAAO,EAAE,SAAS,CAAC,CAAC,CAAC;GACrD,UAAU,OAAO,EAAE,YAAY,EAAE;GACjC,UAAW;IAAC;IAAY;IAAS;IAAS;GAAM,CAAC,CAAW,SAAS,EAAE,QAAQ,IAC3E,EAAE,WACF;EACN,EAAE;EAEF,MAAM,eAAe,SAAS,QAAQ,MAAM,EAAE,WAAW,EAAE,SAAS,CAAC,CAAC,CAAC;EACvE,IAAI,YAAY;EAChB,IAAI,mBAAmB;EACvB,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,IAAI,aAAa,IAAI,EAAE,OAAO,KAAK;GACzC,aAAa;GACb,oBAAoB,IAAI,EAAE;EAC5B;EACA,MAAM,WACJ,YAAY,IACR,mBAAmB,YACnB,SAAS,QAAQ,GAAG,MAAM,IAAI,EAAE,OAAO,CAAC,IAAI,KAAK,IAAI,GAAG,SAAS,MAAM;EAE7E,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO,QAAQ,WAAW,GAAA,CAAI,QAAQ,CAAC,CAAC;GACxC;GACA;GACA;GACA,SAAS,OAAO,MAAM,WAAW,EAAE;GACnC,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,KAAK,QAAQ,cAAc,OAAO,KAAK,QAAQ;GACxD,WAAW;EACb;CACF,SAAS,KAAK;EACZ,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO;GACP,cAAc;GACd;GACA,UAAU,CAAC;GACX,SAAS;GACT,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,WAAW,CAAC,QAAQ,cAAc,QAAQ,UAAU;GAC7D,WAAW;GACX,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;EACxD;CACF;AACF;;;;;AAMA,SAAgB,2BACd,UAAuC,CAAC,GACmC;CAC3E,QAAQ,UAAU,wBAAwB,OAAO,OAAO;AAC1D"}
@@ -5,7 +5,8 @@ import "./index-C2fkZhv_.js";
5
5
  import { s as TraceStore } from "./store-CT9YIIve.js";
6
6
  import { o as LlmClientOptions } from "./llm-client-BiK4HW0u.js";
7
7
  import { f as TraceAnalystSpan } from "./store-CxJry_cs.js";
8
- import { C as AnalystContext, S as Analyst, T as AnalystFinding, u as TraceAnalystKindSpec } from "./default-registry-Cl3pHo4n.js";
8
+ import { i as AnalystFinding, n as AnalystContext, t as Analyst } from "./types-DVjczBM9.js";
9
+ import { u as TraceAnalystKindSpec } from "./default-registry-Brxr728w.js";
9
10
  import { AxAIArgs, AxAIService } from "@ax-llm/ax";
10
11
  import { z } from "zod";
11
12
  //#region src/run-score.d.ts
@@ -531,4 +532,4 @@ declare class SkillUsageAnalyst implements Analyst<SkillUsageReport> {
531
532
  declare const SKILL_USAGE_ANALYST: SkillUsageAnalyst;
532
533
  //#endregion
533
534
  export { aggregateRunScore as $, BehavioralTokenSequence as A, DEFAULT_COMPLEXITY_WEIGHTS as B, FindingSubjectKind as C, parseFindingSubject as D, findingSubjectGrammarPromptFor as E, createAnalystAi as F, createSemanticConceptJudge as G, SemanticConceptJudgeInput as H, ConceptComplexity as I, RunCriticOptions as J, runSemanticConceptJudge as K, ConceptFinding as L, SuboptimalSignal as M, computeTraceMetrics as N, renderFindingSubject as O, CreateAnalystAiConfig as P, RunScoreWeights as Q, ConceptSpec as R, FindingSubject as S, KIND_EXPECTED_SUBJECTS as T, SemanticConceptJudgeOptions as U, SEMANTIC_CONCEPT_JUDGE_VERSION as V, SemanticConceptJudgeResult as W, DEFAULT_RUN_SCORE_WEIGHTS as X, RunTrace as Y, RunScore as Z, defaultIsMaterial as _, SkillUsageScanConfig as a, FINDING_SUBJECT_KINDS as b, DEFAULT_TRACE_ANALYST_KINDS as c, IMPROVEMENT_KIND_SPEC as d, clamp01 as et, FAILURE_MODE_KIND_SPEC as f, PersistedFinding as g, FindingsStore as h, SkillUsageReport as i, SuboptimalCode as j, BehavioralMetrics as k, KNOWLEDGE_POISONING_KIND_SPEC as l, FindingsDiff as m, SkillUsageAnalyst as n, buildSkillUsageReport as o, DiffPolicy as p, RunCritic as q, SkillUsageRecord as r, emitSkillUsageFindings as s, SKILL_USAGE_ANALYST as t, KNOWLEDGE_GAP_KIND_SPEC as u, diffFindings as v, FindingSubjectStringSchema as w, FINDING_SUBJECT_SYNTAX as x, FINDING_SUBJECT_GRAMMAR_PROMPT as y, ConceptWeightStrategy as z };
534
- //# sourceMappingURL=skill-usage-BaaxFSJR.d.ts.map
535
+ //# sourceMappingURL=skill-usage-BDQVPIG1.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"skill-usage-BDQVPIG1.d.ts","names":[],"sources":["../src/run-score.ts","../src/run-critic.ts","../src/semantic-concept-judge.ts","../src/analyst/ax-service.ts","../src/trace-analyst/behavioral-metrics.ts","../src/analyst/finding-subject.ts","../src/analyst/findings-store.ts","../src/analyst/kinds/failure-mode.ts","../src/analyst/kinds/improvement.ts","../src/analyst/kinds/knowledge-gap.ts","../src/analyst/kinds/knowledge-poisoning.ts","../src/analyst/kinds/index.ts","../src/analyst/kinds/skill-usage.ts"],"mappings":";;;;;;;;;;;;UAAiB;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;cAGW,2BAA2B;iBAcxB,kBAAkB,OAAO,UAAU,UAAS,QAAQ;iBAiBpD,QAAQ;;;UCxDP;EACf,KAAK;EACL,OAAO;EACP,QAAQ;EACR,WAAW;EACX,QAAQ;;UAGO;EACf,UAAU,QAAQ;EAClB,gBAAgB;;cAYL;mBACM;mBACA;EAEjB,YAAY,UAAS;EAKf,MAAM,OAAO,YAAY,gBAAgB,QAAQ;EAYvD,WAAW,OAAO,WAAW;EAmH7B,KAAK,OAAO;UAIJ;;;;;;;;;;;;;;;;;;;;;;;;;KC/GE;UAEK;EACf;;EAEA;;EAEA;;EAEA,aAAa;;UAGE;EACf;EACA;;EAEA;EACA;EACA,UAAU;;UAGK;;EAEf;;EAEA;;EAEA,aAAa;IAAQ;IAAc;;;EAEnC,kBAAkB;;EAElB;EACA;;UAGe;EACf;EACA;;EAEA;EACA;EACA;EACA,UAAU;EACV;EACA;EACA;;EAEA;EACA;;;;;;;;KASU;cAEC,4BAA4B,OAAO;UAM/B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,MAAM;EACN,aAAa;EACb;EACA,WAAW;EACX,SAAS;;;;;;EAMT,iBAAiB;;EAEjB,oBAAoB,QAAQ,OAAO;;cAKxB;;;;;;iBAkGS,wBACpB,OAAO,2BACP,UAAS,8BACR,QAAQ;;;;;iBAsJK,2BACd,UAAS,+BACP,OAAO,8BAA8B,QAAQ;;;UC/YhC;;;EAGf;;;EAGA;;EAEA,UAAU;;EAEV;;EAEA,WAAW;;;;;;;;;;;;;;iBAeG,gBAAgB,QAAQ,wBAAwB;;;KCTpD;UAMK;EACf,MAAM;EACN;;EAEA;;EAEA,UAAU;;UAGK;;EAEf;EACA;;EAEA,gBAAgB;;EAEhB;EACA;EACA,eAAe;EACf;EACA;;EAEA;EACA;EACA,SAAS;;UAGM;EACf;EACA;EACA,sBAAsB;EACtB,uBAAuB;;;;;;iBA+DT,oBAAoB,gBAAgB,qBAAqB;;;;;;;;;;KC7E7D;EAEN;EAAwB;EAAc;;EACtC;EAAyB;;EACzB;EAAuB;;EACvB;EAAyB;;EAGzB;EAAuB;;EACvB;EAAe;;EACf;EAAkB;EAAc;;EAChC;EAAkB;;EAClB;EAAa;EAAgB;;EAC7B;EAAc;;EACd;EAAkB;;EAClB;EAAkB;;EAClB;EAAwB;;EACxB;EAAuB;;EACvB;EAAc;;EACd;EAAa;EAAgB;;EAC7B;EAAgB;;EAChB;EAAqB;;EACrB;EAAuB;;EAEvB;EAA4B;;EAC5B;EAA2B;;EAE3B;EAAiB;;KAEX,qBAAqB;cAEpB,uBAAuB,cAAc;;;;;;;;;;;;;;;;;iBA2ClC,oBAAoB,iCAAiC;;;;;;;iBAoHrD,qBAAqB,GAAG;;;;;;;;;;cA8D3B,wBAAwB,SAAS,OAAO;cA0DxC;;;;;;;;;cAYA,wBAAwB,eAAe,cAAc;;iBAoDlD,+BAA+B;;;;;;;;;;;;cAmBlC,4BAA0B,EAAA;;;;;;;;UCzZtB,yBAAyB;EACxC;;cAGW;WAGiB;mBAFX;EAEjB,YAA4B;EAItB,OAAO,eAAe,UAAU,mBAAmB;;EAQzD,WAAW;;EAkBX,QAAQ,gBAAgB;;UAOT;;EAEf,UAAU;;EAEV,aAAa;;EAEb,WAAW;;;;;EAKX,SAAS;IAAQ,UAAU;IAAkB,SAAS;;;UAGvC;;;;;;;;EAQf,cAAc,UAAU,gBAAgB,SAAS;;;;;;iBAOnC,kBAAkB,GAAG,gBAAgB,GAAG;;;;;iBAWxC,aACd,UAAU,oBACV,SAAS,oBACT,SAAQ,aACP;;;cChFU,wBAAwB;;;cCmBxB,uBAAuB;;;cCGvB,yBAAyB;;;cCPzB,+BAA+B;;;;;;;;cC1B/B,sCAAsC;;;KCCvC;;UAGK;EACf;EACA,MAAM;;EAEN;EACA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;EACA;EACA;;EAEA;;EAEA;;UAGe;EACf;EACA,SAAS;;UAGM;;EAEf;;EAEA;IAAc;IAAc,MAAM;;;EAElC;;;EAGA,kBAAkB;;EAElB;;;iBA2Ec,sBAAsB,QAAQ,uBAAuB;;iBAiHrD,uBACd,QAAQ,kBACR,qBACC;cA2HU,6BAA6B,QAAQ;WACvC;WACA;WAEA;WACA;IAAS;IAAgC;;WACzC;EAEH,QAAQ,OAAO,kBAAkB,KAAK,iBAAiB,QAAQ;;cAS1D,qBAAmB"}
@@ -1,7 +1,7 @@
1
1
  import { i as JudgeError, s as ValidationError, t as AgentEvalError } from "./errors-8YnH8WlF.js";
2
2
  import { c as costForTokenPricing, i as CostLedger, t as CostAccountingIncompleteError } from "./cost-ledger-BrJxbrMy.js";
3
3
  import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, m as stripFencedJson, u as costReceiptFromLlmError } from "./llm-client-ClPW-dWB.js";
4
- import { i as combineAbortSignals, r as clamp01 } from "./run-score-iEEAWiBY.js";
4
+ import { a as clamp01, o as combineAbortSignals, t as assertProposalFindings } from "./proposal-findings-DCawte-y.js";
5
5
  import { n as mapConcurrent } from "./concurrency-MUjT7VjM.js";
6
6
  import { C as pairedBootstrap, D as pairedSignTest, I as weightedComposite, u as confidenceInterval } from "./statistics-RwRNu2__.js";
7
7
  import { t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
@@ -3880,7 +3880,7 @@ function describeExternalScenario(scenario, label, maxChars, describe) {
3880
3880
  assertJsonValue(data, `${label} scenario '${scenario.id}'`);
3881
3881
  const serializedChars = JSON.stringify(data).length;
3882
3882
  if (serializedChars > maxChars) throw new Error(`${label} scenario '${scenario.id}' exceeds maxEvidenceChars (${serializedChars} > ${maxChars})`);
3883
- return deepFreeze({
3883
+ return deepFreeze$1({
3884
3884
  id: scenario.id,
3885
3885
  data
3886
3886
  });
@@ -3931,9 +3931,9 @@ async function scoreOneScenario(args) {
3931
3931
  function cloneExternalTextCandidate$1(candidate) {
3932
3932
  return typeof candidate === "string" ? candidate : { ...candidate };
3933
3933
  }
3934
- function deepFreeze(value) {
3934
+ function deepFreeze$1(value) {
3935
3935
  if (value && typeof value === "object") {
3936
- for (const child of Object.values(value)) deepFreeze(child);
3936
+ for (const child of Object.values(value)) deepFreeze$1(child);
3937
3937
  Object.freeze(value);
3938
3938
  }
3939
3939
  return value;
@@ -6508,6 +6508,8 @@ async function runOptimization(opts) {
6508
6508
  const candidateConcurrency = opts.candidateConcurrency ?? 1;
6509
6509
  if (typeof opts.runDir !== "string" || opts.runDir.trim().length === 0) throw new Error("runOptimization: runDir is required and must be a non-empty string");
6510
6510
  if (!Number.isInteger(candidateConcurrency) || candidateConcurrency < 1) throw new Error("runOptimization: candidateConcurrency must be a positive integer");
6511
+ const initialFindings = immutableProposalSnapshot(assertProposalFindings(opts.findings ?? [], "runOptimization initial proposal findings"), "initial findings");
6512
+ const baselineSurface = immutableProposalSnapshot(opts.baselineSurface, "baseline surface");
6511
6513
  opts.runDir = resolveRunDir(opts.runDir, opts.repo);
6512
6514
  const storage = opts.storage ?? fsCampaignStorage();
6513
6515
  const costLedger = opts.costLedger ?? createRunCostLedger({
@@ -6520,7 +6522,7 @@ async function runOptimization(opts) {
6520
6522
  const premeasuredBaseline = opts.premeasuredBaseline;
6521
6523
  const baselineCampaign = premeasuredBaseline ? validatedPremeasuredBaseline({
6522
6524
  input: premeasuredBaseline,
6523
- baselineSurface: opts.baselineSurface,
6525
+ baselineSurface,
6524
6526
  scenarios: opts.scenarios,
6525
6527
  reps,
6526
6528
  seed: opts.seed ?? 42
@@ -6528,7 +6530,7 @@ async function runOptimization(opts) {
6528
6530
  ...opts,
6529
6531
  costLedger,
6530
6532
  costPhase: "search.baseline",
6531
- dispatch: (scenario, ctx) => opts.dispatchWithSurface(opts.baselineSurface, scenario, ctx),
6533
+ dispatch: (scenario, ctx) => opts.dispatchWithSurface(baselineSurface, scenario, ctx),
6532
6534
  runDir: `${opts.runDir}/baseline`
6533
6535
  });
6534
6536
  const baselineCoverage = campaignCoverage(baselineCampaign.cells, opts.scenarios, reps, requireJudgeScore);
@@ -6538,10 +6540,10 @@ async function runOptimization(opts) {
6538
6540
  }
6539
6541
  const generations = [];
6540
6542
  const history = [];
6541
- let currentFindings = opts.findings ?? [];
6543
+ let currentFindings = initialFindings;
6542
6544
  const selectionRankKey = opts.selectionRankKey ?? ((campaign) => [campaignMeanComposite(campaign)]);
6543
- let winnerSurface = opts.baselineSurface;
6544
- let winnerSurfaceHash = surfaceHash(opts.baselineSurface);
6545
+ let winnerSurface = baselineSurface;
6546
+ let winnerSurfaceHash = surfaceHash(baselineSurface);
6545
6547
  let winnerComposite = campaignMeanComposite(baselineCampaign);
6546
6548
  let winnerRankKey = selectionRankKey(baselineCampaign);
6547
6549
  assertFiniteRankKey(winnerRankKey, "selectionRankKey for baseline");
@@ -6549,7 +6551,7 @@ async function runOptimization(opts) {
6549
6551
  let winnerOutcome = baselineOutcome;
6550
6552
  let winnerLabel;
6551
6553
  let winnerRationale;
6552
- const scored = [toParetoParent(opts.baselineSurface, winnerSurfaceHash, baselineCampaign, -1)];
6554
+ const scored = [toParetoParent(baselineSurface, winnerSurfaceHash, baselineCampaign, -1)];
6553
6555
  if (opts.analyzeGeneration && opts.maxGenerations > 0 && baselineCampaign.cells.length > 0) {
6554
6556
  const fresh = await opts.analyzeGeneration({
6555
6557
  generation: -1,
@@ -6563,31 +6565,34 @@ async function runOptimization(opts) {
6563
6565
  costLedger,
6564
6566
  costPhase: "analysis.baseline"
6565
6567
  });
6566
- if (Array.isArray(fresh)) currentFindings = fresh;
6568
+ if (!Array.isArray(fresh)) throw new TypeError("runOptimization: analyzeGeneration must return an array");
6569
+ currentFindings = immutableProposalSnapshot(assertProposalFindings(fresh, "runOptimization baseline analysis findings"), "baseline analysis findings");
6567
6570
  }
6568
6571
  for (let gen = 0; gen < opts.maxGenerations; gen++) {
6569
- if (proposer.decide?.({ history }).stop) break;
6572
+ const proposalHistory = immutableProposalSnapshot(history, "history");
6573
+ if (proposer.decide?.({ history: proposalHistory }).stop) break;
6570
6574
  const paretoParents = computeParetoFrontier(scored);
6571
6575
  const parentSurfaceHash = winnerSurfaceHash;
6572
6576
  const parentComposite = winnerComposite;
6573
- const proposed = await proposer.propose({
6574
- currentSurface: winnerSurface,
6575
- history,
6576
- findings: currentFindings,
6577
+ const proposalContext = Object.freeze({
6578
+ currentSurface: immutableProposalSnapshot(winnerSurface, "current surface"),
6579
+ history: proposalHistory,
6580
+ findings: immutableProposalSnapshot(assertProposalFindings(currentFindings, "runOptimization proposal findings"), "findings"),
6577
6581
  populationSize: opts.populationSize,
6578
6582
  generation: gen,
6579
- signal: new AbortController().signal,
6580
- baselineOutcome,
6581
- incumbentOutcome: winnerOutcome,
6582
- report: opts.report,
6583
- dataset: opts.labeledStore && opts.labeledStore !== "off" ? opts.labeledStore : void 0,
6583
+ signal: opts.signal ?? new AbortController().signal,
6584
+ baselineOutcome: immutableProposalSnapshot(baselineOutcome, "baseline outcome"),
6585
+ incumbentOutcome: immutableProposalSnapshot(winnerOutcome, "incumbent outcome"),
6584
6586
  maxImprovementShots: opts.maxImprovementShots,
6585
- paretoParents,
6587
+ paretoParents: immutableProposalSnapshot(paretoParents, "Pareto parents"),
6586
6588
  costLedger,
6587
6589
  costPhase: "search.proposal"
6588
6590
  });
6589
- if (proposed.length === 0) break;
6590
- const surfaceResults = await mapConcurrent(proposed.map((p) => isProposedCandidate(p) ? p : {
6591
+ const proposed = await proposer.propose(proposalContext);
6592
+ if (!Array.isArray(proposed)) throw new TypeError("runOptimization: proposer must return an array");
6593
+ const proposalSnapshot = immutableProposalSnapshot(proposed, "candidate outputs");
6594
+ if (proposalSnapshot.length === 0) break;
6595
+ const surfaceResults = await mapConcurrent(proposalSnapshot.map((p) => isProposedCandidate(p) ? p : {
6591
6596
  surface: p,
6592
6597
  label: "",
6593
6598
  rationale: ""
@@ -6684,11 +6689,13 @@ async function runOptimization(opts) {
6684
6689
  costLedger,
6685
6690
  costPhase: "analysis.generation"
6686
6691
  });
6687
- if (Array.isArray(fresh)) currentFindings = fresh;
6692
+ if (!Array.isArray(fresh)) throw new TypeError("runOptimization: analyzeGeneration must return an array");
6693
+ currentFindings = immutableProposalSnapshot(assertProposalFindings(fresh, "runOptimization generation analysis findings"), "generation analysis findings");
6688
6694
  }
6689
6695
  }
6690
6696
  return {
6691
6697
  generations,
6698
+ baselineSurface,
6692
6699
  winnerSurface,
6693
6700
  winnerSurfaceHash,
6694
6701
  winnerLabel,
@@ -6698,6 +6705,19 @@ async function runOptimization(opts) {
6698
6705
  cost: costLedger.summary()
6699
6706
  };
6700
6707
  }
6708
+ function immutableProposalSnapshot(value, label) {
6709
+ try {
6710
+ return deepFreeze(structuredClone(value));
6711
+ } catch (cause) {
6712
+ throw new TypeError(`runOptimization: proposal ${label} must contain snapshot-safe data`, { cause });
6713
+ }
6714
+ }
6715
+ function deepFreeze(value, seen = /* @__PURE__ */ new WeakSet()) {
6716
+ if (typeof value !== "object" || value === null || seen.has(value)) return value;
6717
+ seen.add(value);
6718
+ for (const descriptor of Object.values(Object.getOwnPropertyDescriptors(value))) if ("value" in descriptor) deepFreeze(descriptor.value, seen);
6719
+ return Object.freeze(value);
6720
+ }
6701
6721
  function validatedPremeasuredBaseline(args) {
6702
6722
  const { input } = args;
6703
6723
  if (input.surfaceHash !== surfaceHash(args.baselineSurface)) throw new Error("runOptimization: premeasured baseline surface hash does not match baselineSurface");
@@ -6807,7 +6827,8 @@ async function runImprovementLoop(opts) {
6807
6827
  dispatchTimeoutMs,
6808
6828
  costLedger
6809
6829
  });
6810
- const winnerIsBaseline = optimization.winnerSurfaceHash === surfaceHash(opts.baselineSurface);
6830
+ const baselineSurface = optimization.baselineSurface;
6831
+ const winnerIsBaseline = optimization.winnerSurfaceHash === surfaceHash(baselineSurface);
6811
6832
  const holdoutDeferred = (opts.holdout ?? "measured") === "deferred";
6812
6833
  const baselineOnHoldout = holdoutDeferred ? await runCampaign({
6813
6834
  ...opts,
@@ -6827,7 +6848,7 @@ async function runImprovementLoop(opts) {
6827
6848
  costPhase: "holdout.baseline",
6828
6849
  dispatchTimeoutMs,
6829
6850
  scenarios: opts.holdoutScenarios,
6830
- dispatch: (scenario, ctx) => opts.dispatchWithSurface(opts.baselineSurface, scenario, ctx),
6851
+ dispatch: (scenario, ctx) => opts.dispatchWithSurface(baselineSurface, scenario, ctx),
6831
6852
  runDir: `${opts.runDir}/holdout-baseline`
6832
6853
  });
6833
6854
  const winnerOnHoldout = winnerIsBaseline || holdoutDeferred ? baselineOnHoldout : await runCampaign({
@@ -6867,7 +6888,7 @@ async function runImprovementLoop(opts) {
6867
6888
  let neutralizedOnHoldout;
6868
6889
  let neutralizedSurface;
6869
6890
  if (opts.neutralize && !winnerIsBaseline && !holdoutDeferred) {
6870
- const surface = opts.neutralize(optimization.winnerSurface, opts.baselineSurface);
6891
+ const surface = opts.neutralize(optimization.winnerSurface, baselineSurface);
6871
6892
  neutralizedSurface = surface;
6872
6893
  neutralizedOnHoldout = await runCampaign({
6873
6894
  ...opts,
@@ -6920,7 +6941,7 @@ async function runImprovementLoop(opts) {
6920
6941
  costPhase: "promotion.gate",
6921
6942
  signal: new AbortController().signal
6922
6943
  });
6923
- const promotedDiff = optimization.winnerSurfaceHash === surfaceHash(opts.baselineSurface) ? "" : renderSurfaceDiff(optimization.winnerSurface, opts.baselineSurface);
6944
+ const promotedDiff = optimization.winnerSurfaceHash === surfaceHash(baselineSurface) ? "" : renderSurfaceDiff(optimization.winnerSurface, baselineSurface);
6924
6945
  let prResult;
6925
6946
  if (opts.autoOnPromote === "pr" && gateResult.decision === "ship") prResult = openAutoPr({
6926
6947
  result: winnerOnHoldout,
@@ -7814,4 +7835,4 @@ function skillOptOptimizationMethod(config) {
7814
7835
  //#endregion
7815
7836
  export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B, summarizeAgentReceiptIntegrity as Bt, detectScale as C, llmJudge as Ct, runCanaries as D, fileVerdictCache as Dt, pairHoldout as E, contentHash as Et, assertCodeSurfaceIdentity as F, minimumPairsForPairedDeltaTest as Ft, DEFAULT_MUTATION_PRIMITIVES as G, campaignBreakdown as H, JudgeParseError as Ht, assertComponentSurface as I, pairedDeltaTest as It, planCampaignRun as J, buildReflectionPrompt as K, codeSurfaceIdentityMaterial as L, BackendIntegrityError as Lt, compareOptimizationMethods as M, paretoFrontierWithCrowding as Mt, costFromLedgerSummary as N, scalarScore as Nt, composeGate as O, inMemoryVerdictCache as Ot, optimizationTokenUsageFromSummary as P, recoverTruncatedJson as Pt, inMemoryCampaignStorage as Q, componentSurfaceIdentityMaterial as R, assertRealAgentReceipts as Rt, defaultProductionGate as S, hashScenarios as St, heldoutSignificance as T, canonicalJson as Tt, campaignMeanComposite as U, surfaceHash as V, summarizeBackendIntegrity as Vt, compareRankKeys as W, createRunCostLedger as X, runCampaign as Y, fsCampaignStorage as Z, buildEvidenceVector as _, redTeamReport as _t, emitLoopProvenance as a, assertCampaignDesign as at, powerPreflight as b, Dataset as bt, provenanceRecordPath as c, campaignSplitDigest as ct, runImprovementLoop as d, REFERENCE_EQUIVALENCE_INPUT_LIMITS as dt, SearchLedgerConflictError as et, runOptimization as f, REFERENCE_EQUIVALENCE_JUDGE_VERSION as ft, gepaOptimizationMethod as g, redTeamDataset as gt, labelTrustRank as h, DEFAULT_RED_TEAM_CORPUS as ht, canonicalDigest as i, tangleTracesRoot as it, assertOptimizationResult as j, paretoFrontier as jt, externalTextOptimizationMethod as k, crowdingDistance as kt, provenanceSpansPath as l, campaignSplitDigestFromIdentities as lt, isProposedCandidate as m, runReferenceEquivalenceJudge as mt, buildLoopProvenanceRecord as n, SearchLedgerIntegrityError as nt, loopProvenanceArgsFromResult as o, assertCampaignSplitIdentity as ot, runEval as p, createReferenceEquivalenceJudge as pt, parseReflectionResponse as q, campaignMeasurementDigest as r, resolveRunDir as rt, loopProvenanceSpans as s, campaignScenarioIdentity as st, skillOptOptimizationMethod as t, SearchLedgerError as tt, verifyLoopProvenanceRecord as u, openAutoPr as ut, paretoPolicy as v, scoreRedTeamOutput as vt, dimensionRegressions as w, cachedJudge as wt, heldOutGate as x, HoldoutLockedError as xt, paretoSignificanceGate as y, toolNamesForRun as yt, renderSurfaceDiff as z, assertRealBackend as zt };
7816
7837
 
7817
- //# sourceMappingURL=skillopt-optimization-method-vvJ4bMNI.js.map
7838
+ //# sourceMappingURL=skillopt-optimization-method-BY6vKLJB.js.map