@tangle-network/agent-eval 0.123.1 → 0.123.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/CHANGELOG.md +5 -0
  2. package/README.md +151 -161
  3. package/dist/analyst/index.d.ts +9 -1
  4. package/dist/analyst/index.js +5 -5
  5. package/dist/authenticity/index.js +3 -2
  6. package/dist/authenticity/index.js.map +1 -1
  7. package/dist/benchmarks/index.d.ts +2 -1
  8. package/dist/benchmarks/index.js +6 -6
  9. package/dist/campaign/index.d.ts +29 -33
  10. package/dist/campaign/index.js +6 -6
  11. package/dist/{chunk-A5S77LSE.js → chunk-4SOQ4ND2.js} +2 -2
  12. package/dist/{chunk-VJ7T5WIO.js → chunk-5YMKIFYP.js} +3 -3
  13. package/dist/{chunk-U5CHZ5M3.js → chunk-DNVPOYUS.js} +4 -4
  14. package/dist/{chunk-6WX7CBAR.js → chunk-E3HAD4A3.js} +19 -8
  15. package/dist/chunk-E3HAD4A3.js.map +1 -0
  16. package/dist/{chunk-LBAHQOBI.js → chunk-EBDOTTZJ.js} +37 -11
  17. package/dist/chunk-EBDOTTZJ.js.map +1 -0
  18. package/dist/{chunk-XJYR7XFV.js → chunk-GC4ATIKK.js} +1 -1
  19. package/dist/chunk-GC4ATIKK.js.map +1 -0
  20. package/dist/{chunk-HZJF4IUO.js → chunk-HQY7LBV2.js} +3 -3
  21. package/dist/{chunk-NJC7U437.js → chunk-J7S4YM27.js} +6 -5
  22. package/dist/chunk-J7S4YM27.js.map +1 -0
  23. package/dist/{chunk-S3UZOQ5Y.js → chunk-LOW3U7JZ.js} +2 -2
  24. package/dist/{chunk-OYZAPX5G.js → chunk-R226UZOI.js} +2 -2
  25. package/dist/{chunk-GS3FJGUF.js → chunk-RQP5UTK5.js} +120 -14
  26. package/dist/chunk-RQP5UTK5.js.map +1 -0
  27. package/dist/{chunk-G2GPNLSX.js → chunk-WMJR67FX.js} +3 -3
  28. package/dist/{chunk-FC5NDO3E.js → chunk-WXQTVEKM.js} +3 -3
  29. package/dist/cli.js +100 -10
  30. package/dist/cli.js.map +1 -1
  31. package/dist/contract/index.d.ts +97 -5
  32. package/dist/contract/index.js +9 -7
  33. package/dist/contract/index.js.map +1 -1
  34. package/dist/control.js +3 -3
  35. package/dist/fuzz.js +3 -2
  36. package/dist/fuzz.js.map +1 -1
  37. package/dist/hosted/index.d.ts +8 -2
  38. package/dist/index.d.ts +10 -2
  39. package/dist/index.js +13 -13
  40. package/dist/index.js.map +1 -1
  41. package/dist/openapi.json +1 -1
  42. package/dist/rl.d.ts +9 -1
  43. package/dist/rl.js +4 -4
  44. package/dist/storyboard/index.js +1 -1
  45. package/dist/storyboard/index.js.map +1 -1
  46. package/dist/traces.js +3 -3
  47. package/dist/wire/index.d.ts +61 -4
  48. package/dist/wire/index.js +2 -2
  49. package/docs/adapters-observability.md +6 -6
  50. package/docs/building-doctrine.md +5 -5
  51. package/docs/concepts.md +29 -29
  52. package/docs/customer-journeys.md +80 -155
  53. package/docs/design/loop-taxonomy.md +26 -27
  54. package/docs/design.md +70 -0
  55. package/docs/distributed-driver.md +14 -14
  56. package/docs/eval-surface-map.md +11 -11
  57. package/docs/hosted-ingest-spec.md +4 -4
  58. package/docs/improvement-glossary.md +38 -38
  59. package/docs/insight-report.md +32 -27
  60. package/docs/multi-shot-optimization.md +8 -8
  61. package/docs/research-report-methodology.md +9 -9
  62. package/docs/self-improvement-map.md +13 -13
  63. package/docs/trace-analysis.md +2 -2
  64. package/docs/wire-protocol.md +16 -16
  65. package/package.json +2 -1
  66. package/dist/chunk-6WX7CBAR.js.map +0 -1
  67. package/dist/chunk-GS3FJGUF.js.map +0 -1
  68. package/dist/chunk-LBAHQOBI.js.map +0 -1
  69. package/dist/chunk-NJC7U437.js.map +0 -1
  70. package/dist/chunk-XJYR7XFV.js.map +0 -1
  71. package/docs/auto-research-loop-end-to-end.md +0 -186
  72. /package/dist/{chunk-A5S77LSE.js.map → chunk-4SOQ4ND2.js.map} +0 -0
  73. /package/dist/{chunk-VJ7T5WIO.js.map → chunk-5YMKIFYP.js.map} +0 -0
  74. /package/dist/{chunk-U5CHZ5M3.js.map → chunk-DNVPOYUS.js.map} +0 -0
  75. /package/dist/{chunk-HZJF4IUO.js.map → chunk-HQY7LBV2.js.map} +0 -0
  76. /package/dist/{chunk-S3UZOQ5Y.js.map → chunk-LOW3U7JZ.js.map} +0 -0
  77. /package/dist/{chunk-OYZAPX5G.js.map → chunk-R226UZOI.js.map} +0 -0
  78. /package/dist/{chunk-G2GPNLSX.js.map → chunk-WMJR67FX.js.map} +0 -0
  79. /package/dist/{chunk-FC5NDO3E.js.map → chunk-WXQTVEKM.js.map} +0 -0
@@ -1 +1 @@
1
- {"version":3,"sources":["../../src/authenticity/index.ts"],"sourcesContent":["/**\n * Authenticity — \"is this real, or convincing BS?\"\n *\n * Pass/build-style scoring rewards anything that compiles and renders, so an\n * agent can ship a polished frontend with a FAKE in-browser engine and zero of\n * the required on-chain/contract work, and outscore a half-finished real\n * implementation. This module scores what buildability does not: did the agent\n * actually build the intended thing on the intended infra, or fake it.\n *\n * Two layers:\n * - DETERMINISTIC `scoreAuthenticity` — calibrated by construction (no LLM,\n * trustworthy today). Structural signals over the produced files, driven by\n * a domain `AuthenticitySignals` config: required artifact present, real\n * implementation of the hard part, real infra calls, wiring, fake-shim\n * detection, mock/stub density.\n * - LLM NUANCE `scoreAuthenticityNuance` — mocked% / fake% / unique% for the\n * \"looks real but is hollow\" cases structure can't see.\n *\n * `gateRealness` is the anti-Goodhart gate: a submission missing the required\n * artifact (or faking it) is capped and cannot rank high regardless of how\n * buildable it is. Domain-agnostic; ships a Solidity/Fhenix preset.\n *\n * Input is the produced-state currency: `{ path, content }[]` — exactly what\n * `extractProducedState(...).artifacts` yields, so any consumer can feed a run's\n * produced state straight in.\n */\n\nexport interface ProducedFile {\n path: string\n content?: string\n}\n\nexport interface AuthenticitySignals {\n /** Human label for the domain (e.g. 'fhenix-fhe'). */\n label: string\n /** A file the task REQUIRES (e.g. /\\.sol$/ for an on-chain task). */\n requiredArtifact?: RegExp\n /** Vendored/3rd-party paths to exclude from required-artifact detection. */\n vendored?: RegExp\n /** Real implementation of the hard part, inside the required artifact\n * (e.g. Fhenix encrypted types + FHE.* ops). Matched against content, so it\n * fails on comments/strings only if the regex is written tightly. */\n realImpl: RegExp\n /** Real use of the intended client infra (e.g. cofhejs.encrypt() calls). */\n realInfra: RegExp\n /** Evidence the artifact is actually wired/used (e.g. contract writes). */\n wiring?: RegExp\n /** A fake shim standing in for the real thing — matched on file path AND body. */\n fakeShim: RegExp\n /** Mock/stub/TODO markers. Defaults to a generic set. */\n mock?: RegExp\n /** Score weights (default 40/25/20/15). */\n weights?: { artifact?: number; impl?: number; infra?: number; wiring?: number }\n}\n\nexport interface AuthenticityResult {\n /** Deterministic realness, 0 (BS) … 100 (real on real infra). */\n realness: number\n requiredArtifactPresent: boolean\n requiredArtifactCount: number\n usesRealImpl: boolean\n realInfra: boolean\n wired: boolean\n /** The required artifact is actually referenced/imported by other (non-artifact)\n * files — i.e. wired into the rest of the system, not dead code. Domain-agnostic:\n * a deliverable nothing else uses is suspect in any vertical. */\n artifactReferenced: boolean\n /** Convenience: the artifact is connected to the running system, via either the\n * domain wiring signal OR a structural reference. */\n artifactWired: boolean\n fakeShim: boolean\n /** mock/stub markers per 1000 LOC, capped at 100. */\n mockDensity: number\n /** Human-readable BS flags — what's missing or faked. */\n flags: string[]\n}\n\nconst DEFAULT_MOCK =\n /\\bmock|\\bfake|\\bdummy|\\bstub\\b|simulat|hardcoded|placeholder|TODO|not\\s+implemented|FIXME/i\n\nfunction basename(p: string): string {\n return p.split('/').pop() ?? p\n}\n\nfunction escapeRe(s: string): string {\n return s.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&')\n}\n\n/** Top-level symbols a source file declares (contract/library/class/etc.), used to\n * test whether other files reference the artifact. Language-agnostic keyword set. */\nfunction declaredNames(content: string): string[] {\n const names = new Set<string>()\n const re =\n /\\b(?:contract|library|interface|abstract\\s+contract|class|enum|struct|module|package)\\s+([A-Za-z_]\\w*)/g\n let m: RegExpExecArray | null\n while ((m = re.exec(content))) {\n const name = m[1]\n if (name && name.length >= 4) names.add(name)\n }\n return [...names]\n}\n\n/** Is a required artifact referenced/imported by any non-artifact file? Catches the\n * \"decorative / dead-code artifact\" facade (a real-looking deliverable nothing in\n * the running system imports, deploys, or calls). Purely structural — no domain. */\nfunction isArtifactReferenced(\n required: readonly ProducedFile[],\n others: readonly ProducedFile[],\n): boolean {\n if (!required.length || !others.length) return false\n return required.some((rf) => {\n const stem = rf.path.replace(/\\.[^.]+$/, '') // import-path stem (no ext)\n const base = basename(rf.path) // filename incl. ext\n const names = declaredNames(rf.content ?? '')\n return others.some((o) => {\n const c = o.content ?? ''\n if (!c) return false\n if (c.includes(base) || c.includes(stem)) return true // import of the path\n return names.some((n) => new RegExp(`\\\\b${escapeRe(n)}\\\\b`).test(c)) // symbol reference\n })\n })\n}\n\n/** Deterministic authenticity scan of produced files. Pure — same files in,\n * same score out. No LLM, no IO. */\nexport function scoreAuthenticity(\n files: readonly ProducedFile[],\n signals: AuthenticitySignals,\n): AuthenticityResult {\n const w = {\n artifact: signals.weights?.artifact ?? 40,\n impl: signals.weights?.impl ?? 25,\n infra: signals.weights?.infra ?? 20,\n wiring: signals.weights?.wiring ?? 15,\n }\n const mockRe = signals.mock ?? DEFAULT_MOCK\n\n const required = signals.requiredArtifact\n ? files.filter(\n (f) => signals.requiredArtifact!.test(f.path) && !(signals.vendored?.test(f.path) ?? false),\n )\n : []\n const others = signals.requiredArtifact ? files.filter((f) => !required.includes(f)) : files\n\n const requiredText = required.map((f) => f.content ?? '').join('\\n')\n const otherText = others.map((f) => f.content ?? '').join('\\n')\n const allText = files.map((f) => f.content ?? '').join('\\n')\n\n const requiredArtifactPresent = signals.requiredArtifact ? required.length > 0 : true\n // Real impl looked for in the required artifact when there is one, else anywhere.\n const usesRealImpl = signals.realImpl.test(signals.requiredArtifact ? requiredText : allText)\n const realInfra = signals.realInfra.test(allText)\n const wired = signals.wiring ? signals.wiring.test(otherText || allText) : false\n // Structural: is the required artifact actually used by the rest of the system?\n const artifactReferenced = isArtifactReferenced(required, others)\n const artifactWired = wired || artifactReferenced\n const fakeShim = files.some(\n (f) => signals.fakeShim.test(basename(f.path)) || signals.fakeShim.test(f.content ?? ''),\n )\n\n const mockHits = (\n allText.match(\n new RegExp(mockRe.source, mockRe.flags.includes('g') ? mockRe.flags : `${mockRe.flags}g`),\n ) ?? []\n ).length\n const loc = Math.max(1, allText.split('\\n').length)\n const mockDensity = Math.min(100, Math.round((mockHits / loc) * 1000))\n\n // A real-looking artifact that nothing in the system imports/deploys/calls is\n // decorative (dead code) — a common facade. We REPORT this (flag + signal) but do\n // NOT auto-penalize the score: structural reference detection is noisy (an ABI or\n // placeholder-address file makes a dead contract look \"referenced\", while a strong\n // contract-only submission looks \"dead\"), so a score penalty manufactures false\n // negatives on legitimately-partial work. Gate on it only via opts.requireArtifactWired,\n // and let the LLM-nuance layer resolve the ambiguous middle band.\n const decorativeArtifact = requiredArtifactPresent && usesRealImpl && !artifactWired\n\n let realness = 0\n if (requiredArtifactPresent) realness += w.artifact\n if (usesRealImpl) realness += w.impl\n if (realInfra) realness += w.infra\n if (wired) realness += w.wiring\n if (fakeShim) realness -= 25\n realness -= Math.min(20, mockDensity)\n realness = Math.max(0, Math.min(100, realness))\n\n const flags: string[] = []\n if (signals.requiredArtifact && !requiredArtifactPresent) {\n flags.push(\n `NO_REQUIRED_ARTIFACT: task needs ${signals.label} artifact (${signals.requiredArtifact}); none produced`,\n )\n }\n if (requiredArtifactPresent && signals.requiredArtifact && !usesRealImpl) {\n flags.push('ARTIFACT_NO_REAL_IMPL: required artifact exists but lacks the real implementation')\n }\n if (fakeShim) flags.push('FAKE_SHIM: ships a client-side stand-in simulating the real infra')\n if (!realInfra && !requiredArtifactPresent)\n flags.push('NO_REAL_INFRA: no real infra calls — cosmetic at best')\n if (mockDensity >= 8)\n flags.push(`HIGH_MOCK_DENSITY: ${mockDensity} mock/stub markers per 1000 LOC`)\n if (signals.wiring && requiredArtifactPresent && !wired)\n flags.push('NOT_WIRED: artifact exists but is never used by the client')\n if (decorativeArtifact)\n flags.push(\n 'DEAD_ARTIFACT: required artifact is not referenced/imported anywhere — decorative or dead code',\n )\n\n return {\n realness,\n requiredArtifactPresent,\n requiredArtifactCount: required.length,\n usesRealImpl,\n realInfra,\n wired,\n artifactReferenced,\n artifactWired,\n fakeShim,\n mockDensity,\n flags,\n }\n}\n\nexport interface RealnessGate {\n gated: boolean\n reason?: string\n}\n\n/** Anti-Goodhart gate: a required-artifact-missing or faked submission is\n * capped and cannot rank high regardless of buildability. */\nexport function gateRealness(\n r: AuthenticityResult,\n opts: { floor?: number; requireArtifact?: boolean; requireArtifactWired?: boolean } = {},\n): RealnessGate {\n const floor = opts.floor ?? 30\n if ((opts.requireArtifact ?? true) && !r.requiredArtifactPresent) {\n return { gated: true, reason: 'required artifact missing' }\n }\n if (r.fakeShim && !r.usesRealImpl) {\n return { gated: true, reason: 'fake shim with no real implementation' }\n }\n // Opt-in (default off): a vertical where the deliverable MUST be wired into the\n // running system can reject a decorative/dead artifact. Off by default because a\n // contract-only (incomplete-but-real) submission is legitimately partial, not fake.\n if (\n opts.requireArtifactWired &&\n r.requiredArtifactPresent &&\n r.usesRealImpl &&\n !r.artifactWired\n ) {\n return { gated: true, reason: 'required artifact present but never wired into the system' }\n }\n if (r.realness < floor)\n return { gated: true, reason: `realness ${r.realness} below floor ${floor}` }\n return { gated: false }\n}\n\n// ── LLM nuance layer ─────────────────────────────────────────────────────────\n\nexport interface AuthenticityNuance {\n /** 0 (nothing mocked) … 100 (entirely mocked). */\n mockedPct: number\n /** 0 (genuine) … 100 (a hollow facade / cargo-culted). */\n fakePct: number\n /** 0 (boilerplate/template clone) … 100 (distinctive real work). */\n uniquePct: number\n verdict: string\n}\n\n/** A minimal completion fn — inject your model caller (router/tcloud). Keeps\n * this module free of any specific LLM client. */\nexport type CompleteFn = (system: string, user: string) => Promise<string>\n\nfunction fileDigest(\n files: readonly ProducedFile[],\n opts: { maxFiles?: number; perFile?: number; prioritize?: RegExp } = {},\n): string {\n const maxFiles = opts.maxFiles ?? 14\n const perFile = opts.perFile ?? 1200\n // Lead with the required-artifact files (e.g. .sol) so a truncated digest\n // never hides the very thing the judge must assess.\n const ordered = opts.prioritize\n ? [...files].sort(\n (a, b) => Number(opts.prioritize!.test(b.path)) - Number(opts.prioritize!.test(a.path)),\n )\n : files\n return ordered\n .slice(0, maxFiles)\n .map((f) => `// ${f.path}\\n${(f.content ?? '').slice(0, perFile)}`)\n .join('\\n\\n')\n}\n\nfunction clampPct(v: unknown): number {\n const n = typeof v === 'number' ? v : Number(v)\n return Number.isFinite(n) ? Math.max(0, Math.min(100, Math.round(n))) : 0\n}\n\n/**\n * LLM nuance scoring — judges the \"looks real but is hollow\" axis structure\n * misses. Inject a `complete` caller; returns mocked/fake/unique % + a verdict.\n * Fail-soft: a bad/unparseable response yields a worst-case (fully-fake) read,\n * never a false pass.\n */\nexport async function scoreAuthenticityNuance(\n files: readonly ProducedFile[],\n complete: CompleteFn,\n opts: { intent?: string; prioritize?: RegExp } = {},\n): Promise<AuthenticityNuance> {\n const system =\n 'You audit whether an agent BUILT THE REAL THING or faked it. Be skeptical: ' +\n 'a pretty UI, cosmetic labels, simulated/in-memory stand-ins for real infra, ' +\n 'and cargo-culted imports do NOT count as real. Respond with ONLY JSON: ' +\n '{\"mockedPct\":0-100,\"fakePct\":0-100,\"uniquePct\":0-100,\"verdict\":\"one sentence\"}. ' +\n 'mockedPct = how much is mocked/stubbed; fakePct = how hollow/facade it is; ' +\n 'uniquePct = how distinctive vs boilerplate.'\n const user =\n (opts.intent ? `Intended deliverable: ${opts.intent}\\n\\n` : '') +\n `Produced files:\\n${fileDigest(files, { prioritize: opts.prioritize })}`\n try {\n const raw = await complete(system, user)\n const m = raw.match(/\\{[\\s\\S]*\\}/)\n if (!m)\n return { mockedPct: 100, fakePct: 100, uniquePct: 0, verdict: 'unparseable judge response' }\n const j = JSON.parse(m[0]) as Record<string, unknown>\n return {\n mockedPct: clampPct(j.mockedPct),\n fakePct: clampPct(j.fakePct),\n uniquePct: clampPct(j.uniquePct),\n verdict: typeof j.verdict === 'string' ? j.verdict : '',\n }\n } catch (err) {\n return {\n mockedPct: 100,\n fakePct: 100,\n uniquePct: 0,\n verdict: `judge error: ${err instanceof Error ? err.message : String(err)}`,\n }\n }\n}\n\n// ── Realness-direct LLM judge ─────────────────────────────────────────────────\n\nexport interface RealnessJudgment {\n /** 0 (facade/simulator) … 100 (real implementation on the intended infra). */\n isReal: number\n rationale: string\n}\n\n/**\n * Ask an LLM to rate realness DIRECTLY on a 0-100 scale — the axis that matched\n * human blind-labels in validation (F1 0.80→0.88 on the gray band; a fakePct/\n * hollowness proxy over-penalized \"real core + stubbed periphery\" partials, and a\n * weak judge model over-flagged — use a strong one). Domain-agnostic skeleton; the\n * consumer supplies `intent` (what the deliverable should be) and `rubric` (domain\n * specifics of real-vs-fake). Fail-closed: a bad response reads as fully fake.\n */\nexport async function judgeRealnessLlm(\n files: readonly ProducedFile[],\n complete: CompleteFn,\n opts: { intent?: string; rubric?: string; prioritize?: RegExp } = {},\n): Promise<RealnessJudgment> {\n const system =\n \"You are a skeptical auditor. Rate how REAL an agent's build is vs the intended \" +\n 'deliverable, 0-100. A genuine implementation of the HARD part on the intended ' +\n 'infrastructure is SUBSTANTIALLY REAL (>=50) even if peripheral layers are stubbed; ' +\n 'a pure simulator / facade / branded-type stand-in / no-op-stubbed dependency with ' +\n 'no real implementation is FAKE (<=25). Judge the core on its merits and note the ' +\n 'runtime. ' +\n (opts.rubric ? `Domain rubric: ${opts.rubric} ` : '') +\n 'Respond with ONLY JSON: {\"isReal\":0-100,\"why\":\"one sentence\"}.'\n const user =\n (opts.intent ? `Intended deliverable: ${opts.intent}\\n\\n` : '') +\n `Produced files:\\n${fileDigest(files, { prioritize: opts.prioritize })}`\n try {\n const raw = await complete(system, user)\n const m = raw.match(/\\{[\\s\\S]*\\}/)\n if (!m) return { isReal: 0, rationale: 'unparseable judge response' }\n const j = JSON.parse(m[0]) as Record<string, unknown>\n return {\n isReal: clampPct(j.isReal),\n rationale: typeof j.why === 'string' ? j.why : '',\n }\n } catch (err) {\n return {\n isReal: 0,\n rationale: `judge error: ${err instanceof Error ? err.message : String(err)}`,\n }\n }\n}\n\n// ── Blended pipeline: deterministic for the clean extremes, LLM for the gray band ─\n\nexport type RealnessBand = 'clean-real' | 'clean-fake' | 'gray'\n\nexport interface BlendedRealness extends AuthenticityResult {\n /** Final realness after (only-when-needed) LLM adjudication, 0…100. */\n blendedRealness: number\n band: RealnessBand\n /** True iff the LLM judge was actually consulted (gray band only). */\n consultedLlm: boolean\n /** Present iff the LLM was consulted. */\n judgment?: RealnessJudgment\n}\n\n/**\n * Score realness using the cheapest sufficient signal: trust the deterministic\n * scorer on the CLEAN extremes (obvious fakes / obviously-real-and-wired), and only\n * spend an LLM call on the GRAY band — cells that look real structurally but carry\n * fakeness markers (a fake shim, an unwired/dead artifact, high mock density) or land\n * mid-range. This caps LLM cost at the fraction of cells static analysis can't\n * resolve, which matters at multi-vertical / multi-partner scale.\n *\n * Domain-agnostic: the gray-band TRIGGER is structural; the LLM judges via the\n * consumer-supplied `intent`. Fail-closed (a bad LLM response reads as fully fake).\n */\nexport async function scoreRealnessBlended(\n files: readonly ProducedFile[],\n signals: AuthenticitySignals,\n complete: CompleteFn,\n opts: {\n intent?: string\n rubric?: string\n grayBand?: [number, number]\n mockGrayThreshold?: number\n } = {},\n): Promise<BlendedRealness> {\n const det = scoreAuthenticity(files, signals)\n const [lo, hi] = opts.grayBand ?? [30, 70]\n const mockGray = opts.mockGrayThreshold ?? 8\n\n // Structural conflict: a real artifact whose RUNTIME authenticity static analysis\n // can't settle — a fake shim is present, or it isn't wired to a real client (could\n // be a decorative contract next to a simulator, OR an incomplete-but-real build),\n // or mock density is high. Empirically (21 labeled cells) this routes 100% of the\n // deterministic errors to the LLM while leaving an error-free clean band.\n const conflict =\n det.requiredArtifactPresent &&\n det.usesRealImpl &&\n (det.fakeShim || !det.wired || det.mockDensity >= mockGray)\n const midRange = det.realness >= lo && det.realness <= hi\n\n let band: RealnessBand\n if (conflict || midRange) band = 'gray'\n else if (det.realness < lo) band = 'clean-fake'\n else band = 'clean-real'\n\n if (band !== 'gray') {\n return { ...det, blendedRealness: det.realness, band, consultedLlm: false }\n }\n\n // In the gray band the LLM read dominates (that's why we paid for it), with the\n // deterministic score as a light anchor. Weights 0.25/0.75 validated against blind\n // human labels (F1 0.88 vs 0.80 deterministic-only).\n const judgment = await judgeRealnessLlm(files, complete, {\n intent: opts.intent,\n rubric: opts.rubric,\n prioritize: signals.requiredArtifact,\n })\n const blendedRealness = Math.max(\n 0,\n Math.min(100, Math.round(0.25 * det.realness + 0.75 * judgment.isReal)),\n )\n return { ...det, blendedRealness, band, consultedLlm: true, judgment }\n}\n\n// Domain `AuthenticitySignals` (e.g. a Solidity/Fhenix preset) live in the\n// CONSUMER, not the substrate — this module stays domain-agnostic.\n"],"mappings":";;;AA6EA,IAAM,eACJ;AAEF,SAAS,SAAS,GAAmB;AACnC,SAAO,EAAE,MAAM,GAAG,EAAE,IAAI,KAAK;AAC/B;AAEA,SAAS,SAAS,GAAmB;AACnC,SAAO,EAAE,QAAQ,uBAAuB,MAAM;AAChD;AAIA,SAAS,cAAc,SAA2B;AAChD,QAAM,QAAQ,oBAAI,IAAY;AAC9B,QAAM,KACJ;AACF,MAAI;AACJ,SAAQ,IAAI,GAAG,KAAK,OAAO,GAAI;AAC7B,UAAM,OAAO,EAAE,CAAC;AAChB,QAAI,QAAQ,KAAK,UAAU,EAAG,OAAM,IAAI,IAAI;AAAA,EAC9C;AACA,SAAO,CAAC,GAAG,KAAK;AAClB;AAKA,SAAS,qBACP,UACA,QACS;AACT,MAAI,CAAC,SAAS,UAAU,CAAC,OAAO,OAAQ,QAAO;AAC/C,SAAO,SAAS,KAAK,CAAC,OAAO;AAC3B,UAAM,OAAO,GAAG,KAAK,QAAQ,YAAY,EAAE;AAC3C,UAAM,OAAO,SAAS,GAAG,IAAI;AAC7B,UAAM,QAAQ,cAAc,GAAG,WAAW,EAAE;AAC5C,WAAO,OAAO,KAAK,CAAC,MAAM;AACxB,YAAM,IAAI,EAAE,WAAW;AACvB,UAAI,CAAC,EAAG,QAAO;AACf,UAAI,EAAE,SAAS,IAAI,KAAK,EAAE,SAAS,IAAI,EAAG,QAAO;AACjD,aAAO,MAAM,KAAK,CAAC,MAAM,IAAI,OAAO,MAAM,SAAS,CAAC,CAAC,KAAK,EAAE,KAAK,CAAC,CAAC;AAAA,IACrE,CAAC;AAAA,EACH,CAAC;AACH;AAIO,SAAS,kBACd,OACA,SACoB;AACpB,QAAM,IAAI;AAAA,IACR,UAAU,QAAQ,SAAS,YAAY;AAAA,IACvC,MAAM,QAAQ,SAAS,QAAQ;AAAA,IAC/B,OAAO,QAAQ,SAAS,SAAS;AAAA,IACjC,QAAQ,QAAQ,SAAS,UAAU;AAAA,EACrC;AACA,QAAM,SAAS,QAAQ,QAAQ;AAE/B,QAAM,WAAW,QAAQ,mBACrB,MAAM;AAAA,IACJ,CAAC,MAAM,QAAQ,iBAAkB,KAAK,EAAE,IAAI,KAAK,EAAE,QAAQ,UAAU,KAAK,EAAE,IAAI,KAAK;AAAA,EACvF,IACA,CAAC;AACL,QAAM,SAAS,QAAQ,mBAAmB,MAAM,OAAO,CAAC,MAAM,CAAC,SAAS,SAAS,CAAC,CAAC,IAAI;AAEvF,QAAM,eAAe,SAAS,IAAI,CAAC,MAAM,EAAE,WAAW,EAAE,EAAE,KAAK,IAAI;AACnE,QAAM,YAAY,OAAO,IAAI,CAAC,MAAM,EAAE,WAAW,EAAE,EAAE,KAAK,IAAI;AAC9D,QAAM,UAAU,MAAM,IAAI,CAAC,MAAM,EAAE,WAAW,EAAE,EAAE,KAAK,IAAI;AAE3D,QAAM,0BAA0B,QAAQ,mBAAmB,SAAS,SAAS,IAAI;AAEjF,QAAM,eAAe,QAAQ,SAAS,KAAK,QAAQ,mBAAmB,eAAe,OAAO;AAC5F,QAAM,YAAY,QAAQ,UAAU,KAAK,OAAO;AAChD,QAAM,QAAQ,QAAQ,SAAS,QAAQ,OAAO,KAAK,aAAa,OAAO,IAAI;AAE3E,QAAM,qBAAqB,qBAAqB,UAAU,MAAM;AAChE,QAAM,gBAAgB,SAAS;AAC/B,QAAM,WAAW,MAAM;AAAA,IACrB,CAAC,MAAM,QAAQ,SAAS,KAAK,SAAS,EAAE,IAAI,CAAC,KAAK,QAAQ,SAAS,KAAK,EAAE,WAAW,EAAE;AAAA,EACzF;AAEA,QAAM,YACJ,QAAQ;AAAA,IACN,IAAI,OAAO,OAAO,QAAQ,OAAO,MAAM,SAAS,GAAG,IAAI,OAAO,QAAQ,GAAG,OAAO,KAAK,GAAG;AAAA,EAC1F,KAAK,CAAC,GACN;AACF,QAAM,MAAM,KAAK,IAAI,GAAG,QAAQ,MAAM,IAAI,EAAE,MAAM;AAClD,QAAM,cAAc,KAAK,IAAI,KAAK,KAAK,MAAO,WAAW,MAAO,GAAI,CAAC;AASrE,QAAM,qBAAqB,2BAA2B,gBAAgB,CAAC;AAEvE,MAAI,WAAW;AACf,MAAI,wBAAyB,aAAY,EAAE;AAC3C,MAAI,aAAc,aAAY,EAAE;AAChC,MAAI,UAAW,aAAY,EAAE;AAC7B,MAAI,MAAO,aAAY,EAAE;AACzB,MAAI,SAAU,aAAY;AAC1B,cAAY,KAAK,IAAI,IAAI,WAAW;AACpC,aAAW,KAAK,IAAI,GAAG,KAAK,IAAI,KAAK,QAAQ,CAAC;AAE9C,QAAM,QAAkB,CAAC;AACzB,MAAI,QAAQ,oBAAoB,CAAC,yBAAyB;AACxD,UAAM;AAAA,MACJ,oCAAoC,QAAQ,KAAK,cAAc,QAAQ,gBAAgB;AAAA,IACzF;AAAA,EACF;AACA,MAAI,2BAA2B,QAAQ,oBAAoB,CAAC,cAAc;AACxE,UAAM,KAAK,mFAAmF;AAAA,EAChG;AACA,MAAI,SAAU,OAAM,KAAK,mEAAmE;AAC5F,MAAI,CAAC,aAAa,CAAC;AACjB,UAAM,KAAK,4DAAuD;AACpE,MAAI,eAAe;AACjB,UAAM,KAAK,sBAAsB,WAAW,iCAAiC;AAC/E,MAAI,QAAQ,UAAU,2BAA2B,CAAC;AAChD,UAAM,KAAK,4DAA4D;AACzE,MAAI;AACF,UAAM;AAAA,MACJ;AAAA,IACF;AAEF,SAAO;AAAA,IACL;AAAA,IACA;AAAA,IACA,uBAAuB,SAAS;AAAA,IAChC;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,EACF;AACF;AASO,SAAS,aACd,GACA,OAAsF,CAAC,GACzE;AACd,QAAM,QAAQ,KAAK,SAAS;AAC5B,OAAK,KAAK,mBAAmB,SAAS,CAAC,EAAE,yBAAyB;AAChE,WAAO,EAAE,OAAO,MAAM,QAAQ,4BAA4B;AAAA,EAC5D;AACA,MAAI,EAAE,YAAY,CAAC,EAAE,cAAc;AACjC,WAAO,EAAE,OAAO,MAAM,QAAQ,wCAAwC;AAAA,EACxE;AAIA,MACE,KAAK,wBACL,EAAE,2BACF,EAAE,gBACF,CAAC,EAAE,eACH;AACA,WAAO,EAAE,OAAO,MAAM,QAAQ,4DAA4D;AAAA,EAC5F;AACA,MAAI,EAAE,WAAW;AACf,WAAO,EAAE,OAAO,MAAM,QAAQ,YAAY,EAAE,QAAQ,gBAAgB,KAAK,GAAG;AAC9E,SAAO,EAAE,OAAO,MAAM;AACxB;AAkBA,SAAS,WACP,OACA,OAAqE,CAAC,GAC9D;AACR,QAAM,WAAW,KAAK,YAAY;AAClC,QAAM,UAAU,KAAK,WAAW;AAGhC,QAAM,UAAU,KAAK,aACjB,CAAC,GAAG,KAAK,EAAE;AAAA,IACT,CAAC,GAAG,MAAM,OAAO,KAAK,WAAY,KAAK,EAAE,IAAI,CAAC,IAAI,OAAO,KAAK,WAAY,KAAK,EAAE,IAAI,CAAC;AAAA,EACxF,IACA;AACJ,SAAO,QACJ,MAAM,GAAG,QAAQ,EACjB,IAAI,CAAC,MAAM,MAAM,EAAE,IAAI;AAAA,GAAM,EAAE,WAAW,IAAI,MAAM,GAAG,OAAO,CAAC,EAAE,EACjE,KAAK,MAAM;AAChB;AAEA,SAAS,SAAS,GAAoB;AACpC,QAAM,IAAI,OAAO,MAAM,WAAW,IAAI,OAAO,CAAC;AAC9C,SAAO,OAAO,SAAS,CAAC,IAAI,KAAK,IAAI,GAAG,KAAK,IAAI,KAAK,KAAK,MAAM,CAAC,CAAC,CAAC,IAAI;AAC1E;AAQA,eAAsB,wBACpB,OACA,UACA,OAAiD,CAAC,GACrB;AAC7B,QAAM,SACJ;AAMF,QAAM,QACH,KAAK,SAAS,yBAAyB,KAAK,MAAM;AAAA;AAAA,IAAS,MAC5D;AAAA,EAAoB,WAAW,OAAO,EAAE,YAAY,KAAK,WAAW,CAAC,CAAC;AACxE,MAAI;AACF,UAAM,MAAM,MAAM,SAAS,QAAQ,IAAI;AACvC,UAAM,IAAI,IAAI,MAAM,aAAa;AACjC,QAAI,CAAC;AACH,aAAO,EAAE,WAAW,KAAK,SAAS,KAAK,WAAW,GAAG,SAAS,6BAA6B;AAC7F,UAAM,IAAI,KAAK,MAAM,EAAE,CAAC,CAAC;AACzB,WAAO;AAAA,MACL,WAAW,SAAS,EAAE,SAAS;AAAA,MAC/B,SAAS,SAAS,EAAE,OAAO;AAAA,MAC3B,WAAW,SAAS,EAAE,SAAS;AAAA,MAC/B,SAAS,OAAO,EAAE,YAAY,WAAW,EAAE,UAAU;AAAA,IACvD;AAAA,EACF,SAAS,KAAK;AACZ,WAAO;AAAA,MACL,WAAW;AAAA,MACX,SAAS;AAAA,MACT,WAAW;AAAA,MACX,SAAS,gBAAgB,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG,CAAC;AAAA,IAC3E;AAAA,EACF;AACF;AAkBA,eAAsB,iBACpB,OACA,UACA,OAAkE,CAAC,GACxC;AAC3B,QAAM,SACJ,kaAMC,KAAK,SAAS,kBAAkB,KAAK,MAAM,MAAM,MAClD;AACF,QAAM,QACH,KAAK,SAAS,yBAAyB,KAAK,MAAM;AAAA;AAAA,IAAS,MAC5D;AAAA,EAAoB,WAAW,OAAO,EAAE,YAAY,KAAK,WAAW,CAAC,CAAC;AACxE,MAAI;AACF,UAAM,MAAM,MAAM,SAAS,QAAQ,IAAI;AACvC,UAAM,IAAI,IAAI,MAAM,aAAa;AACjC,QAAI,CAAC,EAAG,QAAO,EAAE,QAAQ,GAAG,WAAW,6BAA6B;AACpE,UAAM,IAAI,KAAK,MAAM,EAAE,CAAC,CAAC;AACzB,WAAO;AAAA,MACL,QAAQ,SAAS,EAAE,MAAM;AAAA,MACzB,WAAW,OAAO,EAAE,QAAQ,WAAW,EAAE,MAAM;AAAA,IACjD;AAAA,EACF,SAAS,KAAK;AACZ,WAAO;AAAA,MACL,QAAQ;AAAA,MACR,WAAW,gBAAgB,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG,CAAC;AAAA,IAC7E;AAAA,EACF;AACF;AA2BA,eAAsB,qBACpB,OACA,SACA,UACA,OAKI,CAAC,GACqB;AAC1B,QAAM,MAAM,kBAAkB,OAAO,OAAO;AAC5C,QAAM,CAAC,IAAI,EAAE,IAAI,KAAK,YAAY,CAAC,IAAI,EAAE;AACzC,QAAM,WAAW,KAAK,qBAAqB;AAO3C,QAAM,WACJ,IAAI,2BACJ,IAAI,iBACH,IAAI,YAAY,CAAC,IAAI,SAAS,IAAI,eAAe;AACpD,QAAM,WAAW,IAAI,YAAY,MAAM,IAAI,YAAY;AAEvD,MAAI;AACJ,MAAI,YAAY,SAAU,QAAO;AAAA,WACxB,IAAI,WAAW,GAAI,QAAO;AAAA,MAC9B,QAAO;AAEZ,MAAI,SAAS,QAAQ;AACnB,WAAO,EAAE,GAAG,KAAK,iBAAiB,IAAI,UAAU,MAAM,cAAc,MAAM;AAAA,EAC5E;AAKA,QAAM,WAAW,MAAM,iBAAiB,OAAO,UAAU;AAAA,IACvD,QAAQ,KAAK;AAAA,IACb,QAAQ,KAAK;AAAA,IACb,YAAY,QAAQ;AAAA,EACtB,CAAC;AACD,QAAM,kBAAkB,KAAK;AAAA,IAC3B;AAAA,IACA,KAAK,IAAI,KAAK,KAAK,MAAM,OAAO,IAAI,WAAW,OAAO,SAAS,MAAM,CAAC;AAAA,EACxE;AACA,SAAO,EAAE,GAAG,KAAK,iBAAiB,MAAM,cAAc,MAAM,SAAS;AACvE;","names":[]}
1
+ {"version":3,"sources":["../../src/authenticity/index.ts"],"sourcesContent":["/**\n * Authenticity — \"is this real, or convincing BS?\"\n *\n * Pass/build-style scoring rewards anything that compiles and renders, so an\n * agent can ship a polished frontend with a FAKE in-browser engine and zero of\n * the required on-chain/contract work, and outscore a half-finished real\n * implementation. This module scores what buildability does not: did the agent\n * actually build the intended thing on the intended infra, or fake it.\n *\n * Two layers:\n * - DETERMINISTIC `scoreAuthenticity` — calibrated by construction (no LLM,\n * trustworthy today). Structural signals over the produced files, driven by\n * a domain `AuthenticitySignals` config: required artifact present, real\n * implementation of the hard part, real infra calls, wiring, fake-shim\n * detection, mock/stub density.\n * - LLM NUANCE `scoreAuthenticityNuance` — mocked% / fake% / unique% for the\n * \"looks real but is hollow\" cases structure can't see.\n *\n * `gateRealness` is the anti-Goodhart gate: a submission missing the required\n * artifact (or faking it) is capped and cannot rank high regardless of how\n * buildable it is. Domain-agnostic; ships a Solidity/Fhenix preset.\n *\n * Input is the produced-state currency: `{ path, content }[]` — exactly what\n * `extractProducedState(...).artifacts` yields, so any consumer can feed a run's\n * produced state straight in.\n */\n\nexport interface ProducedFile {\n path: string\n content?: string\n}\n\nexport interface AuthenticitySignals {\n /** Human label for the domain (e.g. 'fhenix-fhe'). */\n label: string\n /** A file the task REQUIRES (e.g. /\\.sol$/ for an on-chain task). */\n requiredArtifact?: RegExp\n /** Vendored/3rd-party paths to exclude from required-artifact detection. */\n vendored?: RegExp\n /** Real implementation of the hard part, inside the required artifact\n * (e.g. Fhenix encrypted types + FHE.* ops). Matched against content, so it\n * fails on comments/strings only if the regex is written tightly. */\n realImpl: RegExp\n /** Real use of the intended client infra (e.g. cofhejs.encrypt() calls). */\n realInfra: RegExp\n /** Evidence the artifact is actually wired/used (e.g. contract writes). */\n wiring?: RegExp\n /** A fake shim standing in for the real thing — matched on file path AND body. */\n fakeShim: RegExp\n /** Mock/stub/TODO markers. Defaults to a generic set. */\n mock?: RegExp\n /** Score weights (default 40/25/20/15). */\n weights?: { artifact?: number; impl?: number; infra?: number; wiring?: number }\n}\n\nexport interface AuthenticityResult {\n /** Deterministic realness, 0 (BS) … 100 (real on real infra). */\n realness: number\n requiredArtifactPresent: boolean\n requiredArtifactCount: number\n usesRealImpl: boolean\n realInfra: boolean\n wired: boolean\n /** The required artifact is actually referenced/imported by other (non-artifact)\n * files — i.e. wired into the rest of the system, not dead code. Domain-agnostic:\n * a deliverable nothing else uses is suspect in any vertical. */\n artifactReferenced: boolean\n /** Convenience: the artifact is connected to the running system, via either the\n * domain wiring signal OR a structural reference. */\n artifactWired: boolean\n fakeShim: boolean\n /** mock/stub markers per 1000 LOC, capped at 100. */\n mockDensity: number\n /** Human-readable BS flags — what's missing or faked. */\n flags: string[]\n}\n\nconst DEFAULT_MOCK =\n /\\bmock|\\bfake|\\bdummy|\\bstub\\b|simulat|hardcoded|placeholder|TODO|not\\s+implemented|FIXME/i\n\nfunction basename(p: string): string {\n return p.split('/').pop() ?? p\n}\n\nfunction escapeRe(s: string): string {\n return s.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&')\n}\n\n/** Top-level symbols a source file declares (contract/library/class/etc.), used to\n * test whether other files reference the artifact. Language-agnostic keyword set. */\nfunction declaredNames(content: string): string[] {\n const names = new Set<string>()\n const re =\n /\\b(?:contract|library|interface|abstract\\s+contract|class|enum|struct|module|package)\\s+([A-Za-z_]\\w*)/g\n let m = re.exec(content)\n while (m) {\n const name = m[1]\n if (name && name.length >= 4) names.add(name)\n m = re.exec(content)\n }\n return [...names]\n}\n\n/** Is a required artifact referenced/imported by any non-artifact file? Catches the\n * \"decorative / dead-code artifact\" facade (a real-looking deliverable nothing in\n * the running system imports, deploys, or calls). Purely structural — no domain. */\nfunction isArtifactReferenced(\n required: readonly ProducedFile[],\n others: readonly ProducedFile[],\n): boolean {\n if (!required.length || !others.length) return false\n return required.some((rf) => {\n const stem = rf.path.replace(/\\.[^.]+$/, '') // import-path stem (no ext)\n const base = basename(rf.path) // filename incl. ext\n const names = declaredNames(rf.content ?? '')\n return others.some((o) => {\n const c = o.content ?? ''\n if (!c) return false\n if (c.includes(base) || c.includes(stem)) return true // import of the path\n return names.some((n) => new RegExp(`\\\\b${escapeRe(n)}\\\\b`).test(c)) // symbol reference\n })\n })\n}\n\n/** Deterministic authenticity scan of produced files. Pure — same files in,\n * same score out. No LLM, no IO. */\nexport function scoreAuthenticity(\n files: readonly ProducedFile[],\n signals: AuthenticitySignals,\n): AuthenticityResult {\n const w = {\n artifact: signals.weights?.artifact ?? 40,\n impl: signals.weights?.impl ?? 25,\n infra: signals.weights?.infra ?? 20,\n wiring: signals.weights?.wiring ?? 15,\n }\n const mockRe = signals.mock ?? DEFAULT_MOCK\n\n const required = signals.requiredArtifact\n ? files.filter(\n (f) => signals.requiredArtifact!.test(f.path) && !(signals.vendored?.test(f.path) ?? false),\n )\n : []\n const others = signals.requiredArtifact ? files.filter((f) => !required.includes(f)) : files\n\n const requiredText = required.map((f) => f.content ?? '').join('\\n')\n const otherText = others.map((f) => f.content ?? '').join('\\n')\n const allText = files.map((f) => f.content ?? '').join('\\n')\n\n const requiredArtifactPresent = signals.requiredArtifact ? required.length > 0 : true\n // Real impl looked for in the required artifact when there is one, else anywhere.\n const usesRealImpl = signals.realImpl.test(signals.requiredArtifact ? requiredText : allText)\n const realInfra = signals.realInfra.test(allText)\n const wired = signals.wiring ? signals.wiring.test(otherText || allText) : false\n // Structural: is the required artifact actually used by the rest of the system?\n const artifactReferenced = isArtifactReferenced(required, others)\n const artifactWired = wired || artifactReferenced\n const fakeShim = files.some(\n (f) => signals.fakeShim.test(basename(f.path)) || signals.fakeShim.test(f.content ?? ''),\n )\n\n const mockHits = (\n allText.match(\n new RegExp(mockRe.source, mockRe.flags.includes('g') ? mockRe.flags : `${mockRe.flags}g`),\n ) ?? []\n ).length\n const loc = Math.max(1, allText.split('\\n').length)\n const mockDensity = Math.min(100, Math.round((mockHits / loc) * 1000))\n\n // A real-looking artifact that nothing in the system imports/deploys/calls is\n // decorative (dead code) — a common facade. We REPORT this (flag + signal) but do\n // NOT auto-penalize the score: structural reference detection is noisy (an ABI or\n // placeholder-address file makes a dead contract look \"referenced\", while a strong\n // contract-only submission looks \"dead\"), so a score penalty manufactures false\n // negatives on legitimately-partial work. Gate on it only via opts.requireArtifactWired,\n // and let the LLM-nuance layer resolve the ambiguous middle band.\n const decorativeArtifact = requiredArtifactPresent && usesRealImpl && !artifactWired\n\n let realness = 0\n if (requiredArtifactPresent) realness += w.artifact\n if (usesRealImpl) realness += w.impl\n if (realInfra) realness += w.infra\n if (wired) realness += w.wiring\n if (fakeShim) realness -= 25\n realness -= Math.min(20, mockDensity)\n realness = Math.max(0, Math.min(100, realness))\n\n const flags: string[] = []\n if (signals.requiredArtifact && !requiredArtifactPresent) {\n flags.push(\n `NO_REQUIRED_ARTIFACT: task needs ${signals.label} artifact (${signals.requiredArtifact}); none produced`,\n )\n }\n if (requiredArtifactPresent && signals.requiredArtifact && !usesRealImpl) {\n flags.push('ARTIFACT_NO_REAL_IMPL: required artifact exists but lacks the real implementation')\n }\n if (fakeShim) flags.push('FAKE_SHIM: ships a client-side stand-in simulating the real infra')\n if (!realInfra && !requiredArtifactPresent)\n flags.push('NO_REAL_INFRA: no real infra calls — cosmetic at best')\n if (mockDensity >= 8)\n flags.push(`HIGH_MOCK_DENSITY: ${mockDensity} mock/stub markers per 1000 LOC`)\n if (signals.wiring && requiredArtifactPresent && !wired)\n flags.push('NOT_WIRED: artifact exists but is never used by the client')\n if (decorativeArtifact)\n flags.push(\n 'DEAD_ARTIFACT: required artifact is not referenced/imported anywhere — decorative or dead code',\n )\n\n return {\n realness,\n requiredArtifactPresent,\n requiredArtifactCount: required.length,\n usesRealImpl,\n realInfra,\n wired,\n artifactReferenced,\n artifactWired,\n fakeShim,\n mockDensity,\n flags,\n }\n}\n\nexport interface RealnessGate {\n gated: boolean\n reason?: string\n}\n\n/** Anti-Goodhart gate: a required-artifact-missing or faked submission is\n * capped and cannot rank high regardless of buildability. */\nexport function gateRealness(\n r: AuthenticityResult,\n opts: { floor?: number; requireArtifact?: boolean; requireArtifactWired?: boolean } = {},\n): RealnessGate {\n const floor = opts.floor ?? 30\n if ((opts.requireArtifact ?? true) && !r.requiredArtifactPresent) {\n return { gated: true, reason: 'required artifact missing' }\n }\n if (r.fakeShim && !r.usesRealImpl) {\n return { gated: true, reason: 'fake shim with no real implementation' }\n }\n // Opt-in (default off): a vertical where the deliverable MUST be wired into the\n // running system can reject a decorative/dead artifact. Off by default because a\n // contract-only (incomplete-but-real) submission is legitimately partial, not fake.\n if (\n opts.requireArtifactWired &&\n r.requiredArtifactPresent &&\n r.usesRealImpl &&\n !r.artifactWired\n ) {\n return { gated: true, reason: 'required artifact present but never wired into the system' }\n }\n if (r.realness < floor)\n return { gated: true, reason: `realness ${r.realness} below floor ${floor}` }\n return { gated: false }\n}\n\n// ── LLM nuance layer ─────────────────────────────────────────────────────────\n\nexport interface AuthenticityNuance {\n /** 0 (nothing mocked) … 100 (entirely mocked). */\n mockedPct: number\n /** 0 (genuine) … 100 (a hollow facade / cargo-culted). */\n fakePct: number\n /** 0 (boilerplate/template clone) … 100 (distinctive real work). */\n uniquePct: number\n verdict: string\n}\n\n/** A minimal completion fn — inject your model caller (router/tcloud). Keeps\n * this module free of any specific LLM client. */\nexport type CompleteFn = (system: string, user: string) => Promise<string>\n\nfunction fileDigest(\n files: readonly ProducedFile[],\n opts: { maxFiles?: number; perFile?: number; prioritize?: RegExp } = {},\n): string {\n const maxFiles = opts.maxFiles ?? 14\n const perFile = opts.perFile ?? 1200\n // Lead with the required-artifact files (e.g. .sol) so a truncated digest\n // never hides the very thing the judge must assess.\n const ordered = opts.prioritize\n ? [...files].sort(\n (a, b) => Number(opts.prioritize!.test(b.path)) - Number(opts.prioritize!.test(a.path)),\n )\n : files\n return ordered\n .slice(0, maxFiles)\n .map((f) => `// ${f.path}\\n${(f.content ?? '').slice(0, perFile)}`)\n .join('\\n\\n')\n}\n\nfunction clampPct(v: unknown): number {\n const n = typeof v === 'number' ? v : Number(v)\n return Number.isFinite(n) ? Math.max(0, Math.min(100, Math.round(n))) : 0\n}\n\n/**\n * LLM nuance scoring — judges the \"looks real but is hollow\" axis structure\n * misses. Inject a `complete` caller; returns mocked/fake/unique % + a verdict.\n * Fail-soft: a bad/unparseable response yields a worst-case (fully-fake) read,\n * never a false pass.\n */\nexport async function scoreAuthenticityNuance(\n files: readonly ProducedFile[],\n complete: CompleteFn,\n opts: { intent?: string; prioritize?: RegExp } = {},\n): Promise<AuthenticityNuance> {\n const system =\n 'You audit whether an agent BUILT THE REAL THING or faked it. Be skeptical: ' +\n 'a pretty UI, cosmetic labels, simulated/in-memory stand-ins for real infra, ' +\n 'and cargo-culted imports do NOT count as real. Respond with ONLY JSON: ' +\n '{\"mockedPct\":0-100,\"fakePct\":0-100,\"uniquePct\":0-100,\"verdict\":\"one sentence\"}. ' +\n 'mockedPct = how much is mocked/stubbed; fakePct = how hollow/facade it is; ' +\n 'uniquePct = how distinctive vs boilerplate.'\n const user =\n (opts.intent ? `Intended deliverable: ${opts.intent}\\n\\n` : '') +\n `Produced files:\\n${fileDigest(files, { prioritize: opts.prioritize })}`\n try {\n const raw = await complete(system, user)\n const m = raw.match(/\\{[\\s\\S]*\\}/)\n if (!m)\n return { mockedPct: 100, fakePct: 100, uniquePct: 0, verdict: 'unparseable judge response' }\n const j = JSON.parse(m[0]) as Record<string, unknown>\n return {\n mockedPct: clampPct(j.mockedPct),\n fakePct: clampPct(j.fakePct),\n uniquePct: clampPct(j.uniquePct),\n verdict: typeof j.verdict === 'string' ? j.verdict : '',\n }\n } catch (err) {\n return {\n mockedPct: 100,\n fakePct: 100,\n uniquePct: 0,\n verdict: `judge error: ${err instanceof Error ? err.message : String(err)}`,\n }\n }\n}\n\n// ── Realness-direct LLM judge ─────────────────────────────────────────────────\n\nexport interface RealnessJudgment {\n /** 0 (facade/simulator) … 100 (real implementation on the intended infra). */\n isReal: number\n rationale: string\n}\n\n/**\n * Ask an LLM to rate realness DIRECTLY on a 0-100 scale — the axis that matched\n * human blind-labels in validation (F1 0.80→0.88 on the gray band; a fakePct/\n * hollowness proxy over-penalized \"real core + stubbed periphery\" partials, and a\n * weak judge model over-flagged — use a strong one). Domain-agnostic skeleton; the\n * consumer supplies `intent` (what the deliverable should be) and `rubric` (domain\n * specifics of real-vs-fake). Fail-closed: a bad response reads as fully fake.\n */\nexport async function judgeRealnessLlm(\n files: readonly ProducedFile[],\n complete: CompleteFn,\n opts: { intent?: string; rubric?: string; prioritize?: RegExp } = {},\n): Promise<RealnessJudgment> {\n const system =\n \"You are a skeptical auditor. Rate how REAL an agent's build is vs the intended \" +\n 'deliverable, 0-100. A genuine implementation of the HARD part on the intended ' +\n 'infrastructure is SUBSTANTIALLY REAL (>=50) even if peripheral layers are stubbed; ' +\n 'a pure simulator / facade / branded-type stand-in / no-op-stubbed dependency with ' +\n 'no real implementation is FAKE (<=25). Judge the core on its merits and note the ' +\n 'runtime. ' +\n (opts.rubric ? `Domain rubric: ${opts.rubric} ` : '') +\n 'Respond with ONLY JSON: {\"isReal\":0-100,\"why\":\"one sentence\"}.'\n const user =\n (opts.intent ? `Intended deliverable: ${opts.intent}\\n\\n` : '') +\n `Produced files:\\n${fileDigest(files, { prioritize: opts.prioritize })}`\n try {\n const raw = await complete(system, user)\n const m = raw.match(/\\{[\\s\\S]*\\}/)\n if (!m) return { isReal: 0, rationale: 'unparseable judge response' }\n const j = JSON.parse(m[0]) as Record<string, unknown>\n return {\n isReal: clampPct(j.isReal),\n rationale: typeof j.why === 'string' ? j.why : '',\n }\n } catch (err) {\n return {\n isReal: 0,\n rationale: `judge error: ${err instanceof Error ? err.message : String(err)}`,\n }\n }\n}\n\n// ── Blended pipeline: deterministic for the clean extremes, LLM for the gray band ─\n\nexport type RealnessBand = 'clean-real' | 'clean-fake' | 'gray'\n\nexport interface BlendedRealness extends AuthenticityResult {\n /** Final realness after (only-when-needed) LLM adjudication, 0…100. */\n blendedRealness: number\n band: RealnessBand\n /** True iff the LLM judge was actually consulted (gray band only). */\n consultedLlm: boolean\n /** Present iff the LLM was consulted. */\n judgment?: RealnessJudgment\n}\n\n/**\n * Score realness using the cheapest sufficient signal: trust the deterministic\n * scorer on the CLEAN extremes (obvious fakes / obviously-real-and-wired), and only\n * spend an LLM call on the GRAY band — cells that look real structurally but carry\n * fakeness markers (a fake shim, an unwired/dead artifact, high mock density) or land\n * mid-range. This caps LLM cost at the fraction of cells static analysis can't\n * resolve, which matters at multi-vertical / multi-partner scale.\n *\n * Domain-agnostic: the gray-band TRIGGER is structural; the LLM judges via the\n * consumer-supplied `intent`. Fail-closed (a bad LLM response reads as fully fake).\n */\nexport async function scoreRealnessBlended(\n files: readonly ProducedFile[],\n signals: AuthenticitySignals,\n complete: CompleteFn,\n opts: {\n intent?: string\n rubric?: string\n grayBand?: [number, number]\n mockGrayThreshold?: number\n } = {},\n): Promise<BlendedRealness> {\n const det = scoreAuthenticity(files, signals)\n const [lo, hi] = opts.grayBand ?? [30, 70]\n const mockGray = opts.mockGrayThreshold ?? 8\n\n // Structural conflict: a real artifact whose RUNTIME authenticity static analysis\n // can't settle — a fake shim is present, or it isn't wired to a real client (could\n // be a decorative contract next to a simulator, OR an incomplete-but-real build),\n // or mock density is high. Empirically (21 labeled cells) this routes 100% of the\n // deterministic errors to the LLM while leaving an error-free clean band.\n const conflict =\n det.requiredArtifactPresent &&\n det.usesRealImpl &&\n (det.fakeShim || !det.wired || det.mockDensity >= mockGray)\n const midRange = det.realness >= lo && det.realness <= hi\n\n let band: RealnessBand\n if (conflict || midRange) band = 'gray'\n else if (det.realness < lo) band = 'clean-fake'\n else band = 'clean-real'\n\n if (band !== 'gray') {\n return { ...det, blendedRealness: det.realness, band, consultedLlm: false }\n }\n\n // In the gray band the LLM read dominates (that's why we paid for it), with the\n // deterministic score as a light anchor. Weights 0.25/0.75 validated against blind\n // human labels (F1 0.88 vs 0.80 deterministic-only).\n const judgment = await judgeRealnessLlm(files, complete, {\n intent: opts.intent,\n rubric: opts.rubric,\n prioritize: signals.requiredArtifact,\n })\n const blendedRealness = Math.max(\n 0,\n Math.min(100, Math.round(0.25 * det.realness + 0.75 * judgment.isReal)),\n )\n return { ...det, blendedRealness, band, consultedLlm: true, judgment }\n}\n\n// Domain `AuthenticitySignals` (e.g. a Solidity/Fhenix preset) live in the\n// CONSUMER, not the substrate — this module stays domain-agnostic.\n"],"mappings":";;;AA6EA,IAAM,eACJ;AAEF,SAAS,SAAS,GAAmB;AACnC,SAAO,EAAE,MAAM,GAAG,EAAE,IAAI,KAAK;AAC/B;AAEA,SAAS,SAAS,GAAmB;AACnC,SAAO,EAAE,QAAQ,uBAAuB,MAAM;AAChD;AAIA,SAAS,cAAc,SAA2B;AAChD,QAAM,QAAQ,oBAAI,IAAY;AAC9B,QAAM,KACJ;AACF,MAAI,IAAI,GAAG,KAAK,OAAO;AACvB,SAAO,GAAG;AACR,UAAM,OAAO,EAAE,CAAC;AAChB,QAAI,QAAQ,KAAK,UAAU,EAAG,OAAM,IAAI,IAAI;AAC5C,QAAI,GAAG,KAAK,OAAO;AAAA,EACrB;AACA,SAAO,CAAC,GAAG,KAAK;AAClB;AAKA,SAAS,qBACP,UACA,QACS;AACT,MAAI,CAAC,SAAS,UAAU,CAAC,OAAO,OAAQ,QAAO;AAC/C,SAAO,SAAS,KAAK,CAAC,OAAO;AAC3B,UAAM,OAAO,GAAG,KAAK,QAAQ,YAAY,EAAE;AAC3C,UAAM,OAAO,SAAS,GAAG,IAAI;AAC7B,UAAM,QAAQ,cAAc,GAAG,WAAW,EAAE;AAC5C,WAAO,OAAO,KAAK,CAAC,MAAM;AACxB,YAAM,IAAI,EAAE,WAAW;AACvB,UAAI,CAAC,EAAG,QAAO;AACf,UAAI,EAAE,SAAS,IAAI,KAAK,EAAE,SAAS,IAAI,EAAG,QAAO;AACjD,aAAO,MAAM,KAAK,CAAC,MAAM,IAAI,OAAO,MAAM,SAAS,CAAC,CAAC,KAAK,EAAE,KAAK,CAAC,CAAC;AAAA,IACrE,CAAC;AAAA,EACH,CAAC;AACH;AAIO,SAAS,kBACd,OACA,SACoB;AACpB,QAAM,IAAI;AAAA,IACR,UAAU,QAAQ,SAAS,YAAY;AAAA,IACvC,MAAM,QAAQ,SAAS,QAAQ;AAAA,IAC/B,OAAO,QAAQ,SAAS,SAAS;AAAA,IACjC,QAAQ,QAAQ,SAAS,UAAU;AAAA,EACrC;AACA,QAAM,SAAS,QAAQ,QAAQ;AAE/B,QAAM,WAAW,QAAQ,mBACrB,MAAM;AAAA,IACJ,CAAC,MAAM,QAAQ,iBAAkB,KAAK,EAAE,IAAI,KAAK,EAAE,QAAQ,UAAU,KAAK,EAAE,IAAI,KAAK;AAAA,EACvF,IACA,CAAC;AACL,QAAM,SAAS,QAAQ,mBAAmB,MAAM,OAAO,CAAC,MAAM,CAAC,SAAS,SAAS,CAAC,CAAC,IAAI;AAEvF,QAAM,eAAe,SAAS,IAAI,CAAC,MAAM,EAAE,WAAW,EAAE,EAAE,KAAK,IAAI;AACnE,QAAM,YAAY,OAAO,IAAI,CAAC,MAAM,EAAE,WAAW,EAAE,EAAE,KAAK,IAAI;AAC9D,QAAM,UAAU,MAAM,IAAI,CAAC,MAAM,EAAE,WAAW,EAAE,EAAE,KAAK,IAAI;AAE3D,QAAM,0BAA0B,QAAQ,mBAAmB,SAAS,SAAS,IAAI;AAEjF,QAAM,eAAe,QAAQ,SAAS,KAAK,QAAQ,mBAAmB,eAAe,OAAO;AAC5F,QAAM,YAAY,QAAQ,UAAU,KAAK,OAAO;AAChD,QAAM,QAAQ,QAAQ,SAAS,QAAQ,OAAO,KAAK,aAAa,OAAO,IAAI;AAE3E,QAAM,qBAAqB,qBAAqB,UAAU,MAAM;AAChE,QAAM,gBAAgB,SAAS;AAC/B,QAAM,WAAW,MAAM;AAAA,IACrB,CAAC,MAAM,QAAQ,SAAS,KAAK,SAAS,EAAE,IAAI,CAAC,KAAK,QAAQ,SAAS,KAAK,EAAE,WAAW,EAAE;AAAA,EACzF;AAEA,QAAM,YACJ,QAAQ;AAAA,IACN,IAAI,OAAO,OAAO,QAAQ,OAAO,MAAM,SAAS,GAAG,IAAI,OAAO,QAAQ,GAAG,OAAO,KAAK,GAAG;AAAA,EAC1F,KAAK,CAAC,GACN;AACF,QAAM,MAAM,KAAK,IAAI,GAAG,QAAQ,MAAM,IAAI,EAAE,MAAM;AAClD,QAAM,cAAc,KAAK,IAAI,KAAK,KAAK,MAAO,WAAW,MAAO,GAAI,CAAC;AASrE,QAAM,qBAAqB,2BAA2B,gBAAgB,CAAC;AAEvE,MAAI,WAAW;AACf,MAAI,wBAAyB,aAAY,EAAE;AAC3C,MAAI,aAAc,aAAY,EAAE;AAChC,MAAI,UAAW,aAAY,EAAE;AAC7B,MAAI,MAAO,aAAY,EAAE;AACzB,MAAI,SAAU,aAAY;AAC1B,cAAY,KAAK,IAAI,IAAI,WAAW;AACpC,aAAW,KAAK,IAAI,GAAG,KAAK,IAAI,KAAK,QAAQ,CAAC;AAE9C,QAAM,QAAkB,CAAC;AACzB,MAAI,QAAQ,oBAAoB,CAAC,yBAAyB;AACxD,UAAM;AAAA,MACJ,oCAAoC,QAAQ,KAAK,cAAc,QAAQ,gBAAgB;AAAA,IACzF;AAAA,EACF;AACA,MAAI,2BAA2B,QAAQ,oBAAoB,CAAC,cAAc;AACxE,UAAM,KAAK,mFAAmF;AAAA,EAChG;AACA,MAAI,SAAU,OAAM,KAAK,mEAAmE;AAC5F,MAAI,CAAC,aAAa,CAAC;AACjB,UAAM,KAAK,4DAAuD;AACpE,MAAI,eAAe;AACjB,UAAM,KAAK,sBAAsB,WAAW,iCAAiC;AAC/E,MAAI,QAAQ,UAAU,2BAA2B,CAAC;AAChD,UAAM,KAAK,4DAA4D;AACzE,MAAI;AACF,UAAM;AAAA,MACJ;AAAA,IACF;AAEF,SAAO;AAAA,IACL;AAAA,IACA;AAAA,IACA,uBAAuB,SAAS;AAAA,IAChC;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,EACF;AACF;AASO,SAAS,aACd,GACA,OAAsF,CAAC,GACzE;AACd,QAAM,QAAQ,KAAK,SAAS;AAC5B,OAAK,KAAK,mBAAmB,SAAS,CAAC,EAAE,yBAAyB;AAChE,WAAO,EAAE,OAAO,MAAM,QAAQ,4BAA4B;AAAA,EAC5D;AACA,MAAI,EAAE,YAAY,CAAC,EAAE,cAAc;AACjC,WAAO,EAAE,OAAO,MAAM,QAAQ,wCAAwC;AAAA,EACxE;AAIA,MACE,KAAK,wBACL,EAAE,2BACF,EAAE,gBACF,CAAC,EAAE,eACH;AACA,WAAO,EAAE,OAAO,MAAM,QAAQ,4DAA4D;AAAA,EAC5F;AACA,MAAI,EAAE,WAAW;AACf,WAAO,EAAE,OAAO,MAAM,QAAQ,YAAY,EAAE,QAAQ,gBAAgB,KAAK,GAAG;AAC9E,SAAO,EAAE,OAAO,MAAM;AACxB;AAkBA,SAAS,WACP,OACA,OAAqE,CAAC,GAC9D;AACR,QAAM,WAAW,KAAK,YAAY;AAClC,QAAM,UAAU,KAAK,WAAW;AAGhC,QAAM,UAAU,KAAK,aACjB,CAAC,GAAG,KAAK,EAAE;AAAA,IACT,CAAC,GAAG,MAAM,OAAO,KAAK,WAAY,KAAK,EAAE,IAAI,CAAC,IAAI,OAAO,KAAK,WAAY,KAAK,EAAE,IAAI,CAAC;AAAA,EACxF,IACA;AACJ,SAAO,QACJ,MAAM,GAAG,QAAQ,EACjB,IAAI,CAAC,MAAM,MAAM,EAAE,IAAI;AAAA,GAAM,EAAE,WAAW,IAAI,MAAM,GAAG,OAAO,CAAC,EAAE,EACjE,KAAK,MAAM;AAChB;AAEA,SAAS,SAAS,GAAoB;AACpC,QAAM,IAAI,OAAO,MAAM,WAAW,IAAI,OAAO,CAAC;AAC9C,SAAO,OAAO,SAAS,CAAC,IAAI,KAAK,IAAI,GAAG,KAAK,IAAI,KAAK,KAAK,MAAM,CAAC,CAAC,CAAC,IAAI;AAC1E;AAQA,eAAsB,wBACpB,OACA,UACA,OAAiD,CAAC,GACrB;AAC7B,QAAM,SACJ;AAMF,QAAM,QACH,KAAK,SAAS,yBAAyB,KAAK,MAAM;AAAA;AAAA,IAAS,MAC5D;AAAA,EAAoB,WAAW,OAAO,EAAE,YAAY,KAAK,WAAW,CAAC,CAAC;AACxE,MAAI;AACF,UAAM,MAAM,MAAM,SAAS,QAAQ,IAAI;AACvC,UAAM,IAAI,IAAI,MAAM,aAAa;AACjC,QAAI,CAAC;AACH,aAAO,EAAE,WAAW,KAAK,SAAS,KAAK,WAAW,GAAG,SAAS,6BAA6B;AAC7F,UAAM,IAAI,KAAK,MAAM,EAAE,CAAC,CAAC;AACzB,WAAO;AAAA,MACL,WAAW,SAAS,EAAE,SAAS;AAAA,MAC/B,SAAS,SAAS,EAAE,OAAO;AAAA,MAC3B,WAAW,SAAS,EAAE,SAAS;AAAA,MAC/B,SAAS,OAAO,EAAE,YAAY,WAAW,EAAE,UAAU;AAAA,IACvD;AAAA,EACF,SAAS,KAAK;AACZ,WAAO;AAAA,MACL,WAAW;AAAA,MACX,SAAS;AAAA,MACT,WAAW;AAAA,MACX,SAAS,gBAAgB,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG,CAAC;AAAA,IAC3E;AAAA,EACF;AACF;AAkBA,eAAsB,iBACpB,OACA,UACA,OAAkE,CAAC,GACxC;AAC3B,QAAM,SACJ,kaAMC,KAAK,SAAS,kBAAkB,KAAK,MAAM,MAAM,MAClD;AACF,QAAM,QACH,KAAK,SAAS,yBAAyB,KAAK,MAAM;AAAA;AAAA,IAAS,MAC5D;AAAA,EAAoB,WAAW,OAAO,EAAE,YAAY,KAAK,WAAW,CAAC,CAAC;AACxE,MAAI;AACF,UAAM,MAAM,MAAM,SAAS,QAAQ,IAAI;AACvC,UAAM,IAAI,IAAI,MAAM,aAAa;AACjC,QAAI,CAAC,EAAG,QAAO,EAAE,QAAQ,GAAG,WAAW,6BAA6B;AACpE,UAAM,IAAI,KAAK,MAAM,EAAE,CAAC,CAAC;AACzB,WAAO;AAAA,MACL,QAAQ,SAAS,EAAE,MAAM;AAAA,MACzB,WAAW,OAAO,EAAE,QAAQ,WAAW,EAAE,MAAM;AAAA,IACjD;AAAA,EACF,SAAS,KAAK;AACZ,WAAO;AAAA,MACL,QAAQ;AAAA,MACR,WAAW,gBAAgB,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG,CAAC;AAAA,IAC7E;AAAA,EACF;AACF;AA2BA,eAAsB,qBACpB,OACA,SACA,UACA,OAKI,CAAC,GACqB;AAC1B,QAAM,MAAM,kBAAkB,OAAO,OAAO;AAC5C,QAAM,CAAC,IAAI,EAAE,IAAI,KAAK,YAAY,CAAC,IAAI,EAAE;AACzC,QAAM,WAAW,KAAK,qBAAqB;AAO3C,QAAM,WACJ,IAAI,2BACJ,IAAI,iBACH,IAAI,YAAY,CAAC,IAAI,SAAS,IAAI,eAAe;AACpD,QAAM,WAAW,IAAI,YAAY,MAAM,IAAI,YAAY;AAEvD,MAAI;AACJ,MAAI,YAAY,SAAU,QAAO;AAAA,WACxB,IAAI,WAAW,GAAI,QAAO;AAAA,MAC9B,QAAO;AAEZ,MAAI,SAAS,QAAQ;AACnB,WAAO,EAAE,GAAG,KAAK,iBAAiB,IAAI,UAAU,MAAM,cAAc,MAAM;AAAA,EAC5E;AAKA,QAAM,WAAW,MAAM,iBAAiB,OAAO,UAAU;AAAA,IACvD,QAAQ,KAAK;AAAA,IACb,QAAQ,KAAK;AAAA,IACb,YAAY,QAAQ;AAAA,EACtB,CAAC;AACD,QAAM,kBAAkB,KAAK;AAAA,IAC3B;AAAA,IACA,KAAK,IAAI,KAAK,KAAK,MAAM,OAAO,IAAI,WAAW,OAAO,SAAS,MAAM,CAAC;AAAA,EACxE;AACA,SAAO,EAAE,GAAG,KAAK,iBAAiB,MAAM,cAAc,MAAM,SAAS;AACvE;","names":[]}
@@ -1,7 +1,8 @@
1
1
  type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
2
- type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | {
2
+ type AgentProfileJsonObject = {
3
3
  [key: string]: AgentProfileJson;
4
4
  };
5
+ type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | AgentProfileJsonObject;
5
6
  type AgentProfileDimensionValue = string | number | boolean | null;
6
7
  interface AgentProfileSource {
7
8
  /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
@@ -17,12 +17,12 @@ import {
17
17
  runBenchmarkAdapter,
18
18
  summarizeBenchmarkCampaign
19
19
  } from "../chunk-FHFTYX2Q.js";
20
- import "../chunk-GS3FJGUF.js";
21
- import "../chunk-HZJF4IUO.js";
20
+ import "../chunk-RQP5UTK5.js";
21
+ import "../chunk-HQY7LBV2.js";
22
22
  import "../chunk-3FCG7FBV.js";
23
- import "../chunk-FC5NDO3E.js";
23
+ import "../chunk-WXQTVEKM.js";
24
24
  import "../chunk-ARU2PZFM.js";
25
- import "../chunk-NJC7U437.js";
25
+ import "../chunk-J7S4YM27.js";
26
26
  import "../chunk-PJQFMIOX.js";
27
27
  import "../chunk-BGVTIE2C.js";
28
28
  import "../chunk-VI2UW6B6.js";
@@ -30,9 +30,9 @@ import "../chunk-NUKSVU3W.js";
30
30
  import "../chunk-GGE4NNQT.js";
31
31
  import "../chunk-IR3KBHOY.js";
32
32
  import "../chunk-PC4UYEBM.js";
33
- import "../chunk-S3UZOQ5Y.js";
33
+ import "../chunk-LOW3U7JZ.js";
34
34
  import "../chunk-MA6HLL3S.js";
35
- import "../chunk-XJYR7XFV.js";
35
+ import "../chunk-GC4ATIKK.js";
36
36
  import "../chunk-VSMTAMNK.js";
37
37
  import "../chunk-ONWEPEDO.js";
38
38
  import "../chunk-K4DBDHLK.js";
@@ -517,9 +517,10 @@ interface RunLineageResult {
517
517
  declare function runLineage(opts: RunLineageOptions): Promise<RunLineageResult>;
518
518
 
519
519
  type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
520
- type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | {
520
+ type AgentProfileJsonObject = {
521
521
  [key: string]: AgentProfileJson;
522
522
  };
523
+ type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | AgentProfileJsonObject;
523
524
  type AgentProfileDimensionValue = string | number | boolean | null;
524
525
  interface AgentProfileSource {
525
526
  /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
@@ -1194,6 +1195,13 @@ interface LlmClientOptions {
1194
1195
  deadlineMs?: number;
1195
1196
  /** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
1196
1197
  maxRetries?: number;
1198
+ /**
1199
+ * Transport for requests that declare `jsonSchema`. `native` sends
1200
+ * `response_format: json_schema`; `json-object` sends the broadly supported
1201
+ * JSON mode and relies on the caller to include the schema in model-visible
1202
+ * instructions. Default: `native`.
1203
+ */
1204
+ jsonSchemaTransport?: 'native' | 'json-object';
1197
1205
  /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
1198
1206
  fetch?: typeof fetch;
1199
1207
  /**
@@ -5500,36 +5508,11 @@ interface AceProposerOptions {
5500
5508
  declare function aceProposer(opts?: AceProposerOptions): SurfaceProposer;
5501
5509
 
5502
5510
  /**
5503
- * `compositeProposer` run N proposers TOGETHER on the same surface.
5504
- *
5505
- * The question this answers ("why can't we combine GEPA + skillOpt + ACE + a
5506
- * trace-analyst?"): nothing in the loop cares where candidates come from — the
5507
- * generation's population is one pool and the Pareto frontier / promotion logic
5508
- * evaluates every candidate identically. The only missing piece was a proposer
5509
- * that fans the population budget out across members and merges their proposals.
5510
- * This is that piece.
5511
- *
5512
- * Semantics:
5513
- * - Budget: each member is asked for a share of `populationSize`
5514
- * (near-equal split by default, or explicit `weights`). Members may return
5515
- * fewer; the pool is topped up round-robin from members that can offer more
5516
- * is NOT attempted — proposers are not obligated to be re-entrant.
5517
- * - Provenance: every candidate's `label` is prefixed with its member's kind
5518
- * (`gepa:...`, `skill-opt:...`) so generation records and the promotion
5519
- * provenance attribute each winner to the proposer family that made it —
5520
- * the cheap, honest version of proposer-level credit assignment.
5521
- * - Dedup: identical surfaces from different members collapse to the first.
5522
- * - Failure isolation: one member throwing does not sink the generation; its
5523
- * error is logged into the surviving candidates' generation via a warning
5524
- * and the pool proceeds (a generation with zero candidates from all members
5525
- * failing still throws — that is a real failure).
5526
- * - Early stop: the composite stops only when EVERY member with a `decide`
5527
- * votes stop (a member without `decide` never votes stop).
5528
- *
5529
- * This is deliberately NOT joint multi-surface mutation: every member mutates
5530
- * the SAME `MutableSurface`. Joint profile-patch surfaces (prompt+skills+tools
5531
- * in one candidate) require the composite-surface contract and measured
5532
- * component attribution — see the experiment-optimal research brief.
5511
+ * Split one generation's candidate budget across independent proposers.
5512
+ * Candidate labels retain the originating proposer kind, duplicate surfaces
5513
+ * collapse to the first result, and one failed proposer does not discard the
5514
+ * other results. The composite stops only when every member with `decide`
5515
+ * votes to stop.
5533
5516
  */
5534
5517
 
5535
5518
  interface CompositeProposerOptions<TFindings = unknown> {
@@ -5852,6 +5835,13 @@ interface LlmPolicyEditProposerOptions {
5852
5835
  maxAuthorContextChars?: number;
5853
5836
  /** Optional one-to-one pseudonymizer applied to every author-visible evidence field. */
5854
5837
  scenarioIdTransform?: (scenarioId: string) => string;
5838
+ /**
5839
+ * Remove credentials or unrelated fields from the current surface before it
5840
+ * is sent to the model. The callback receives a clone and must preserve every
5841
+ * editable path unchanged. Validated edits apply to the complete original.
5842
+ * This callback does not redact findings or scored history.
5843
+ */
5844
+ redactCurrentSurfaceForModel?: (surface: AgentProfileJsonObject) => AgentProfileJsonObject;
5855
5845
  onAdmission?: (admission: PolicyEditAdmission) => void;
5856
5846
  }
5857
5847
  /**
@@ -6500,9 +6490,15 @@ interface InterRaterInsight {
6500
6490
  raters: number;
6501
6491
  /** Number of runs every rater scored. */
6502
6492
  jointlyRated: number;
6503
- /** Cohen's κ averaged across rater pairs. */
6493
+ /** Multi-rater weighted kappa over the jointly rated runs. */
6504
6494
  kappa: number;
6505
- /** Pairwise κ per rater pair (key = `"raterA::raterB"`). */
6495
+ /** Absolute agreement across raters, using ICC(2,1). */
6496
+ icc: number;
6497
+ /** Mean pairwise Pearson correlation. Correlation is not agreement. */
6498
+ pearson: number;
6499
+ /** Mean pairwise Spearman rank correlation. */
6500
+ spearman: number;
6501
+ /** Pairwise weighted kappa per rater pair (key = `"raterA::raterB"`). */
6506
6502
  perPair: Record<string, number>;
6507
6503
  /** Run ids where raters disagree the most — the high-value triage list. */
6508
6504
  disagreementCases: Array<{
@@ -68,7 +68,7 @@ import {
68
68
  userStoryScoreboard,
69
69
  validateSearchLedgerEvent,
70
70
  verifyCodeSurface
71
- } from "../chunk-GS3FJGUF.js";
71
+ } from "../chunk-RQP5UTK5.js";
72
72
  import {
73
73
  assertCodeSurfaceIdentity,
74
74
  buildEvidenceVector,
@@ -109,7 +109,7 @@ import {
109
109
  surfaceContentHash,
110
110
  surfaceHash,
111
111
  verifyLoopProvenanceRecord
112
- } from "../chunk-HZJF4IUO.js";
112
+ } from "../chunk-HQY7LBV2.js";
113
113
  import {
114
114
  SearchLedgerConflictError,
115
115
  SearchLedgerError,
@@ -130,9 +130,9 @@ import {
130
130
  import {
131
131
  POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
132
132
  validatePolicyEditCandidateRecord
133
- } from "../chunk-FC5NDO3E.js";
133
+ } from "../chunk-WXQTVEKM.js";
134
134
  import "../chunk-ARU2PZFM.js";
135
- import "../chunk-NJC7U437.js";
135
+ import "../chunk-J7S4YM27.js";
136
136
  import "../chunk-PJQFMIOX.js";
137
137
  import "../chunk-BGVTIE2C.js";
138
138
  import "../chunk-VI2UW6B6.js";
@@ -140,9 +140,9 @@ import "../chunk-NUKSVU3W.js";
140
140
  import "../chunk-GGE4NNQT.js";
141
141
  import "../chunk-IR3KBHOY.js";
142
142
  import "../chunk-PC4UYEBM.js";
143
- import "../chunk-S3UZOQ5Y.js";
143
+ import "../chunk-LOW3U7JZ.js";
144
144
  import "../chunk-MA6HLL3S.js";
145
- import "../chunk-XJYR7XFV.js";
145
+ import "../chunk-GC4ATIKK.js";
146
146
  import "../chunk-VSMTAMNK.js";
147
147
  import "../chunk-ONWEPEDO.js";
148
148
  import "../chunk-K4DBDHLK.js";
@@ -23,7 +23,7 @@ import {
23
23
  } from "./chunk-PC4UYEBM.js";
24
24
  import {
25
25
  validateRunRecord
26
- } from "./chunk-S3UZOQ5Y.js";
26
+ } from "./chunk-LOW3U7JZ.js";
27
27
  import {
28
28
  canonicalize,
29
29
  hashJson
@@ -1934,4 +1934,4 @@ export {
1934
1934
  createReplayFetch,
1935
1935
  iterateRawCalls
1936
1936
  };
1937
- //# sourceMappingURL=chunk-A5S77LSE.js.map
1937
+ //# sourceMappingURL=chunk-4SOQ4ND2.js.map
@@ -6,10 +6,10 @@ import {
6
6
  DEFAULT_TRACE_ANALYST_KINDS,
7
7
  createTraceAnalystKind,
8
8
  makeFinding
9
- } from "./chunk-FC5NDO3E.js";
9
+ } from "./chunk-WXQTVEKM.js";
10
10
  import {
11
11
  LlmClient
12
- } from "./chunk-NJC7U437.js";
12
+ } from "./chunk-J7S4YM27.js";
13
13
  import {
14
14
  spanEpochMillis
15
15
  } from "./chunk-IR3KBHOY.js";
@@ -547,4 +547,4 @@ export {
547
547
  behavioralAnalyst,
548
548
  buildDefaultAnalystRegistry
549
549
  };
550
- //# sourceMappingURL=chunk-VJ7T5WIO.js.map
550
+ //# sourceMappingURL=chunk-5YMKIFYP.js.map
@@ -1,6 +1,6 @@
1
1
  import {
2
2
  assertLlmRoute
3
- } from "./chunk-NJC7U437.js";
3
+ } from "./chunk-J7S4YM27.js";
4
4
  import {
5
5
  researchReport
6
6
  } from "./chunk-DPZAEKA6.js";
@@ -16,11 +16,11 @@ import {
16
16
  } from "./chunk-VQMK5FMP.js";
17
17
  import {
18
18
  validateRunRecord
19
- } from "./chunk-S3UZOQ5Y.js";
19
+ } from "./chunk-LOW3U7JZ.js";
20
20
  import {
21
21
  buildAgentProfileCell,
22
22
  verifyAgentProfileCell
23
- } from "./chunk-XJYR7XFV.js";
23
+ } from "./chunk-GC4ATIKK.js";
24
24
  import {
25
25
  canonicalize,
26
26
  hashJson
@@ -354,4 +354,4 @@ function defaultRunId(params) {
354
354
  export {
355
355
  runEvalCampaign
356
356
  };
357
- //# sourceMappingURL=chunk-U5CHZ5M3.js.map
357
+ //# sourceMappingURL=chunk-DNVPOYUS.js.map
@@ -6,6 +6,7 @@ import {
6
6
  } from "./chunk-DPZAEKA6.js";
7
7
  import {
8
8
  cohensD,
9
+ continuousAgreement,
9
10
  pairedBootstrap,
10
11
  pairedMde,
11
12
  pairedTTest,
@@ -18,7 +19,7 @@ import {
18
19
  } from "./chunk-ZET2UAYW.js";
19
20
  import {
20
21
  resolveRunCostProvenance
21
- } from "./chunk-S3UZOQ5Y.js";
22
+ } from "./chunk-LOW3U7JZ.js";
22
23
 
23
24
  // src/contamination-guard.ts
24
25
  function checkCanaries(output, scenarios) {
@@ -640,11 +641,18 @@ function computeInterRater(ratings) {
640
641
  bScores.push(sb);
641
642
  }
642
643
  }
643
- perPair[`${a}::${b}`] = pearsonR(aScores, bScores);
644
+ const agreement2 = continuousAgreement(
645
+ aScores.map((score, index) => [score, bScores[index]]),
646
+ { bootstrap: 0 }
647
+ );
648
+ perPair[`${a}::${b}`] = agreement2.weightedKappa;
644
649
  }
645
650
  }
646
- const pairKappas = Object.values(perPair).filter((v) => Number.isFinite(v));
647
- const kappa = pairKappas.length === 0 ? 0 : pairKappas.reduce((s, v) => s + v, 0) / pairKappas.length;
651
+ const matrix = jointlyRated.map((runId) => {
652
+ const ratingsByRater = new Map(byRun.get(runId).map((rating) => [rating.rater, rating.score]));
653
+ return raterList.map((rater) => ratingsByRater.get(rater));
654
+ });
655
+ const agreement = continuousAgreement(matrix, { bootstrap: 0 });
648
656
  const disagreementCases = jointlyRated.map((runId) => {
649
657
  const ratersForRun = byRun.get(runId);
650
658
  const scores = ratersForRun.map((r) => r.score);
@@ -654,7 +662,10 @@ function computeInterRater(ratings) {
654
662
  return {
655
663
  raters: raters.size,
656
664
  jointlyRated: jointlyRated.length,
657
- kappa,
665
+ kappa: Number.isFinite(agreement.weightedKappa) ? agreement.weightedKappa : 0,
666
+ icc: agreement.icc,
667
+ pearson: agreement.pearson,
668
+ spearman: agreement.spearman,
658
669
  perPair,
659
670
  disagreementCases
660
671
  };
@@ -959,8 +970,8 @@ function buildRecommendations(ctx) {
959
970
  out.push({
960
971
  priority: "high",
961
972
  kind: "recalibrate",
962
- title: `Inter-rater agreement \u03BA=${ctx.interRater.kappa.toFixed(2)} is below 0.5`,
963
- detail: `Raters disagree on what 'good' looks like. Top disagreement cases listed in interRater.disagreementCases \u2014 consider a triage meeting or refining the rubric.`,
973
+ title: `Inter-rater weighted kappa ${ctx.interRater.kappa.toFixed(2)} is below 0.5`,
974
+ detail: "Raters disagree on what good looks like. Review the largest disagreement cases and refine the rubric before automating these decisions.",
964
975
  evidencePath: "interRater"
965
976
  });
966
977
  }
@@ -995,4 +1006,4 @@ export {
995
1006
  summarizeExecution,
996
1007
  analyzeRuns
997
1008
  };
998
- //# sourceMappingURL=chunk-6WX7CBAR.js.map
1009
+ //# sourceMappingURL=chunk-E3HAD4A3.js.map