@tangle-network/agent-app 0.47.24 → 0.48.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/.claude/skills/eval-architect/SKILL.md +34 -31
  2. package/.claude/skills/improve-conductor/SKILL.md +46 -41
  3. package/.claude/skills/measurement-validation/SKILL.md +40 -34
  4. package/README.md +7 -0
  5. package/dist/DesignCanvas-MWEYD6PR.js +11 -0
  6. package/dist/DesignCanvasEditor-UZ2S6BDD.js +14 -0
  7. package/dist/app-auth/index.js +2 -2
  8. package/dist/chat-routes/index.js +2 -2
  9. package/dist/{chunk-G4CPEJNN.js → chunk-AKMQFCDL.js} +2 -2
  10. package/dist/{chunk-TXM4SS46.js → chunk-CSMJM7YT.js} +2 -2
  11. package/dist/{chunk-3YUI7WNW.js → chunk-IAY2DMKW.js} +30 -3
  12. package/dist/chunk-IAY2DMKW.js.map +1 -0
  13. package/dist/{chunk-54KJ54SY.js → chunk-JAT6SOPK.js} +3 -3
  14. package/dist/{chunk-HUYTN7RI.js → chunk-JF423ASS.js} +5 -5
  15. package/dist/{chunk-WOZSOPUZ.js → chunk-KX2GA3L6.js} +3 -3
  16. package/dist/{chunk-42NBWR4U.js → chunk-SYWEMKZR.js} +2 -2
  17. package/dist/{chunk-QEPLXYOE.js → chunk-VHMZ3KUH.js} +2 -2
  18. package/dist/{chunk-A2XMTCER.js → chunk-WYHFHTSQ.js} +2 -2
  19. package/dist/design-canvas/index.js +2 -2
  20. package/dist/design-canvas-react/engine.js +4 -4
  21. package/dist/design-canvas-react/index.js +7 -7
  22. package/dist/design-canvas-react/lazy.js +1 -1
  23. package/dist/eval-campaign/index.d.ts +4 -4
  24. package/dist/eval-campaign/index.js +3 -2
  25. package/dist/eval-campaign/index.js.map +1 -1
  26. package/dist/eval-campaign/trust-gate.d.ts +10 -33
  27. package/dist/platform/index.js +2 -2
  28. package/dist/public-consultation/index.js +2 -2
  29. package/dist/sequences/index.js +1 -1
  30. package/dist/web/core.d.ts +84 -0
  31. package/dist/web/index.d.ts +3 -84
  32. package/dist/web/index.js +3 -1
  33. package/dist/web/message-groups.d.ts +21 -0
  34. package/package.json +13 -13
  35. package/dist/DesignCanvas-UPRRA4YG.js +0 -11
  36. package/dist/DesignCanvasEditor-GVJO5SMH.js +0 -14
  37. package/dist/chunk-3YUI7WNW.js.map +0 -1
  38. /package/dist/{DesignCanvas-UPRRA4YG.js.map → DesignCanvas-MWEYD6PR.js.map} +0 -0
  39. /package/dist/{DesignCanvasEditor-GVJO5SMH.js.map → DesignCanvasEditor-UZ2S6BDD.js.map} +0 -0
  40. /package/dist/{chunk-G4CPEJNN.js.map → chunk-AKMQFCDL.js.map} +0 -0
  41. /package/dist/{chunk-TXM4SS46.js.map → chunk-CSMJM7YT.js.map} +0 -0
  42. /package/dist/{chunk-54KJ54SY.js.map → chunk-JAT6SOPK.js.map} +0 -0
  43. /package/dist/{chunk-HUYTN7RI.js.map → chunk-JF423ASS.js.map} +0 -0
  44. /package/dist/{chunk-WOZSOPUZ.js.map → chunk-KX2GA3L6.js.map} +0 -0
  45. /package/dist/{chunk-42NBWR4U.js.map → chunk-SYWEMKZR.js.map} +0 -0
  46. /package/dist/{chunk-QEPLXYOE.js.map → chunk-VHMZ3KUH.js.map} +0 -0
  47. /package/dist/{chunk-A2XMTCER.js.map → chunk-WYHFHTSQ.js.map} +0 -0
@@ -3,7 +3,7 @@ import {
3
3
  } from "./chunk-ZSSCVT6H.js";
4
4
  import {
5
5
  parseJsonObjectBody
6
- } from "./chunk-3YUI7WNW.js";
6
+ } from "./chunk-IAY2DMKW.js";
7
7
  import {
8
8
  mentionInputToPart,
9
9
  toChatMessageParts
@@ -1046,4 +1046,4 @@ export {
1046
1046
  createChatTurnRoutes,
1047
1047
  streamChatRouteAsSandboxEvents
1048
1048
  };
1049
- //# sourceMappingURL=chunk-42NBWR4U.js.map
1049
+ //# sourceMappingURL=chunk-SYWEMKZR.js.map
@@ -4,7 +4,7 @@ import {
4
4
  boundsIntersect,
5
5
  elementAabb,
6
6
  scaleForPreset
7
- } from "./chunk-A2XMTCER.js";
7
+ } from "./chunk-WYHFHTSQ.js";
8
8
 
9
9
  // src/design-canvas-react/engine/snap.ts
10
10
  var KIND_PRIORITY = {
@@ -447,4 +447,4 @@ export {
447
447
  buildInsertImageOp,
448
448
  DEFAULT_INSERT_TEMPLATES
449
449
  };
450
- //# sourceMappingURL=chunk-QEPLXYOE.js.map
450
+ //# sourceMappingURL=chunk-VHMZ3KUH.js.map
@@ -1,6 +1,6 @@
1
1
  import {
2
2
  assertMediaUrl
3
- } from "./chunk-3YUI7WNW.js";
3
+ } from "./chunk-IAY2DMKW.js";
4
4
 
5
5
  // src/design-canvas/model.ts
6
6
  var SCENE_SCHEMA_VERSION = 1;
@@ -1001,4 +1001,4 @@ export {
1001
1001
  scalePageForChannelPreset,
1002
1002
  bleedAwareExportRect
1003
1003
  };
1004
- //# sourceMappingURL=chunk-A2XMTCER.js.map
1004
+ //# sourceMappingURL=chunk-WYHFHTSQ.js.map
@@ -31,12 +31,12 @@ import {
31
31
  validateSceneOperation,
32
32
  validateSceneOperations,
33
33
  validateSlotValue
34
- } from "../chunk-A2XMTCER.js";
34
+ } from "../chunk-WYHFHTSQ.js";
35
35
  import {
36
36
  buildScopedMcpServerEntry,
37
37
  createMcpToolHandler
38
38
  } from "../chunk-6A7MYOUI.js";
39
- import "../chunk-3YUI7WNW.js";
39
+ import "../chunk-IAY2DMKW.js";
40
40
  import "../chunk-TXD5HXLE.js";
41
41
 
42
42
  // src/design-canvas/operations.ts
@@ -20,7 +20,7 @@ import {
20
20
  nudgeDelta,
21
21
  resolveExportParams,
22
22
  resolveNodeCachePixelRatio
23
- } from "../chunk-QEPLXYOE.js";
23
+ } from "../chunk-VHMZ3KUH.js";
24
24
  import {
25
25
  SCENE_COMMAND_HISTORY_LIMIT,
26
26
  addElementCommand,
@@ -39,12 +39,12 @@ import {
39
39
  setPageGuidesCommand,
40
40
  setPagePropsCommand,
41
41
  ungroupElementCommand
42
- } from "../chunk-G4CPEJNN.js";
42
+ } from "../chunk-AKMQFCDL.js";
43
43
  import {
44
44
  bleedAwareExportBounds,
45
45
  scaleForPreset
46
- } from "../chunk-A2XMTCER.js";
47
- import "../chunk-3YUI7WNW.js";
46
+ } from "../chunk-WYHFHTSQ.js";
47
+ import "../chunk-IAY2DMKW.js";
48
48
  import "../chunk-TXD5HXLE.js";
49
49
  export {
50
50
  DEFAULT_INSERT_TEMPLATES,
@@ -12,7 +12,7 @@ import {
12
12
  downloadDataUrl,
13
13
  exportDocumentJson,
14
14
  exportPageDataUrl
15
- } from "../chunk-HUYTN7RI.js";
15
+ } from "../chunk-JF423ASS.js";
16
16
  import {
17
17
  BTN,
18
18
  BTN_ACTIVE,
@@ -41,7 +41,7 @@ import {
41
41
  buildRulerTicks,
42
42
  formatRulerLabel,
43
43
  selectTickStep
44
- } from "../chunk-54KJ54SY.js";
44
+ } from "../chunk-JAT6SOPK.js";
45
45
  import "../chunk-Q2QKEV6J.js";
46
46
  import "../chunk-ZN5J47UX.js";
47
47
  import {
@@ -65,7 +65,7 @@ import {
65
65
  nudgeDelta,
66
66
  resolveExportParams,
67
67
  resolveNodeCachePixelRatio
68
- } from "../chunk-QEPLXYOE.js";
68
+ } from "../chunk-VHMZ3KUH.js";
69
69
  import {
70
70
  SCENE_COMMAND_HISTORY_LIMIT,
71
71
  addElementCommand,
@@ -84,17 +84,17 @@ import {
84
84
  setPageGuidesCommand,
85
85
  setPagePropsCommand,
86
86
  ungroupElementCommand
87
- } from "../chunk-G4CPEJNN.js";
87
+ } from "../chunk-AKMQFCDL.js";
88
88
  import {
89
89
  assertSceneMediaSrc,
90
90
  bleedAwareExportBounds,
91
91
  scaleForPreset
92
- } from "../chunk-A2XMTCER.js";
92
+ } from "../chunk-WYHFHTSQ.js";
93
93
  import {
94
94
  DesignCanvasChromeLazy,
95
95
  DesignCanvasLazy
96
- } from "../chunk-WOZSOPUZ.js";
97
- import "../chunk-3YUI7WNW.js";
96
+ } from "../chunk-KX2GA3L6.js";
97
+ import "../chunk-IAY2DMKW.js";
98
98
  import "../chunk-TXD5HXLE.js";
99
99
 
100
100
  // src/design-canvas-react/components/CanvasInsertPanel.tsx
@@ -1,7 +1,7 @@
1
1
  import {
2
2
  DesignCanvasChromeLazy,
3
3
  DesignCanvasLazy
4
- } from "../chunk-WOZSOPUZ.js";
4
+ } from "../chunk-KX2GA3L6.js";
5
5
  export {
6
6
  DesignCanvasChromeLazy,
7
7
  DesignCanvasLazy
@@ -27,6 +27,8 @@ import type { JudgeConfig, Scenario } from '@tangle-network/agent-eval/campaign'
27
27
  export interface EnsembleJudgeConfig<TArtifact, TScenario extends Scenario, D extends string> {
28
28
  /** Judge name — appears in traces and scorecards. */
29
29
  name: string;
30
+ /** Scoring revision for campaign caches. Change it when rubric, model, or callback settings change. */
31
+ judgeVersion?: JudgeConfig<TArtifact, TScenario>['judgeVersion'];
30
32
  /** Stable-ordered rubric dimensions. Drives the `JudgeDimension` list AND the
31
33
  * reducer keys, so a judge that omits a dimension scores it 0 (never silently
32
34
  * dropped). */
@@ -38,11 +40,9 @@ export interface EnsembleJudgeConfig<TArtifact, TScenario extends Scenario, D ex
38
40
  * `{ model, perDimension: null }` to record a judge failure WITHOUT killing
39
41
  * the ensemble; throw only on an unrecoverable error (the whole rep is then
40
42
  * treated as a failed judge).
43
+ * Use the supplied costLedger, costPhase, and costTags to record paid judge calls.
41
44
  */
42
- scoreOne: (input: {
43
- artifact: TArtifact;
44
- scenario: TScenario;
45
- signal: AbortSignal;
45
+ scoreOne: (input: Parameters<JudgeConfig<TArtifact, TScenario>['score']>[0] & {
46
46
  rep: number;
47
47
  }) => Promise<JudgeVerdict<D>>;
48
48
  /** Independent judge calls per artifact, reduced by `aggregateJudgeVerdicts`.
@@ -115,10 +115,11 @@ function buildEnsembleJudge(cfg) {
115
115
  }
116
116
  return {
117
117
  name: cfg.name,
118
+ ...cfg.judgeVersion === void 0 ? {} : { judgeVersion: cfg.judgeVersion },
118
119
  dimensions: cfg.rubric.map((key) => ({ key, description: cfg.describe?.(key) ?? key })),
119
- async score({ artifact, scenario, signal }) {
120
+ async score(input) {
120
121
  const settled = await Promise.allSettled(
121
- Array.from({ length: reps }, (_, rep) => cfg.scoreOne({ artifact, scenario, signal, rep }))
122
+ Array.from({ length: reps }, (_, rep) => cfg.scoreOne({ ...input, rep }))
122
123
  );
123
124
  const verdicts = settled.map(
124
125
  (r, rep) => r.status === "fulfilled" ? r.value : { model: `${cfg.name}-rep${rep}`, perDimension: null, rationale: String(r.reason) }
@@ -1 +1 @@
1
- {"version":3,"sources":["../../src/eval-campaign/index.ts","../../src/eval-campaign/trust-gate.ts"],"sourcesContent":["/**\n * Eval-campaign — the app-shell's curated surface for a product's\n * self-improvement loop, NOT a reimplementation.\n *\n * The loop ENGINE lives in `@tangle-network/agent-eval` (a peer dependency):\n * `selfImprove` already owns execution, scoring, data separation, release\n * decisions, durable provenance, and hosted ingest. Candidate search is always\n * explicit: pass an official optimization `method`, or pass a caller-owned\n * `SurfaceProposer`.\n *\n * This module adds the one piece `selfImprove` does not own and which every\n * multi-model product re-hand-rolls — the ensemble judge:\n *\n * {@link buildEnsembleJudge} — turn a per-rubric `scoreOne` into a\n * `JudgeConfig` that fans out N uncorrelated judge calls and reduces them via\n * the substrate's `aggregateJudgeVerdicts` (survivor-mean, inter-rater spread,\n * fail-loud on all-failed). A product writes its rubric + one judge call; the\n * fan-out, partial-failure handling, and composite are the scaffold's.\n *\n * Everything else is a curated re-export so a product has ONE eval import:\n * `selfImprove` + release policies + optimization methods + their types. See\n * `.claude/skills/eval-campaign/SKILL.md` for the wiring contract.\n */\n\nimport {\n aggregateJudgeVerdicts,\n type JudgeVerdict,\n} from '@tangle-network/agent-eval'\nimport type {\n JudgeConfig,\n JudgeScore,\n Scenario,\n} from '@tangle-network/agent-eval/campaign'\n\n/** Config for {@link buildEnsembleJudge}. `D` = the rubric's dimension union. */\nexport interface EnsembleJudgeConfig<TArtifact, TScenario extends Scenario, D extends string> {\n /** Judge name — appears in traces and scorecards. */\n name: string\n /** Stable-ordered rubric dimensions. Drives the `JudgeDimension` list AND the\n * reducer keys, so a judge that omits a dimension scores it 0 (never silently\n * dropped). */\n rubric: readonly D[]\n /**\n * Score ONE artifact on the rubric → a raw per-dimension verdict. Called\n * `judgeReps` times per artifact; vary the model by `rep` for an uncorrelated\n * ensemble (judges that share a base model share its bias). Return\n * `{ model, perDimension: null }` to record a judge failure WITHOUT killing\n * the ensemble; throw only on an unrecoverable error (the whole rep is then\n * treated as a failed judge).\n */\n scoreOne: (input: {\n artifact: TArtifact\n scenario: TScenario\n signal: AbortSignal\n rep: number\n }) => Promise<JudgeVerdict<D>>\n /** Independent judge calls per artifact, reduced by `aggregateJudgeVerdicts`.\n * Default 1. Raise (with model variety in `scoreOne`) for inter-rater bands. */\n judgeReps?: number\n /** Per-dimension composite weights. Default: uniform over `rubric`. A partial\n * map selects-and-weights exactly the named dimensions. */\n weights?: Partial<Record<D, number>>\n /** Optional human-readable dimension descriptions. Default: the key itself. */\n describe?: (dim: D) => string\n}\n\n/**\n * Build a `JudgeConfig` whose `score()` fans out `judgeReps` independent\n * `scoreOne` calls and reduces them with the substrate's\n * `aggregateJudgeVerdicts`. A single judge call failing does NOT fail the cell\n * (it is recorded and dropped); only ALL judges failing throws — which the\n * campaign records as a failed cell, never a silent zero.\n *\n * Pass the result straight to `selfImprove({ judge })` (or `runCampaign`).\n */\nexport function buildEnsembleJudge<TArtifact, TScenario extends Scenario, D extends string>(\n cfg: EnsembleJudgeConfig<TArtifact, TScenario, D>,\n): JudgeConfig<TArtifact, TScenario> {\n const reps = cfg.judgeReps ?? 1\n if (reps < 1) {\n throw new Error(`buildEnsembleJudge: judgeReps must be >= 1 (got ${reps})`)\n }\n if (cfg.rubric.length === 0) {\n throw new Error('buildEnsembleJudge: rubric is empty')\n }\n return {\n name: cfg.name,\n dimensions: cfg.rubric.map((key) => ({ key, description: cfg.describe?.(key) ?? key })),\n async score({ artifact, scenario, signal }): Promise<JudgeScore> {\n const settled = await Promise.allSettled(\n Array.from({ length: reps }, (_, rep) => cfg.scoreOne({ artifact, scenario, signal, rep })),\n )\n const verdicts: JudgeVerdict<D>[] = settled.map((r, rep) =>\n r.status === 'fulfilled'\n ? r.value\n : { model: `${cfg.name}-rep${rep}`, perDimension: null, rationale: String(r.reason) },\n )\n // Throws iff EVERY rep failed → the campaign records a failed cell.\n const agg = aggregateJudgeVerdicts(verdicts, cfg.rubric, cfg.weights)\n return { composite: agg.composite, dimensions: agg.perDimension, notes: agg.rationale }\n },\n }\n}\n\n// ── Trust gate — the after-gate (\"is this result allowed to be believed\") ────\n// One level up from `aggregateJudgeVerdicts`: it audits the raters ACROSS items\n// and reports whether the composites are believable before a lift is reported.\nexport {\n trustVerdicts,\n type TrustItem,\n type TrustThresholds,\n type TrustVerdict,\n} from './trust-gate'\n\n// ── Curated re-exports — the one eval import for a product loop ──────────────\n// The loop engine + gates + drivers + the ensemble reducer, so a product wires\n// its self-improvement loop from a single module instead of reaching across\n// three agent-eval subpaths. All DOWNWARD imports (agent-app consumes the\n// substrate); the layering rule is preserved.\n\nexport { aggregateJudgeVerdicts } from '@tangle-network/agent-eval'\nexport type {\n EnsembleAggregate,\n JudgeVerdict,\n RunRecord,\n} from '@tangle-network/agent-eval'\nexport {\n compareOptimizationMethods,\n defaultProductionGate,\n externalTextOptimizationMethod,\n gepaOptimizationMethod,\n paretoSignificanceGate,\n runCampaign,\n skillOptOptimizationMethod,\n} from '@tangle-network/agent-eval/campaign'\nexport type {\n CampaignResult,\n CompareOptimizationMethodsOptions,\n DispatchContext,\n ExternalTextOptimizationMethodConfig,\n Gate,\n GepaOptimizationMethodConfig,\n JudgeConfig,\n JudgeDimension,\n JudgeScore,\n LabeledScenarioStore,\n MutableSurface,\n OptimizationMethod,\n OptimizationMethodResult,\n Scenario,\n SkillOptOptimizationMethodConfig,\n SurfaceProposer,\n} from '@tangle-network/agent-eval/campaign'\nexport { selfImprove } from '@tangle-network/agent-eval/contract'\nexport type {\n SelfImproveBudget,\n SelfImproveOptions,\n SelfImproveResult,\n} from '@tangle-network/agent-eval/contract'\n","/**\n * Trust gate — decides whether an ensemble's scores are allowed to be BELIEVED,\n * one level up from {@link aggregateJudgeVerdicts} (which only reduces ONE\n * artifact's raters to a composite). A composite is a number; this is the check\n * that the number means anything. It is the code \"Enforced by\" for the\n * measurement-validation skill's after-gate (\"is this result allowed to be\n * believed\").\n *\n * Three checks, each fail-loud and named in `trustReasons`:\n * (1) inter-rater reliability over the corpus ≥ `irrFloor` — raters that\n * disagree no better than chance carry no signal to optimize against.\n * (2) per-item rater spread ≤ `spreadCeiling` — for EACH item, raters must\n * converge on THAT item.\n * (3) surviving raters per item ≥ `minSurvivors` — a mean over one or two\n * raters is an anecdote, not an ensemble.\n *\n * CRITICAL metric semantics — per-item spread is rater disagreement about the\n * SAME item: `max(score) − min(score)` across the raters that scored THAT item\n * (max over its dimensions), never pooled across different items or across the\n * baseline/candidate sides. Pooling reads a genuine quality gap BETWEEN items as\n * \"the raters split\" and so trips the gate exactly when the finding is largest —\n * the failure mode the after-gate exists to prevent. The corpus IRR (check 1)\n * leans on the substrate's `interRaterReliability`, whose expected-disagreement\n * denominator already pools across items, so genuine item-to-item variation\n * RAISES reliability rather than lowering it.\n */\n\nimport {\n interRaterReliability,\n type JudgeScore,\n type JudgeVerdict,\n} from '@tangle-network/agent-eval'\n\n/** One item's raters: the per-judge verdicts {@link aggregateJudgeVerdicts}\n * reduces, tagged with the item they scored so spread stays within-item. */\nexport interface TrustItem<D extends string = string> {\n /** Stable item identifier — surfaces in `perItemSpread` and `trustReasons`. */\n itemId: string\n /** The raters' verdicts for THIS item (one per judge call). A failed judge\n * (`perDimension: null`) is dropped before spread/IRR, never folded as 0. */\n verdicts: readonly JudgeVerdict<D>[]\n}\n\n/** Thresholds for {@link trustVerdicts}. All overridable; defaults are the\n * conservative after-gate bar. */\nexport interface TrustThresholds {\n /** Minimum corpus inter-rater reliability (Krippendorff-style α). Below this\n * the raters agree no better than chance. Default 0.2. */\n irrFloor?: number\n /** Maximum per-item rater spread (`max − min` over a single item's surviving\n * raters, across its dimensions). Above this the raters split ON THAT ITEM.\n * Default 0.5. */\n spreadCeiling?: number\n /** Minimum surviving (non-failed) raters required per item. Default 3. */\n minSurvivors?: number\n}\n\n/** Result of the trust gate. `trustworthy` iff every check passed; `trustReasons`\n * is empty iff `trustworthy`. */\nexport interface TrustVerdict {\n /** True iff IRR ≥ floor AND every item's spread ≤ ceiling AND every item has\n * ≥ `minSurvivors` surviving raters. */\n trustworthy: boolean\n /** One entry per FAILED check, each naming its number + the offending value.\n * Empty iff `trustworthy`. */\n trustReasons: string[]\n /** Corpus inter-rater reliability actually measured (the check-1 value). */\n interRaterReliability: number\n /** Per-item spread (`max − min` over surviving raters, max over dimensions),\n * keyed by `itemId`. The check-2 input, surfaced for drill-down. */\n perItemSpread: Record<string, number>\n}\n\nconst DEFAULT_IRR_FLOOR = 0.2\nconst DEFAULT_SPREAD_CEILING = 0.5\nconst DEFAULT_MIN_SURVIVORS = 3\n\n/** Surviving (non-failed) verdicts for an item — those with a real\n * `perDimension` map. A failed judge carries no scores and is excluded from\n * every statistic (it is NOT a zero rater). */\nfunction survivors<D extends string>(item: TrustItem<D>): JudgeVerdict<D>[] {\n return item.verdicts.filter((v) => v.perDimension !== null)\n}\n\n/**\n * Within-item rater spread: for each dimension, `max − min` across the item's\n * surviving raters; the item's spread is the max over its dimensions (the worst\n * dimension the raters split on). Pooled ONLY within this one item — never\n * across items — so a quality gap between items cannot inflate it.\n */\nfunction itemSpread<D extends string>(survivorVerdicts: JudgeVerdict<D>[]): number {\n if (survivorVerdicts.length < 2) return 0\n const dims = new Set<string>()\n for (const v of survivorVerdicts) {\n for (const d of Object.keys(v.perDimension as Record<string, number>)) dims.add(d)\n }\n let worst = 0\n for (const d of dims) {\n let min = Infinity\n let max = -Infinity\n for (const v of survivorVerdicts) {\n const score = (v.perDimension as Record<string, number>)[d]\n if (score === undefined) continue\n if (score < min) min = score\n if (score > max) max = score\n }\n if (max > -Infinity && max - min > worst) worst = max - min\n }\n return worst\n}\n\n/**\n * Decide whether an ensemble's per-item verdicts are trustworthy enough to\n * believe a lift computed from them. Pure: no LLM, no I/O, no clock, no random —\n * the same `items` + `thresholds` always yield the same verdict.\n *\n * Sibling to {@link aggregateJudgeVerdicts}: that reduces ONE item's raters to a\n * composite; this audits the raters ACROSS items and reports whether the\n * composites are believable. Run it on the corpus of held-out items before\n * reporting any lift over their scores.\n *\n * @throws if `items` is empty — an empty corpus has no measurable trust, and a\n * silent `trustworthy: true` over zero evidence is the exact lie the gate\n * exists to refuse.\n */\nexport function trustVerdicts<D extends string>(\n items: readonly TrustItem<D>[],\n thresholds: TrustThresholds = {},\n): TrustVerdict {\n if (items.length === 0) {\n throw new Error('trustVerdicts: items is empty — no evidence to trust')\n }\n const irrFloor = thresholds.irrFloor ?? DEFAULT_IRR_FLOOR\n const spreadCeiling = thresholds.spreadCeiling ?? DEFAULT_SPREAD_CEILING\n const minSurvivors = thresholds.minSurvivors ?? DEFAULT_MIN_SURVIVORS\n\n // Rater-major JudgeScore series for the substrate's IRR. Each item's surviving\n // raters are assigned a stable column index so the same rater across items\n // lines up; per (item, dimension) one JudgeScore per rater, in item-then-\n // dimension order — the layout interRaterReliability chunks back into items.\n const maxRaters = items.reduce((m, it) => Math.max(m, survivors(it).length), 0)\n const raterSeries: JudgeScore[][] = Array.from({ length: maxRaters }, () => [])\n const perItemSpread: Record<string, number> = {}\n const splitItems: Array<{ itemId: string; spread: number }> = []\n const starvedItems: Array<{ itemId: string; n: number }> = []\n\n for (const item of items) {\n const surv = survivors(item)\n if (surv.length < minSurvivors) starvedItems.push({ itemId: item.itemId, n: surv.length })\n\n const spread = itemSpread(surv)\n perItemSpread[item.itemId] = spread\n if (spread > spreadCeiling) splitItems.push({ itemId: item.itemId, spread })\n\n if (surv.length >= 2) {\n const dims = Array.from(\n new Set(surv.flatMap((v) => Object.keys(v.perDimension as Record<string, number>))),\n ).sort()\n surv.forEach((v, raterIdx) => {\n // raterIdx < surv.length ≤ maxRaters = raterSeries.length, so the column\n // always exists; the ??= keeps the access provably defined for the type.\n const column = (raterSeries[raterIdx] ??= [])\n const pd = v.perDimension as Record<string, number>\n for (const d of dims) {\n const score = pd[d]\n if (score === undefined) continue\n column.push({\n judgeName: v.model,\n dimension: `${item.itemId}::${d}`,\n score,\n reasoning: v.rationale ?? '',\n })\n }\n })\n }\n }\n\n const irr = interRaterReliability(raterSeries)\n\n const trustReasons: string[] = []\n if (irr < irrFloor) {\n trustReasons.push(`(1) IRR ${round(irr)} < ${irrFloor}`)\n }\n for (const { itemId, spread } of splitItems) {\n trustReasons.push(`(2) item ${itemId} spread ${round(spread)} > ${spreadCeiling} — raters split`)\n }\n for (const { itemId, n } of starvedItems) {\n trustReasons.push(`(3) item ${itemId}: ${n} surviving raters < ${minSurvivors}`)\n }\n\n return {\n trustworthy: trustReasons.length === 0,\n trustReasons,\n interRaterReliability: irr,\n perItemSpread,\n }\n}\n\n/** Round to 2 decimals for stable, readable reason strings. */\nfunction round(n: number): number {\n return Math.round(n * 100) / 100\n}\n"],"mappings":";AAwBA;AAAA,EACE;AAAA,OAEK;;;ACAP;AAAA,EACE;AAAA,OAGK;AA0CP,IAAM,oBAAoB;AAC1B,IAAM,yBAAyB;AAC/B,IAAM,wBAAwB;AAK9B,SAAS,UAA4B,MAAuC;AAC1E,SAAO,KAAK,SAAS,OAAO,CAAC,MAAM,EAAE,iBAAiB,IAAI;AAC5D;AAQA,SAAS,WAA6B,kBAA6C;AACjF,MAAI,iBAAiB,SAAS,EAAG,QAAO;AACxC,QAAM,OAAO,oBAAI,IAAY;AAC7B,aAAW,KAAK,kBAAkB;AAChC,eAAW,KAAK,OAAO,KAAK,EAAE,YAAsC,EAAG,MAAK,IAAI,CAAC;AAAA,EACnF;AACA,MAAI,QAAQ;AACZ,aAAW,KAAK,MAAM;AACpB,QAAI,MAAM;AACV,QAAI,MAAM;AACV,eAAW,KAAK,kBAAkB;AAChC,YAAM,QAAS,EAAE,aAAwC,CAAC;AAC1D,UAAI,UAAU,OAAW;AACzB,UAAI,QAAQ,IAAK,OAAM;AACvB,UAAI,QAAQ,IAAK,OAAM;AAAA,IACzB;AACA,QAAI,MAAM,aAAa,MAAM,MAAM,MAAO,SAAQ,MAAM;AAAA,EAC1D;AACA,SAAO;AACT;AAgBO,SAAS,cACd,OACA,aAA8B,CAAC,GACjB;AACd,MAAI,MAAM,WAAW,GAAG;AACtB,UAAM,IAAI,MAAM,2DAAsD;AAAA,EACxE;AACA,QAAM,WAAW,WAAW,YAAY;AACxC,QAAM,gBAAgB,WAAW,iBAAiB;AAClD,QAAM,eAAe,WAAW,gBAAgB;AAMhD,QAAM,YAAY,MAAM,OAAO,CAAC,GAAG,OAAO,KAAK,IAAI,GAAG,UAAU,EAAE,EAAE,MAAM,GAAG,CAAC;AAC9E,QAAM,cAA8B,MAAM,KAAK,EAAE,QAAQ,UAAU,GAAG,MAAM,CAAC,CAAC;AAC9E,QAAM,gBAAwC,CAAC;AAC/C,QAAM,aAAwD,CAAC;AAC/D,QAAM,eAAqD,CAAC;AAE5D,aAAW,QAAQ,OAAO;AACxB,UAAM,OAAO,UAAU,IAAI;AAC3B,QAAI,KAAK,SAAS,aAAc,cAAa,KAAK,EAAE,QAAQ,KAAK,QAAQ,GAAG,KAAK,OAAO,CAAC;AAEzF,UAAM,SAAS,WAAW,IAAI;AAC9B,kBAAc,KAAK,MAAM,IAAI;AAC7B,QAAI,SAAS,cAAe,YAAW,KAAK,EAAE,QAAQ,KAAK,QAAQ,OAAO,CAAC;AAE3E,QAAI,KAAK,UAAU,GAAG;AACpB,YAAM,OAAO,MAAM;AAAA,QACjB,IAAI,IAAI,KAAK,QAAQ,CAAC,MAAM,OAAO,KAAK,EAAE,YAAsC,CAAC,CAAC;AAAA,MACpF,EAAE,KAAK;AACP,WAAK,QAAQ,CAAC,GAAG,aAAa;AAG5B,cAAM,SAAU,YAAY,QAAQ,MAAM,CAAC;AAC3C,cAAM,KAAK,EAAE;AACb,mBAAW,KAAK,MAAM;AACpB,gBAAM,QAAQ,GAAG,CAAC;AAClB,cAAI,UAAU,OAAW;AACzB,iBAAO,KAAK;AAAA,YACV,WAAW,EAAE;AAAA,YACb,WAAW,GAAG,KAAK,MAAM,KAAK,CAAC;AAAA,YAC/B;AAAA,YACA,WAAW,EAAE,aAAa;AAAA,UAC5B,CAAC;AAAA,QACH;AAAA,MACF,CAAC;AAAA,IACH;AAAA,EACF;AAEA,QAAM,MAAM,sBAAsB,WAAW;AAE7C,QAAM,eAAyB,CAAC;AAChC,MAAI,MAAM,UAAU;AAClB,iBAAa,KAAK,WAAW,MAAM,GAAG,CAAC,MAAM,QAAQ,EAAE;AAAA,EACzD;AACA,aAAW,EAAE,QAAQ,OAAO,KAAK,YAAY;AAC3C,iBAAa,KAAK,YAAY,MAAM,WAAW,MAAM,MAAM,CAAC,MAAM,aAAa,sBAAiB;AAAA,EAClG;AACA,aAAW,EAAE,QAAQ,EAAE,KAAK,cAAc;AACxC,iBAAa,KAAK,YAAY,MAAM,KAAK,CAAC,uBAAuB,YAAY,EAAE;AAAA,EACjF;AAEA,SAAO;AAAA,IACL,aAAa,aAAa,WAAW;AAAA,IACrC;AAAA,IACA,uBAAuB;AAAA,IACvB;AAAA,EACF;AACF;AAGA,SAAS,MAAM,GAAmB;AAChC,SAAO,KAAK,MAAM,IAAI,GAAG,IAAI;AAC/B;;;ADjFA,SAAS,0BAAAA,+BAA8B;AAMvC;AAAA,EACE;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,OACK;AAmBP,SAAS,mBAAmB;AA9ErB,SAAS,mBACd,KACmC;AACnC,QAAM,OAAO,IAAI,aAAa;AAC9B,MAAI,OAAO,GAAG;AACZ,UAAM,IAAI,MAAM,mDAAmD,IAAI,GAAG;AAAA,EAC5E;AACA,MAAI,IAAI,OAAO,WAAW,GAAG;AAC3B,UAAM,IAAI,MAAM,qCAAqC;AAAA,EACvD;AACA,SAAO;AAAA,IACL,MAAM,IAAI;AAAA,IACV,YAAY,IAAI,OAAO,IAAI,CAAC,SAAS,EAAE,KAAK,aAAa,IAAI,WAAW,GAAG,KAAK,IAAI,EAAE;AAAA,IACtF,MAAM,MAAM,EAAE,UAAU,UAAU,OAAO,GAAwB;AAC/D,YAAM,UAAU,MAAM,QAAQ;AAAA,QAC5B,MAAM,KAAK,EAAE,QAAQ,KAAK,GAAG,CAAC,GAAG,QAAQ,IAAI,SAAS,EAAE,UAAU,UAAU,QAAQ,IAAI,CAAC,CAAC;AAAA,MAC5F;AACA,YAAM,WAA8B,QAAQ;AAAA,QAAI,CAAC,GAAG,QAClD,EAAE,WAAW,cACT,EAAE,QACF,EAAE,OAAO,GAAG,IAAI,IAAI,OAAO,GAAG,IAAI,cAAc,MAAM,WAAW,OAAO,EAAE,MAAM,EAAE;AAAA,MACxF;AAEA,YAAM,MAAM,uBAAuB,UAAU,IAAI,QAAQ,IAAI,OAAO;AACpE,aAAO,EAAE,WAAW,IAAI,WAAW,YAAY,IAAI,cAAc,OAAO,IAAI,UAAU;AAAA,IACxF;AAAA,EACF;AACF;","names":["aggregateJudgeVerdicts"]}
1
+ {"version":3,"sources":["../../src/eval-campaign/index.ts","../../src/eval-campaign/trust-gate.ts"],"sourcesContent":["/**\n * Eval-campaign — the app-shell's curated surface for a product's\n * self-improvement loop, NOT a reimplementation.\n *\n * The loop ENGINE lives in `@tangle-network/agent-eval` (a peer dependency):\n * `selfImprove` already owns execution, scoring, data separation, release\n * decisions, durable provenance, and hosted ingest. Candidate search is always\n * explicit: pass an official optimization `method`, or pass a caller-owned\n * `SurfaceProposer`.\n *\n * This module adds the one piece `selfImprove` does not own and which every\n * multi-model product re-hand-rolls — the ensemble judge:\n *\n * {@link buildEnsembleJudge} — turn a per-rubric `scoreOne` into a\n * `JudgeConfig` that fans out N uncorrelated judge calls and reduces them via\n * the substrate's `aggregateJudgeVerdicts` (survivor-mean, inter-rater spread,\n * fail-loud on all-failed). A product writes its rubric + one judge call; the\n * fan-out, partial-failure handling, and composite are the scaffold's.\n *\n * Everything else is a curated re-export so a product has ONE eval import:\n * `selfImprove` + release policies + optimization methods + their types. See\n * `.claude/skills/eval-campaign/SKILL.md` for the wiring contract.\n */\n\nimport {\n aggregateJudgeVerdicts,\n type JudgeVerdict,\n} from '@tangle-network/agent-eval'\nimport type {\n JudgeConfig,\n JudgeScore,\n Scenario,\n} from '@tangle-network/agent-eval/campaign'\n\n/** Config for {@link buildEnsembleJudge}. `D` = the rubric's dimension union. */\nexport interface EnsembleJudgeConfig<TArtifact, TScenario extends Scenario, D extends string> {\n /** Judge name — appears in traces and scorecards. */\n name: string\n /** Scoring revision for campaign caches. Change it when rubric, model, or callback settings change. */\n judgeVersion?: JudgeConfig<TArtifact, TScenario>['judgeVersion']\n /** Stable-ordered rubric dimensions. Drives the `JudgeDimension` list AND the\n * reducer keys, so a judge that omits a dimension scores it 0 (never silently\n * dropped). */\n rubric: readonly D[]\n /**\n * Score ONE artifact on the rubric → a raw per-dimension verdict. Called\n * `judgeReps` times per artifact; vary the model by `rep` for an uncorrelated\n * ensemble (judges that share a base model share its bias). Return\n * `{ model, perDimension: null }` to record a judge failure WITHOUT killing\n * the ensemble; throw only on an unrecoverable error (the whole rep is then\n * treated as a failed judge).\n * Use the supplied costLedger, costPhase, and costTags to record paid judge calls.\n */\n scoreOne: (input: Parameters<JudgeConfig<TArtifact, TScenario>['score']>[0] & {\n rep: number\n }) => Promise<JudgeVerdict<D>>\n /** Independent judge calls per artifact, reduced by `aggregateJudgeVerdicts`.\n * Default 1. Raise (with model variety in `scoreOne`) for inter-rater bands. */\n judgeReps?: number\n /** Per-dimension composite weights. Default: uniform over `rubric`. A partial\n * map selects-and-weights exactly the named dimensions. */\n weights?: Partial<Record<D, number>>\n /** Optional human-readable dimension descriptions. Default: the key itself. */\n describe?: (dim: D) => string\n}\n\n/**\n * Build a `JudgeConfig` whose `score()` fans out `judgeReps` independent\n * `scoreOne` calls and reduces them with the substrate's\n * `aggregateJudgeVerdicts`. A single judge call failing does NOT fail the cell\n * (it is recorded and dropped); only ALL judges failing throws — which the\n * campaign records as a failed cell, never a silent zero.\n *\n * Pass the result straight to `selfImprove({ judge })` (or `runCampaign`).\n */\nexport function buildEnsembleJudge<TArtifact, TScenario extends Scenario, D extends string>(\n cfg: EnsembleJudgeConfig<TArtifact, TScenario, D>,\n): JudgeConfig<TArtifact, TScenario> {\n const reps = cfg.judgeReps ?? 1\n if (reps < 1) {\n throw new Error(`buildEnsembleJudge: judgeReps must be >= 1 (got ${reps})`)\n }\n if (cfg.rubric.length === 0) {\n throw new Error('buildEnsembleJudge: rubric is empty')\n }\n return {\n name: cfg.name,\n ...(cfg.judgeVersion === undefined ? {} : { judgeVersion: cfg.judgeVersion }),\n dimensions: cfg.rubric.map((key) => ({ key, description: cfg.describe?.(key) ?? key })),\n async score(input): Promise<JudgeScore> {\n const settled = await Promise.allSettled(\n Array.from({ length: reps }, (_, rep) => cfg.scoreOne({ ...input, rep })),\n )\n const verdicts: JudgeVerdict<D>[] = settled.map((r, rep) =>\n r.status === 'fulfilled'\n ? r.value\n : { model: `${cfg.name}-rep${rep}`, perDimension: null, rationale: String(r.reason) },\n )\n // Throws iff EVERY rep failed → the campaign records a failed cell.\n const agg = aggregateJudgeVerdicts(verdicts, cfg.rubric, cfg.weights)\n return { composite: agg.composite, dimensions: agg.perDimension, notes: agg.rationale }\n },\n }\n}\n\n// Agreement across items is distinct from evaluator accuracy against independent controls.\n// Consumers can opt into auditEvaluator from agent-eval/meta-eval for that separate evidence.\nexport {\n trustVerdicts,\n type TrustItem,\n type TrustThresholds,\n type TrustVerdict,\n} from './trust-gate'\n\n// ── Curated re-exports — the one eval import for a product loop ──────────────\n// The loop engine + gates + drivers + the ensemble reducer, so a product wires\n// its self-improvement loop from a single module instead of reaching across\n// three agent-eval subpaths. All DOWNWARD imports (agent-app consumes the\n// substrate); the layering rule is preserved.\n\nexport { aggregateJudgeVerdicts } from '@tangle-network/agent-eval'\nexport type {\n EnsembleAggregate,\n JudgeVerdict,\n RunRecord,\n} from '@tangle-network/agent-eval'\nexport {\n compareOptimizationMethods,\n defaultProductionGate,\n externalTextOptimizationMethod,\n gepaOptimizationMethod,\n paretoSignificanceGate,\n runCampaign,\n skillOptOptimizationMethod,\n} from '@tangle-network/agent-eval/campaign'\nexport type {\n CampaignResult,\n CompareOptimizationMethodsOptions,\n DispatchContext,\n ExternalTextOptimizationMethodConfig,\n Gate,\n GepaOptimizationMethodConfig,\n JudgeConfig,\n JudgeDimension,\n JudgeScore,\n LabeledScenarioStore,\n MutableSurface,\n OptimizationMethod,\n OptimizationMethodResult,\n Scenario,\n SkillOptOptimizationMethodConfig,\n SurfaceProposer,\n} from '@tangle-network/agent-eval/campaign'\nexport { selfImprove } from '@tangle-network/agent-eval/contract'\nexport type {\n SelfImproveBudget,\n SelfImproveOptions,\n SelfImproveResult,\n} from '@tangle-network/agent-eval/contract'\n","/**\n * Summarize rater agreement, within-item spread, and surviving-judge coverage.\n * Consumers choose the thresholds that a trustworthy result must satisfy.\n * Agreement does not measure evaluator errors against independent controls.\n * Use agent-eval/meta-eval's auditEvaluator when the consumer needs that separate evidence.\n * Spread stays within each item so differences in task quality cannot mimic rater disagreement.\n */\n\nimport {\n interRaterReliability,\n type DimensionJudgeScore,\n type JudgeVerdict,\n} from '@tangle-network/agent-eval'\n\n/** One item's raters: the per-judge verdicts {@link aggregateJudgeVerdicts}\n * reduces, tagged with the item they scored so spread stays within-item. */\nexport interface TrustItem<D extends string = string> {\n /** Stable item identifier — surfaces in `perItemSpread` and `trustReasons`. */\n itemId: string\n /** The raters' verdicts for THIS item (one per judge call). A failed judge\n * (`perDimension: null`) is dropped before spread/IRR, never folded as 0. */\n verdicts: readonly JudgeVerdict<D>[]\n}\n\n/** Configurable agreement and coverage thresholds for {@link trustVerdicts}. */\nexport interface TrustThresholds {\n /** Minimum corpus inter-rater reliability (Krippendorff-style α). Default 0.2. */\n irrFloor?: number\n /** Maximum per-item rater spread (`max − min` over a single item's surviving\n * raters, across its dimensions). Above this the raters split ON THAT ITEM.\n * Default 0.5. */\n spreadCeiling?: number\n /** Minimum surviving (non-failed) raters required per item. Default 3. */\n minSurvivors?: number\n}\n\n/** Result of the trust gate. `trustworthy` iff every check passed; `trustReasons`\n * is empty iff `trustworthy`. */\nexport interface TrustVerdict {\n /** True iff IRR ≥ floor AND every item's spread ≤ ceiling AND every item has\n * ≥ `minSurvivors` surviving raters. */\n trustworthy: boolean\n /** One entry per FAILED check, each naming its number + the offending value.\n * Empty iff `trustworthy`. */\n trustReasons: string[]\n /** Corpus inter-rater reliability actually measured (the check-1 value). */\n interRaterReliability: number\n /** Per-item spread (`max − min` over surviving raters, max over dimensions),\n * keyed by `itemId`. The check-2 input, surfaced for drill-down. */\n perItemSpread: Record<string, number>\n}\n\nconst DEFAULT_IRR_FLOOR = 0.2\nconst DEFAULT_SPREAD_CEILING = 0.5\nconst DEFAULT_MIN_SURVIVORS = 3\n\n/** Surviving (non-failed) verdicts for an item — those with a real\n * `perDimension` map. A failed judge carries no scores and is excluded from\n * every statistic (it is NOT a zero rater). */\nfunction survivors<D extends string>(item: TrustItem<D>): JudgeVerdict<D>[] {\n return item.verdicts.filter((v) => v.perDimension !== null)\n}\n\n/**\n * Within-item rater spread: for each dimension, `max − min` across the item's\n * surviving raters; the item's spread is the max over its dimensions (the worst\n * dimension the raters split on). Pooled ONLY within this one item — never\n * across items — so a quality gap between items cannot inflate it.\n */\nfunction itemSpread<D extends string>(survivorVerdicts: JudgeVerdict<D>[]): number {\n if (survivorVerdicts.length < 2) return 0\n const dims = new Set<string>()\n for (const v of survivorVerdicts) {\n for (const d of Object.keys(v.perDimension as Record<string, number>)) dims.add(d)\n }\n let worst = 0\n for (const d of dims) {\n let min = Infinity\n let max = -Infinity\n for (const v of survivorVerdicts) {\n const score = (v.perDimension as Record<string, number>)[d]\n if (score === undefined) continue\n if (score < min) min = score\n if (score > max) max = score\n }\n if (max > -Infinity && max - min > worst) worst = max - min\n }\n return worst\n}\n\n/**\n * Check an ensemble against its configured agreement and coverage thresholds. Pure: no LLM, no I/O, no clock, no random —\n * the same `items` + `thresholds` always yield the same verdict.\n *\n * Sibling to {@link aggregateJudgeVerdicts}: that reduces ONE item's raters to a\n * composite; this summarizes agreement across the supplied items.\n * A passing result does not establish evaluator accuracy or authorize a release.\n *\n * @throws if `items` is empty — an empty corpus has no measurable trust, and a\n * silent `trustworthy: true` over zero evidence is the exact lie the gate\n * exists to refuse.\n */\nexport function trustVerdicts<D extends string>(\n items: readonly TrustItem<D>[],\n thresholds: TrustThresholds = {},\n): TrustVerdict {\n if (items.length === 0) {\n throw new Error('trustVerdicts: items is empty — no evidence to trust')\n }\n const irrFloor = thresholds.irrFloor ?? DEFAULT_IRR_FLOOR\n const spreadCeiling = thresholds.spreadCeiling ?? DEFAULT_SPREAD_CEILING\n const minSurvivors = thresholds.minSurvivors ?? DEFAULT_MIN_SURVIVORS\n\n // Rater-major JudgeScore series for the substrate's IRR. Each item's surviving\n // raters are assigned a stable column index so the same rater across items\n // lines up; per (item, dimension) one JudgeScore per rater, in item-then-\n // dimension order — the layout interRaterReliability chunks back into items.\n const maxRaters = items.reduce((m, it) => Math.max(m, survivors(it).length), 0)\n const raterSeries: DimensionJudgeScore[][] = Array.from({ length: maxRaters }, () => [])\n const perItemSpread: Record<string, number> = {}\n const splitItems: Array<{ itemId: string; spread: number }> = []\n const starvedItems: Array<{ itemId: string; n: number }> = []\n\n for (const item of items) {\n const surv = survivors(item)\n if (surv.length < minSurvivors) starvedItems.push({ itemId: item.itemId, n: surv.length })\n\n const spread = itemSpread(surv)\n perItemSpread[item.itemId] = spread\n if (spread > spreadCeiling) splitItems.push({ itemId: item.itemId, spread })\n\n if (surv.length >= 2) {\n const dims = Array.from(\n new Set(surv.flatMap((v) => Object.keys(v.perDimension as Record<string, number>))),\n ).sort()\n surv.forEach((v, raterIdx) => {\n // raterIdx < surv.length ≤ maxRaters = raterSeries.length, so the column\n // always exists; the ??= keeps the access provably defined for the type.\n const column = (raterSeries[raterIdx] ??= [])\n const pd = v.perDimension as Record<string, number>\n for (const d of dims) {\n const score = pd[d]\n if (score === undefined) continue\n column.push({\n judgeName: v.model,\n dimension: `${item.itemId}::${d}`,\n score,\n reasoning: v.rationale ?? '',\n })\n }\n })\n }\n }\n\n const irr = interRaterReliability(raterSeries)\n\n const trustReasons: string[] = []\n if (irr < irrFloor) {\n trustReasons.push(`(1) IRR ${round(irr)} < ${irrFloor}`)\n }\n for (const { itemId, spread } of splitItems) {\n trustReasons.push(`(2) item ${itemId} spread ${round(spread)} > ${spreadCeiling} — raters split`)\n }\n for (const { itemId, n } of starvedItems) {\n trustReasons.push(`(3) item ${itemId}: ${n} surviving raters < ${minSurvivors}`)\n }\n\n return {\n trustworthy: trustReasons.length === 0,\n trustReasons,\n interRaterReliability: irr,\n perItemSpread,\n }\n}\n\n/** Round to 2 decimals for stable, readable reason strings. */\nfunction round(n: number): number {\n return Math.round(n * 100) / 100\n}\n"],"mappings":";AAwBA;AAAA,EACE;AAAA,OAEK;;;ACnBP;AAAA,EACE;AAAA,OAGK;AAwCP,IAAM,oBAAoB;AAC1B,IAAM,yBAAyB;AAC/B,IAAM,wBAAwB;AAK9B,SAAS,UAA4B,MAAuC;AAC1E,SAAO,KAAK,SAAS,OAAO,CAAC,MAAM,EAAE,iBAAiB,IAAI;AAC5D;AAQA,SAAS,WAA6B,kBAA6C;AACjF,MAAI,iBAAiB,SAAS,EAAG,QAAO;AACxC,QAAM,OAAO,oBAAI,IAAY;AAC7B,aAAW,KAAK,kBAAkB;AAChC,eAAW,KAAK,OAAO,KAAK,EAAE,YAAsC,EAAG,MAAK,IAAI,CAAC;AAAA,EACnF;AACA,MAAI,QAAQ;AACZ,aAAW,KAAK,MAAM;AACpB,QAAI,MAAM;AACV,QAAI,MAAM;AACV,eAAW,KAAK,kBAAkB;AAChC,YAAM,QAAS,EAAE,aAAwC,CAAC;AAC1D,UAAI,UAAU,OAAW;AACzB,UAAI,QAAQ,IAAK,OAAM;AACvB,UAAI,QAAQ,IAAK,OAAM;AAAA,IACzB;AACA,QAAI,MAAM,aAAa,MAAM,MAAM,MAAO,SAAQ,MAAM;AAAA,EAC1D;AACA,SAAO;AACT;AAcO,SAAS,cACd,OACA,aAA8B,CAAC,GACjB;AACd,MAAI,MAAM,WAAW,GAAG;AACtB,UAAM,IAAI,MAAM,2DAAsD;AAAA,EACxE;AACA,QAAM,WAAW,WAAW,YAAY;AACxC,QAAM,gBAAgB,WAAW,iBAAiB;AAClD,QAAM,eAAe,WAAW,gBAAgB;AAMhD,QAAM,YAAY,MAAM,OAAO,CAAC,GAAG,OAAO,KAAK,IAAI,GAAG,UAAU,EAAE,EAAE,MAAM,GAAG,CAAC;AAC9E,QAAM,cAAuC,MAAM,KAAK,EAAE,QAAQ,UAAU,GAAG,MAAM,CAAC,CAAC;AACvF,QAAM,gBAAwC,CAAC;AAC/C,QAAM,aAAwD,CAAC;AAC/D,QAAM,eAAqD,CAAC;AAE5D,aAAW,QAAQ,OAAO;AACxB,UAAM,OAAO,UAAU,IAAI;AAC3B,QAAI,KAAK,SAAS,aAAc,cAAa,KAAK,EAAE,QAAQ,KAAK,QAAQ,GAAG,KAAK,OAAO,CAAC;AAEzF,UAAM,SAAS,WAAW,IAAI;AAC9B,kBAAc,KAAK,MAAM,IAAI;AAC7B,QAAI,SAAS,cAAe,YAAW,KAAK,EAAE,QAAQ,KAAK,QAAQ,OAAO,CAAC;AAE3E,QAAI,KAAK,UAAU,GAAG;AACpB,YAAM,OAAO,MAAM;AAAA,QACjB,IAAI,IAAI,KAAK,QAAQ,CAAC,MAAM,OAAO,KAAK,EAAE,YAAsC,CAAC,CAAC;AAAA,MACpF,EAAE,KAAK;AACP,WAAK,QAAQ,CAAC,GAAG,aAAa;AAG5B,cAAM,SAAU,YAAY,QAAQ,MAAM,CAAC;AAC3C,cAAM,KAAK,EAAE;AACb,mBAAW,KAAK,MAAM;AACpB,gBAAM,QAAQ,GAAG,CAAC;AAClB,cAAI,UAAU,OAAW;AACzB,iBAAO,KAAK;AAAA,YACV,WAAW,EAAE;AAAA,YACb,WAAW,GAAG,KAAK,MAAM,KAAK,CAAC;AAAA,YAC/B;AAAA,YACA,WAAW,EAAE,aAAa;AAAA,UAC5B,CAAC;AAAA,QACH;AAAA,MACF,CAAC;AAAA,IACH;AAAA,EACF;AAEA,QAAM,MAAM,sBAAsB,WAAW;AAE7C,QAAM,eAAyB,CAAC;AAChC,MAAI,MAAM,UAAU;AAClB,iBAAa,KAAK,WAAW,MAAM,GAAG,CAAC,MAAM,QAAQ,EAAE;AAAA,EACzD;AACA,aAAW,EAAE,QAAQ,OAAO,KAAK,YAAY;AAC3C,iBAAa,KAAK,YAAY,MAAM,WAAW,MAAM,MAAM,CAAC,MAAM,aAAa,sBAAiB;AAAA,EAClG;AACA,aAAW,EAAE,QAAQ,EAAE,KAAK,cAAc;AACxC,iBAAa,KAAK,YAAY,MAAM,KAAK,CAAC,uBAAuB,YAAY,EAAE;AAAA,EACjF;AAEA,SAAO;AAAA,IACL,aAAa,aAAa,WAAW;AAAA,IACrC;AAAA,IACA,uBAAuB;AAAA,IACvB;AAAA,EACF;AACF;AAGA,SAAS,MAAM,GAAmB;AAChC,SAAO,KAAK,MAAM,IAAI,GAAG,IAAI;AAC/B;;;AD1DA,SAAS,0BAAAA,+BAA8B;AAMvC;AAAA,EACE;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,OACK;AAmBP,SAAS,mBAAmB;AA9ErB,SAAS,mBACd,KACmC;AACnC,QAAM,OAAO,IAAI,aAAa;AAC9B,MAAI,OAAO,GAAG;AACZ,UAAM,IAAI,MAAM,mDAAmD,IAAI,GAAG;AAAA,EAC5E;AACA,MAAI,IAAI,OAAO,WAAW,GAAG;AAC3B,UAAM,IAAI,MAAM,qCAAqC;AAAA,EACvD;AACA,SAAO;AAAA,IACL,MAAM,IAAI;AAAA,IACV,GAAI,IAAI,iBAAiB,SAAY,CAAC,IAAI,EAAE,cAAc,IAAI,aAAa;AAAA,IAC3E,YAAY,IAAI,OAAO,IAAI,CAAC,SAAS,EAAE,KAAK,aAAa,IAAI,WAAW,GAAG,KAAK,IAAI,EAAE;AAAA,IACtF,MAAM,MAAM,OAA4B;AACtC,YAAM,UAAU,MAAM,QAAQ;AAAA,QAC5B,MAAM,KAAK,EAAE,QAAQ,KAAK,GAAG,CAAC,GAAG,QAAQ,IAAI,SAAS,EAAE,GAAG,OAAO,IAAI,CAAC,CAAC;AAAA,MAC1E;AACA,YAAM,WAA8B,QAAQ;AAAA,QAAI,CAAC,GAAG,QAClD,EAAE,WAAW,cACT,EAAE,QACF,EAAE,OAAO,GAAG,IAAI,IAAI,OAAO,GAAG,IAAI,cAAc,MAAM,WAAW,OAAO,EAAE,MAAM,EAAE;AAAA,MACxF;AAEA,YAAM,MAAM,uBAAuB,UAAU,IAAI,QAAQ,IAAI,OAAO;AACpE,aAAO,EAAE,WAAW,IAAI,WAAW,YAAY,IAAI,cAAc,OAAO,IAAI,UAAU;AAAA,IACxF;AAAA,EACF;AACF;","names":["aggregateJudgeVerdicts"]}
@@ -1,28 +1,9 @@
1
1
  /**
2
- * Trust gate — decides whether an ensemble's scores are allowed to be BELIEVED,
3
- * one level up from {@link aggregateJudgeVerdicts} (which only reduces ONE
4
- * artifact's raters to a composite). A composite is a number; this is the check
5
- * that the number means anything. It is the code "Enforced by" for the
6
- * measurement-validation skill's after-gate ("is this result allowed to be
7
- * believed").
8
- *
9
- * Three checks, each fail-loud and named in `trustReasons`:
10
- * (1) inter-rater reliability over the corpus ≥ `irrFloor` — raters that
11
- * disagree no better than chance carry no signal to optimize against.
12
- * (2) per-item rater spread ≤ `spreadCeiling` — for EACH item, raters must
13
- * converge on THAT item.
14
- * (3) surviving raters per item ≥ `minSurvivors` — a mean over one or two
15
- * raters is an anecdote, not an ensemble.
16
- *
17
- * CRITICAL metric semantics — per-item spread is rater disagreement about the
18
- * SAME item: `max(score) − min(score)` across the raters that scored THAT item
19
- * (max over its dimensions), never pooled across different items or across the
20
- * baseline/candidate sides. Pooling reads a genuine quality gap BETWEEN items as
21
- * "the raters split" and so trips the gate exactly when the finding is largest —
22
- * the failure mode the after-gate exists to prevent. The corpus IRR (check 1)
23
- * leans on the substrate's `interRaterReliability`, whose expected-disagreement
24
- * denominator already pools across items, so genuine item-to-item variation
25
- * RAISES reliability rather than lowering it.
2
+ * Summarize rater agreement, within-item spread, and surviving-judge coverage.
3
+ * Consumers choose the thresholds that a trustworthy result must satisfy.
4
+ * Agreement does not measure evaluator errors against independent controls.
5
+ * Use agent-eval/meta-eval's auditEvaluator when the consumer needs that separate evidence.
6
+ * Spread stays within each item so differences in task quality cannot mimic rater disagreement.
26
7
  */
27
8
  import { type JudgeVerdict } from '@tangle-network/agent-eval';
28
9
  /** One item's raters: the per-judge verdicts {@link aggregateJudgeVerdicts}
@@ -34,11 +15,9 @@ export interface TrustItem<D extends string = string> {
34
15
  * (`perDimension: null`) is dropped before spread/IRR, never folded as 0. */
35
16
  verdicts: readonly JudgeVerdict<D>[];
36
17
  }
37
- /** Thresholds for {@link trustVerdicts}. All overridable; defaults are the
38
- * conservative after-gate bar. */
18
+ /** Configurable agreement and coverage thresholds for {@link trustVerdicts}. */
39
19
  export interface TrustThresholds {
40
- /** Minimum corpus inter-rater reliability (Krippendorff-style α). Below this
41
- * the raters agree no better than chance. Default 0.2. */
20
+ /** Minimum corpus inter-rater reliability (Krippendorff-style α). Default 0.2. */
42
21
  irrFloor?: number;
43
22
  /** Maximum per-item rater spread (`max − min` over a single item's surviving
44
23
  * raters, across its dimensions). Above this the raters split ON THAT ITEM.
@@ -63,14 +42,12 @@ export interface TrustVerdict {
63
42
  perItemSpread: Record<string, number>;
64
43
  }
65
44
  /**
66
- * Decide whether an ensemble's per-item verdicts are trustworthy enough to
67
- * believe a lift computed from them. Pure: no LLM, no I/O, no clock, no random —
45
+ * Check an ensemble against its configured agreement and coverage thresholds. Pure: no LLM, no I/O, no clock, no random —
68
46
  * the same `items` + `thresholds` always yield the same verdict.
69
47
  *
70
48
  * Sibling to {@link aggregateJudgeVerdicts}: that reduces ONE item's raters to a
71
- * composite; this audits the raters ACROSS items and reports whether the
72
- * composites are believable. Run it on the corpus of held-out items before
73
- * reporting any lift over their scores.
49
+ * composite; this summarizes agreement across the supplied items.
50
+ * A passing result does not establish evaluator accuracy or authorize a release.
74
51
  *
75
52
  * @throws if `items` is empty — an empty corpus has no measurable trust, and a
76
53
  * silent `trustworthy: true` over zero evidence is the exact lie the gate
@@ -13,12 +13,12 @@ import {
13
13
  resolveTangleSsoAccount,
14
14
  signSessionCookieValue,
15
15
  verifySignedSsoState
16
- } from "../chunk-TXM4SS46.js";
16
+ } from "../chunk-CSMJM7YT.js";
17
17
  import {
18
18
  resolveTangleDevOrUserKey,
19
19
  resolveTangleExecutionEnvironment
20
20
  } from "../chunk-JML7WKWU.js";
21
- import "../chunk-3YUI7WNW.js";
21
+ import "../chunk-IAY2DMKW.js";
22
22
  import "../chunk-TXD5HXLE.js";
23
23
 
24
24
  // src/platform/hub.ts
@@ -1,9 +1,9 @@
1
1
  import {
2
2
  createChatTurnRoutes,
3
3
  streamChatRouteAsSandboxEvents
4
- } from "../chunk-42NBWR4U.js";
4
+ } from "../chunk-SYWEMKZR.js";
5
5
  import "../chunk-ZSSCVT6H.js";
6
- import "../chunk-3YUI7WNW.js";
6
+ import "../chunk-IAY2DMKW.js";
7
7
  import "../chunk-4PZE7XAM.js";
8
8
  import "../chunk-ZVEEWGDK.js";
9
9
  import "../chunk-X47R2IVO.js";
@@ -18,7 +18,7 @@ import {
18
18
  } from "../chunk-6A7MYOUI.js";
19
19
  import {
20
20
  assertMediaUrl
21
- } from "../chunk-3YUI7WNW.js";
21
+ } from "../chunk-IAY2DMKW.js";
22
22
  import "../chunk-TXD5HXLE.js";
23
23
 
24
24
  // src/sequences/operations.ts
@@ -0,0 +1,84 @@
1
+ /**
2
+ * Web-boundary utilities every agent app's routes hand-roll: JSON body parsing
3
+ * + narrowing, request-context extraction (real client IP behind Cloudflare),
4
+ * a KV-backed sliding-window rate limiter, the free-route budget policy built
5
+ * on it, and security response headers. Pure mechanism — no DB, no domain. The
6
+ * KV is a structural interface so this needs no `@cloudflare/workers-types`
7
+ * dependency.
8
+ */
9
+ export * from './rate-limit';
10
+ export * from './free-route-limit';
11
+ export type JsonObject = Record<string, unknown>;
12
+ /** Parse + object-narrow a Request body. `[body, null]` on success, `[null,
13
+ * errorResponse]` on a non-object body (callers `if (err) return err`). */
14
+ export declare function parseJsonObjectBody(request: Request): Promise<[JsonObject, null] | [null, Response]>;
15
+ /** Narrow one required string field, 400 if missing/empty. */
16
+ export declare function requireString(body: JsonObject, field: string): string | Response;
17
+ /** Define the context of a request including IP address, user agent, timestamp, and request ID */
18
+ export interface RequestContext {
19
+ ipAddress: string;
20
+ userAgent: string;
21
+ timestamp: string;
22
+ requestId: string;
23
+ }
24
+ /** Extract request context for audit trails. Uses `CF-Connecting-IP` for the
25
+ * real client IP behind Cloudflare. */
26
+ export declare function extractRequestContext(request: Request): RequestContext;
27
+ /** Define options for configuring cookie attributes and behavior */
28
+ export interface CookieOptions {
29
+ name: string;
30
+ /** Default '/'. */
31
+ path?: string;
32
+ /** Default true. */
33
+ httpOnly?: boolean;
34
+ /** Adds the `Secure` attribute. Default false. */
35
+ secure?: boolean;
36
+ /** Default 'Lax'. */
37
+ sameSite?: 'Lax' | 'Strict' | 'None';
38
+ maxAgeSeconds?: number;
39
+ }
40
+ /** Serialize a Set-Cookie header value: `name=encodeURIComponent(value)` plus
41
+ * attributes in Path / HttpOnly / SameSite / Max-Age / Secure order.
42
+ * Throws on `SameSite=None` without `secure` — browsers silently drop that
43
+ * combination, which would otherwise fail invisibly. */
44
+ export declare function serializeCookie(value: string, opts: CookieOptions): string;
45
+ /** Set-Cookie header value that deletes the cookie (empty value, Max-Age=0). */
46
+ export declare function clearCookieHeader(opts: Omit<CookieOptions, 'maxAgeSeconds'>): string;
47
+ /** Read + decode one cookie from a Cookie request header; null when absent. */
48
+ export declare function readCookieValue(cookieHeader: string | null, name: string): string | null;
49
+ /** Define options for configuring security-related HTTP headers including disclaimers and retention labels */
50
+ export interface SecurityHeaderOptions {
51
+ /** Product disclaimer (e.g. "AI-powered tool. Not legal advice."). Omitted if absent. */
52
+ disclaimer?: string;
53
+ /** Data-retention label (e.g. "7-years"). Omitted if absent. */
54
+ retention?: string;
55
+ /** Extra headers to set. */
56
+ extra?: Record<string, string>;
57
+ }
58
+ /** Canonical generic response headers used by {@link addSecurityHeaders}.
59
+ * Exported so static-asset hosts can apply the same policy without copying
60
+ * values that silently drift from Worker/API responses. */
61
+ export declare const STANDARD_SECURITY_HEADERS: Readonly<{
62
+ readonly 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload';
63
+ readonly 'X-Content-Type-Options': 'nosniff';
64
+ readonly 'X-Frame-Options': 'SAMEORIGIN';
65
+ readonly 'Referrer-Policy': 'same-origin';
66
+ readonly 'X-XSS-Protection': '1; mode=block';
67
+ }>;
68
+ /** Set standard security headers on a response (HSTS, nosniff, frame-options,
69
+ * referrer-policy, XSS) + optional product disclaimer/retention. The security
70
+ * set is generic; the disclaimer/retention are the product's. */
71
+ export declare function addSecurityHeaders(response: Response, opts?: SecurityHeaderOptions): Response;
72
+ /**
73
+ * Canonical media-reference boundary shared by every surface that persists a
74
+ * media url (sequences clips, design-canvas image/video src). The ONE rule:
75
+ * remote `http(s)` or a rooted `/api/` path are allowed; everything else is
76
+ * rejected, with a named reason for known-bad local/inline schemes so the
77
+ * thrown message is actionable for an LLM planner. The url is trimmed before
78
+ * the scheme check so leading whitespace cannot smuggle a rejected scheme past
79
+ * a naive `startsWith`.
80
+ *
81
+ * @param what - noun for the error message (e.g. 'media url', 'src').
82
+ */
83
+ export declare function assertMediaUrl(url: string, what?: string): void;
84
+ export { isWorkspaceFileExportable } from './file-export';
@@ -1,84 +1,3 @@
1
- /**
2
- * Web-boundary utilities every agent app's routes hand-roll: JSON body parsing
3
- * + narrowing, request-context extraction (real client IP behind Cloudflare),
4
- * a KV-backed sliding-window rate limiter, the free-route budget policy built
5
- * on it, and security response headers. Pure mechanism — no DB, no domain. The
6
- * KV is a structural interface so this needs no `@cloudflare/workers-types`
7
- * dependency.
8
- */
9
- export * from './rate-limit';
10
- export * from './free-route-limit';
11
- export type JsonObject = Record<string, unknown>;
12
- /** Parse + object-narrow a Request body. `[body, null]` on success, `[null,
13
- * errorResponse]` on a non-object body (callers `if (err) return err`). */
14
- export declare function parseJsonObjectBody(request: Request): Promise<[JsonObject, null] | [null, Response]>;
15
- /** Narrow one required string field, 400 if missing/empty. */
16
- export declare function requireString(body: JsonObject, field: string): string | Response;
17
- /** Define the context of a request including IP address, user agent, timestamp, and request ID */
18
- export interface RequestContext {
19
- ipAddress: string;
20
- userAgent: string;
21
- timestamp: string;
22
- requestId: string;
23
- }
24
- /** Extract request context for audit trails. Uses `CF-Connecting-IP` for the
25
- * real client IP behind Cloudflare. */
26
- export declare function extractRequestContext(request: Request): RequestContext;
27
- /** Define options for configuring cookie attributes and behavior */
28
- export interface CookieOptions {
29
- name: string;
30
- /** Default '/'. */
31
- path?: string;
32
- /** Default true. */
33
- httpOnly?: boolean;
34
- /** Adds the `Secure` attribute. Default false. */
35
- secure?: boolean;
36
- /** Default 'Lax'. */
37
- sameSite?: 'Lax' | 'Strict' | 'None';
38
- maxAgeSeconds?: number;
39
- }
40
- /** Serialize a Set-Cookie header value: `name=encodeURIComponent(value)` plus
41
- * attributes in Path / HttpOnly / SameSite / Max-Age / Secure order.
42
- * Throws on `SameSite=None` without `secure` — browsers silently drop that
43
- * combination, which would otherwise fail invisibly. */
44
- export declare function serializeCookie(value: string, opts: CookieOptions): string;
45
- /** Set-Cookie header value that deletes the cookie (empty value, Max-Age=0). */
46
- export declare function clearCookieHeader(opts: Omit<CookieOptions, 'maxAgeSeconds'>): string;
47
- /** Read + decode one cookie from a Cookie request header; null when absent. */
48
- export declare function readCookieValue(cookieHeader: string | null, name: string): string | null;
49
- /** Define options for configuring security-related HTTP headers including disclaimers and retention labels */
50
- export interface SecurityHeaderOptions {
51
- /** Product disclaimer (e.g. "AI-powered tool. Not legal advice."). Omitted if absent. */
52
- disclaimer?: string;
53
- /** Data-retention label (e.g. "7-years"). Omitted if absent. */
54
- retention?: string;
55
- /** Extra headers to set. */
56
- extra?: Record<string, string>;
57
- }
58
- /** Canonical generic response headers used by {@link addSecurityHeaders}.
59
- * Exported so static-asset hosts can apply the same policy without copying
60
- * values that silently drift from Worker/API responses. */
61
- export declare const STANDARD_SECURITY_HEADERS: Readonly<{
62
- readonly 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload';
63
- readonly 'X-Content-Type-Options': 'nosniff';
64
- readonly 'X-Frame-Options': 'SAMEORIGIN';
65
- readonly 'Referrer-Policy': 'same-origin';
66
- readonly 'X-XSS-Protection': '1; mode=block';
67
- }>;
68
- /** Set standard security headers on a response (HSTS, nosniff, frame-options,
69
- * referrer-policy, XSS) + optional product disclaimer/retention. The security
70
- * set is generic; the disclaimer/retention are the product's. */
71
- export declare function addSecurityHeaders(response: Response, opts?: SecurityHeaderOptions): Response;
72
- /**
73
- * Canonical media-reference boundary shared by every surface that persists a
74
- * media url (sequences clips, design-canvas image/video src). The ONE rule:
75
- * remote `http(s)` or a rooted `/api/` path are allowed; everything else is
76
- * rejected, with a named reason for known-bad local/inline schemes so the
77
- * thrown message is actionable for an LLM planner. The url is trimmed before
78
- * the scheme check so leading whitespace cannot smuggle a rejected scheme past
79
- * a naive `startsWith`.
80
- *
81
- * @param what - noun for the error message (e.g. 'media url', 'src').
82
- */
83
- export declare function assertMediaUrl(url: string, what?: string): void;
84
- export { isWorkspaceFileExportable } from './file-export';
1
+ /** Browser-safe application-boundary helpers. No React or execution-engine peers. */
2
+ export * from './core';
3
+ export * from './message-groups';
package/dist/web/index.js CHANGED
@@ -10,12 +10,13 @@ import {
10
10
  clearCookieHeader,
11
11
  extractRequestContext,
12
12
  freeRouteLimitResponse,
13
+ groupConversationMessages,
13
14
  parseJsonObjectBody,
14
15
  readCookieValue,
15
16
  requireString,
16
17
  serializeCookie,
17
18
  withFreeRouteLimit
18
- } from "../chunk-3YUI7WNW.js";
19
+ } from "../chunk-IAY2DMKW.js";
19
20
  import {
20
21
  isWorkspaceFileExportable
21
22
  } from "../chunk-TXD5HXLE.js";
@@ -31,6 +32,7 @@ export {
31
32
  clearCookieHeader,
32
33
  extractRequestContext,
33
34
  freeRouteLimitResponse,
35
+ groupConversationMessages,
34
36
  isWorkspaceFileExportable,
35
37
  parseJsonObjectBody,
36
38
  readCookieValue,
@@ -0,0 +1,21 @@
1
+ /** Renderer-neutral attribution. Not a transcript merger or an execution state machine. */
2
+ export interface ConversationGroupItem {
3
+ id?: string | number;
4
+ kind?: string;
5
+ role?: string;
6
+ /** A different assistant/persona must never inherit the previous speaker's label. */
7
+ speakerId?: string;
8
+ /** Distinct threads must not be grouped if their rows share a viewport. */
9
+ conversationId?: string;
10
+ }
11
+ export type GroupedConversationItem<T> = T & {
12
+ isContinuation?: boolean;
13
+ groupId?: string;
14
+ };
15
+ /**
16
+ * Annotate assistant rows until an actual user message, speaker or conversation change.
17
+ * Tool/progress notices do not interrupt the group. The first visible assistant is
18
+ * always labeled, including when a renderer has paged earlier history away.
19
+ * Pass display order; stored content, IDs, timestamps and input objects are untouched.
20
+ */
21
+ export declare function groupConversationMessages<T extends ConversationGroupItem>(items?: readonly T[]): GroupedConversationItem<T>[];