@tangle-network/agent-eval 0.133.2 → 0.134.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +218 -0
- package/dist/analyst/index.d.ts +11 -35
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +4 -53
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-BmX-h_yn.d.ts → analyze-runs-DMo3Lb_y.d.ts} +4 -4
- package/dist/{analyze-runs-BmX-h_yn.d.ts.map → analyze-runs-DMo3Lb_y.d.ts.map} +1 -1
- package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
- package/dist/analyze-runs-qk8op0tN.js.map +1 -0
- package/dist/baseline-BaPxoROc.js +149 -0
- package/dist/baseline-BaPxoROc.js.map +1 -0
- package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
- package/dist/baseline-D_fT6277.d.ts.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CJr1H1_a.js → benchmarks-v5piCeDl.js} +3 -3
- package/dist/{benchmarks-CJr1H1_a.js.map → benchmarks-v5piCeDl.js.map} +1 -1
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +5 -4
- package/dist/campaign/index.js +4 -3
- package/dist/{campaign-BJjn1rhw.js → campaign-DEC_7DLn.js} +12 -6
- package/dist/{campaign-BJjn1rhw.js.map → campaign-DEC_7DLn.js.map} +1 -1
- package/dist/{client-COvaLoQG.d.ts → client-BIyh1RCr.d.ts} +29 -15
- package/dist/client-BIyh1RCr.d.ts.map +1 -0
- package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
- package/dist/client-LIuo-KPv.js.map +1 -0
- package/dist/contract/index.d.ts +12 -11
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +12 -18
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-Cl3pHo4n.d.ts → default-registry-Brxr728w.d.ts} +4 -268
- package/dist/default-registry-Brxr728w.d.ts.map +1 -0
- package/dist/{default-registry-D3T9XbuY.js → default-registry-IjYs7T8l.js} +4 -61
- package/dist/default-registry-IjYs7T8l.js.map +1 -0
- package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
- package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-C7Wue8R6.d.ts → index-BoJNQR6n.d.ts} +29 -11
- package/dist/index-BoJNQR6n.d.ts.map +1 -0
- package/dist/{index-BREtv3ZZ.d.ts → index-C21xKtxu.d.ts} +4 -4
- package/dist/{index-BREtv3ZZ.d.ts.map → index-C21xKtxu.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
- package/dist/index-DSC51roc2.d.ts.map +1 -0
- package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
- package/dist/index-DuhJaaiH.d.ts.map +1 -0
- package/dist/index.d.ts +60 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +134 -27
- package/dist/index.js.map +1 -1
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
- package/dist/ledger-core-DAKFKRzi.js.map +1 -0
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
- package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
- package/dist/pipelines/index.d.ts +1 -1
- package/dist/pipelines/index.js +3 -2
- package/dist/pipelines/index.js.map +1 -1
- package/dist/proposal-findings-DCawte-y.js +164 -0
- package/dist/proposal-findings-DCawte-y.js.map +1 -0
- package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
- package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
- package/dist/{release-report-CjHWa8Ia.d.ts → release-report-CuULWKyk.d.ts} +2 -2
- package/dist/{release-report-CjHWa8Ia.d.ts.map → release-report-CuULWKyk.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +4 -4
- package/dist/{researcher-CbSKhK8z.d.ts → researcher-DVtruQ9U.d.ts} +2 -2
- package/dist/{researcher-CbSKhK8z.d.ts.map → researcher-DVtruQ9U.d.ts.map} +1 -1
- package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
- package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
- package/dist/rl.d.ts +2 -2
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +18 -5
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
- package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
- package/dist/{semantic-concept-judge-BypLt6Fw.js → semantic-concept-judge-C0P1VTXD.js} +2 -3
- package/dist/{semantic-concept-judge-BypLt6Fw.js.map → semantic-concept-judge-C0P1VTXD.js.map} +1 -1
- package/dist/{skill-usage-BaaxFSJR.d.ts → skill-usage-BDQVPIG1.d.ts} +3 -2
- package/dist/skill-usage-BDQVPIG1.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-BY6vKLJB.js} +169 -53
- package/dist/skillopt-optimization-method-BY6vKLJB.js.map +1 -0
- package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts → skillopt-optimization-method-DJ3l4w8W.d.ts} +20 -18
- package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +1 -0
- package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
- package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
- package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
- package/dist/statistics-RwRNu2__.js.map +1 -0
- package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
- package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
- package/dist/{summary-report-CFnQgNfg.d.ts → summary-report-DGp0-_XO.d.ts} +61 -4
- package/dist/summary-report-DGp0-_XO.d.ts.map +1 -0
- package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
- package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
- package/dist/types-DVjczBM9.d.ts +276 -0
- package/dist/types-DVjczBM9.d.ts.map +1 -0
- package/dist/{types-BokuXvOG.d.ts → types-DiWLru6Z.d.ts} +20 -37
- package/dist/types-DiWLru6Z.d.ts.map +1 -0
- package/docs/campaign-proposers.md +5 -0
- package/docs/design/statistics-decisions.md +271 -0
- package/docs/design.md +1 -0
- package/docs/insight-report.md +1 -1
- package/docs/research-report-methodology.md +4 -1
- package/package.json +2 -1
- package/dist/analyze-runs-B-afTpCv.js.map +0 -1
- package/dist/baseline-DcX5hQDv.js.map +0 -1
- package/dist/baseline-hG3K85h4.d.ts.map +0 -1
- package/dist/client-COvaLoQG.d.ts.map +0 -1
- package/dist/client-CYzbdJOZ.js.map +0 -1
- package/dist/default-registry-Cl3pHo4n.d.ts.map +0 -1
- package/dist/default-registry-D3T9XbuY.js.map +0 -1
- package/dist/index-C7Wue8R6.d.ts.map +0 -1
- package/dist/index-DSC51roc.d.ts.map +0 -1
- package/dist/index-nhIYz9hn.d.ts.map +0 -1
- package/dist/ledger-core-CPZfcrC2.js.map +0 -1
- package/dist/run-score-iEEAWiBY.js +0 -41
- package/dist/run-score-iEEAWiBY.js.map +0 -1
- package/dist/skill-usage-BaaxFSJR.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CF6a327Q.js.map +0 -1
- package/dist/skillopt-optimization-method-wHF5xsUv.d.ts.map +0 -1
- package/dist/statistics-DWM_AyLe.js.map +0 -1
- package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
- package/dist/summary-report-CFnQgNfg.d.ts.map +0 -1
- package/dist/types-BokuXvOG.d.ts.map +0 -1
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"tool-use-metrics-DEGMKycK.js","names":[],"sources":["../src/failure-taxonomy.ts","../src/tool-use-metrics.ts"],"sourcesContent":["/**\n * Failure taxonomy — canonical classes + a default classifier.\n *\n * Every failed run should end up in a named class. The classifier here\n * is rule-based (fast, deterministic); an LLM fallback can be added by\n * the consumer for novel cases and trained into the rule base over time.\n *\n * Consumers call `classifyFailure(run, spans, events)` and persist the\n * returned class as `Run.outcome.failureClass`.\n */\n\nimport type { FailureClass, Run, Span, TraceEvent } from './trace/schema'\nimport { FAILURE_CLASSES } from './trace/schema'\n\nexport { FAILURE_CLASSES, type FailureClass }\n\nexport interface FailureContext {\n run: Run\n spans: Span[]\n events: TraceEvent[]\n}\n\nexport interface FailureClassification {\n failureClass: FailureClass\n reason: string\n triggerSpanId?: string\n triggerEventId?: string\n}\n\n/** Ordered rules — first match wins. */\nexport interface FailureRule {\n id: string\n match: (ctx: FailureContext) => {\n failureClass: FailureClass\n reason: string\n triggerSpanId?: string\n triggerEventId?: string\n } | null\n}\n\nexport const DEFAULT_RULES: FailureRule[] = [\n // Outcome already named? Respect it.\n {\n id: 'explicit-outcome',\n match: ({ run }) => {\n const fc = run.outcome?.failureClass\n if (fc && fc !== 'unknown')\n return { failureClass: fc, reason: 'outcome.failureClass set explicitly' }\n return null\n },\n },\n {\n id: 'knowledge-readiness-blocked',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'readiness_scored' &&\n e.payload.passed === false,\n )\n return event\n ? {\n failureClass: 'knowledge_readiness_blocked',\n reason: 'knowledge readiness report blocked execution',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'bad-integration-manifest',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_manifest_validated' && e.payload.valid === false) ||\n (e.payload.kind === 'integration_invoke_failed' &&\n e.payload.code === 'manifest_invalid')),\n )\n return event\n ? {\n failureClass: 'bad_integration_manifest',\n reason: 'integration manifest validation failed before launch',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-integration-connection',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_manifest_resolved' &&\n hasResolutionStatus(e.payload, 'missing_connection'),\n )\n return event\n ? {\n failureClass: 'missing_integration_connection',\n reason: 'required integration connection was missing',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-integration-scope',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_manifest_resolved' && hasMissingScopes(e.payload)) ||\n (e.payload.kind === 'integration_invoke_failed' && e.payload.code === 'scope_denied')),\n )\n return event\n ? {\n failureClass: 'missing_integration_scope',\n reason: 'integration grant or connection lacks required scopes',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-approval-required',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_invoke' && e.payload.status === 'approval_required') ||\n (e.payload.kind === 'integration_invoke_failed' &&\n e.payload.code === 'approval_required') ||\n e.payload.kind === 'integration_approval_required'),\n )\n return event\n ? {\n failureClass: 'integration_approval_required',\n reason: 'integration write paused for user approval',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-auth-expired',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n (e.payload.code === 'auth_expired' ||\n e.payload.code === 'connection_not_active' ||\n e.payload.code === 'capability_expired' ||\n e.payload.status === 'expired'),\n )\n return event\n ? {\n failureClass: 'integration_auth_expired',\n reason: 'integration connection or capability expired',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'unsafe-integration-write-denied',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n (e.payload.code === 'unsafe_write_denied' ||\n e.payload.code === 'policy_denied' ||\n e.payload.code === 'action_denied'),\n )\n return event\n ? {\n failureClass: 'unsafe_integration_write_denied',\n reason: 'integration write was denied by policy or capability scope',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-provider-failure',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n ![\n 'scope_denied',\n 'approval_required',\n 'auth_expired',\n 'connection_not_active',\n 'capability_expired',\n 'unsafe_write_denied',\n 'policy_denied',\n 'action_denied',\n 'manifest_invalid',\n ].includes(String(e.payload.code)),\n )\n return event\n ? {\n failureClass: 'integration_provider_failure',\n reason: 'integration provider invocation failed',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-credentials',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.category === 'credential_or_secret',\n )\n return event\n ? {\n failureClass: 'missing_credentials',\n reason: 'required credential or secret was missing',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'bad-retrieval',\n match: ({ run, spans }) => {\n if (run.outcome?.pass !== false) return null\n const retrieval = spans.find(\n (s) =>\n s.kind === 'retrieval' && (s.hits.length === 0 || s.hits.every((hit) => hit.score <= 0)),\n )\n return retrieval\n ? {\n failureClass: 'bad_retrieval',\n reason: 'retrieval returned no useful hits for a failed run',\n triggerSpanId: retrieval.spanId,\n }\n : null\n },\n },\n {\n id: 'insufficient-evidence',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.reason === 'insufficient_evidence',\n )\n return event\n ? {\n failureClass: 'insufficient_evidence',\n reason: 'task proceeded with insufficient supporting evidence',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'contradictory-evidence',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.reason === 'contradictory_evidence',\n )\n return event\n ? {\n failureClass: 'contradictory_evidence',\n reason: 'supporting evidence contradicted itself',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n // Budget breach events\n {\n id: 'budget-breach',\n match: ({ events }) => {\n const breach = events.find((e) => e.kind === 'budget_breach')\n return breach\n ? {\n failureClass: 'budget_exceeded',\n reason: `budget breached on ${breach.payload.dimension ?? 'unknown dimension'}`,\n triggerEventId: breach.eventId,\n }\n : null\n },\n },\n // Policy violations\n {\n id: 'policy-violation',\n match: ({ events }) => {\n const e = events.find((x) => x.kind === 'policy_violation')\n return e\n ? {\n failureClass: 'policy_violation',\n reason: 'policy_violation event emitted',\n triggerEventId: e.eventId,\n }\n : null\n },\n },\n // Sandbox non-zero exit code\n {\n id: 'sandbox-failure',\n match: ({ spans }) => {\n const s = spans.find(\n (x) => x.kind === 'sandbox' && typeof x.exitCode === 'number' && x.exitCode !== 0,\n )\n if (!s) return null\n return {\n failureClass: 'sandbox_failure',\n reason: `sandbox exited ${(s as Extract<Span, { kind: 'sandbox' }>).exitCode}`,\n triggerSpanId: s.spanId,\n }\n },\n },\n // Timeout: run aborted by external signal\n {\n id: 'timeout',\n match: ({ run, events }) => {\n if (run.status !== 'aborted') return null\n const hasTimeout = events.some(\n (e) =>\n e.kind === 'error' &&\n String(e.payload.reason ?? '')\n .toLowerCase()\n .includes('timeout'),\n )\n const note = (run.outcome?.notes ?? '').toLowerCase()\n if (hasTimeout || note.includes('timeout') || note.includes('deadline')) {\n return { failureClass: 'timeout', reason: 'timeout signal observed' }\n }\n return null\n },\n },\n // Tool recovery failure: many consecutive tool errors on the same tool\n {\n id: 'tool-recovery-failure',\n match: ({ spans }) => {\n const tools = spans.filter((s) => s.kind === 'tool')\n const byTool = new Map<string, Span[]>()\n for (const t of tools) {\n const name = (t as Extract<Span, { kind: 'tool' }>).toolName\n const arr = byTool.get(name) ?? []\n arr.push(t)\n byTool.set(name, arr)\n }\n for (const [name, arr] of byTool) {\n const errs = arr.filter((s) => s.status === 'error')\n if (errs.length >= 3 && errs.length === arr.length) {\n return {\n failureClass: 'tool_recovery_failure',\n reason: `${errs.length} consecutive errors on tool \"${name}\"`,\n triggerSpanId: errs[errs.length - 1]!.spanId,\n }\n }\n }\n return null\n },\n },\n // Tool selection error: the run failed and agent called zero tools despite having them\n {\n id: 'tool-selection-error',\n match: ({ run, spans }) => {\n if (run.outcome?.pass !== false) return null\n const hasToolsAvailable = spans.some(\n (s) =>\n s.kind === 'agent' &&\n (s.attributes?.toolsAvailable as number | undefined) !== undefined &&\n (s.attributes?.toolsAvailable as number) > 0,\n )\n const tools = spans.filter((s) => s.kind === 'tool')\n if (hasToolsAvailable && tools.length === 0) {\n return {\n failureClass: 'tool_selection_error',\n reason: 'tools were available but none were called',\n }\n }\n return null\n },\n },\n // Format drift: scored by a judge with dimension='format' below threshold\n {\n id: 'format-drift',\n match: ({ spans }) => {\n const judge = spans.find(\n (s) =>\n s.kind === 'judge' &&\n (s as Extract<Span, { kind: 'judge' }>).dimension === 'format' &&\n (s as Extract<Span, { kind: 'judge' }>).score < 0.5,\n )\n return judge\n ? {\n failureClass: 'format_drift',\n reason: 'format judge scored below 0.5',\n triggerSpanId: judge.spanId,\n }\n : null\n },\n },\n]\n\nfunction hasResolutionStatus(payload: Record<string, unknown>, status: string): boolean {\n if (status === 'missing_connection' && stringArray(payload.missingConnections).length > 0)\n return true\n return resolutionItems(payload).some((item) => item.status === status)\n}\n\nfunction hasMissingScopes(payload: Record<string, unknown>): boolean {\n if (stringArray(payload.missingScopes).length > 0) return true\n return resolutionItems(payload).some(\n (item) => Array.isArray(item.missingScopes) && item.missingScopes.length > 0,\n )\n}\n\nfunction resolutionItems(payload: Record<string, unknown>): Array<Record<string, unknown>> {\n return [\n ...records(payload.missing),\n ...records(payload.optionalMissing),\n ...records(payload.ready),\n ]\n}\n\nfunction records(value: unknown): Array<Record<string, unknown>> {\n if (!Array.isArray(value)) return []\n return value.filter(\n (item): item is Record<string, unknown> =>\n Boolean(item) && typeof item === 'object' && !Array.isArray(item),\n )\n}\n\nfunction stringArray(value: unknown): string[] {\n return Array.isArray(value)\n ? value.filter((item): item is string => typeof item === 'string')\n : []\n}\n\n/** Classify the failure mode of a run using an ordered rule list. */\nexport function classifyFailure(\n ctx: FailureContext,\n rules: FailureRule[] = DEFAULT_RULES,\n): FailureClassification {\n if (ctx.run.outcome?.pass !== false && ctx.run.status === 'completed') {\n return { failureClass: 'success', reason: 'run completed with pass=true (or no explicit fail)' }\n }\n for (const rule of rules) {\n const hit = rule.match(ctx)\n if (hit) return hit\n }\n return { failureClass: 'unknown', reason: 'no rule matched; run failed for unclassified reason' }\n}\n","/**\n * Tool-use metrics — derived purely from trace data.\n *\n * No scoring assumptions: consumers supply optional ground-truth tool\n * selections per turn + optional \"information used downstream\" signals.\n * Without those, we still compute descriptive metrics (error rate,\n * retry rate, duplicate-call rate) that are useful on their own.\n */\n\nimport { argHash, groupBy, hasCapturedToolArgs, toolSpans } from './trace/query'\nimport type { Span } from './trace/schema'\nimport type { TraceStore } from './trace/store'\n\nexport interface ToolUseMetrics {\n runId: string\n totalCalls: number\n /** Calls whose arguments were captured and can be compared for duplication. */\n callsWithCapturedArgs: number\n byTool: Record<string, ToolStats>\n errorRate: number\n /** Ratio of captured-argument calls already seen with the same tool name and arguments. */\n duplicateRate: number\n /** Ratio of error calls followed by ≥1 retry on same tool. */\n retryRate: number\n /** Optional: of the calls agent made, fraction the evaluator marked as \"correct selection\". */\n selectionAccuracy?: number\n}\n\nexport interface ToolStats {\n calls: number\n callsWithCapturedArgs: number\n errors: number\n avgLatencyMs: number\n duplicates: number\n}\n\nexport interface ToolUseOptions {\n /** Map of spanId → whether the evaluator judged the tool selection correct. Optional. */\n selectionLabels?: Record<string, boolean>\n}\n\nexport async function computeToolUseMetrics(\n store: TraceStore,\n runId: string,\n options: ToolUseOptions = {},\n): Promise<ToolUseMetrics> {\n const tools = await toolSpans(store, runId)\n if (tools.length === 0) {\n return {\n runId,\n totalCalls: 0,\n callsWithCapturedArgs: 0,\n byTool: {},\n errorRate: 0,\n duplicateRate: 0,\n retryRate: 0,\n }\n }\n\n const byTool: Record<string, ToolStats> = {}\n let totalErrors = 0\n let totalDuplicates = 0\n let callsWithCapturedArgs = 0\n const sortedTools = [...tools].sort((a, b) => a.startedAt - b.startedAt)\n const seenSignatures = new Set<string>()\n\n // duplicate detection + per-tool aggregation\n for (const t of sortedTools) {\n byTool[t.toolName] ??= {\n calls: 0,\n callsWithCapturedArgs: 0,\n errors: 0,\n avgLatencyMs: 0,\n duplicates: 0,\n }\n const stat = byTool[t.toolName]!\n stat.calls += 1\n if (t.status === 'error') {\n stat.errors += 1\n totalErrors += 1\n }\n if (typeof t.latencyMs === 'number') stat.avgLatencyMs += t.latencyMs\n if (hasCapturedToolArgs(t)) {\n callsWithCapturedArgs += 1\n stat.callsWithCapturedArgs += 1\n const sig = `${t.toolName}|${argHash(t.args)}`\n if (seenSignatures.has(sig)) {\n stat.duplicates += 1\n totalDuplicates += 1\n }\n seenSignatures.add(sig)\n }\n }\n\n for (const stat of Object.values(byTool)) {\n stat.avgLatencyMs = stat.calls > 0 ? stat.avgLatencyMs / stat.calls : 0\n }\n\n // retry detection: per-tool chronological adjacency where error → next same-tool call\n let retryOpportunities = 0\n let retriesFollowed = 0\n for (const [, arr] of groupBy(sortedTools, (t) => t.toolName)) {\n for (let i = 0; i < arr.length; i++) {\n if (arr[i]!.status !== 'error') continue\n retryOpportunities += 1\n if (arr[i + 1]) retriesFollowed += 1\n }\n }\n const retryRate = retryOpportunities > 0 ? retriesFollowed / retryOpportunities : 0\n\n let selectionAccuracy: number | undefined\n if (options.selectionLabels) {\n const labeled = sortedTools.filter((t) => t.spanId in options.selectionLabels!)\n if (labeled.length > 0) {\n selectionAccuracy =\n labeled.filter((t) => options.selectionLabels![t.spanId]).length / labeled.length\n }\n }\n\n return {\n runId,\n totalCalls: sortedTools.length,\n callsWithCapturedArgs,\n byTool,\n errorRate: totalErrors / sortedTools.length,\n duplicateRate: callsWithCapturedArgs > 0 ? totalDuplicates / callsWithCapturedArgs : 0,\n retryRate,\n selectionAccuracy,\n }\n}\n\nexport type { Span }\n"],"mappings":";;AAwCA,MAAa,gBAA+B;CAE1C;EACE,IAAI;EACJ,QAAQ,EAAE,UAAU;GAClB,MAAM,KAAK,IAAI,SAAS;GACxB,IAAI,MAAM,OAAO,WACf,OAAO;IAAE,cAAc;IAAI,QAAQ;GAAsC;GAC3E,OAAO;EACT;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,sBACnB,EAAE,QAAQ,WAAW,KACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,oCAAoC,EAAE,QAAQ,UAAU,SAC1E,EAAE,QAAQ,SAAS,+BAClB,EAAE,QAAQ,SAAS,mBAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mCACnB,oBAAoB,EAAE,SAAS,oBAAoB,CACvD;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,mCAAmC,iBAAiB,EAAE,OAAO,KAC/E,EAAE,QAAQ,SAAS,+BAA+B,EAAE,QAAQ,SAAS,eAC5E;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,wBAAwB,EAAE,QAAQ,WAAW,uBAC/D,EAAE,QAAQ,SAAS,+BAClB,EAAE,QAAQ,SAAS,uBACrB,EAAE,QAAQ,SAAS,gCACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,gCAClB,EAAE,QAAQ,SAAS,kBAClB,EAAE,QAAQ,SAAS,2BACnB,EAAE,QAAQ,SAAS,wBACnB,EAAE,QAAQ,WAAW,UAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,gCAClB,EAAE,QAAQ,SAAS,yBAClB,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,SAAS,gBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,+BACnB,CAAC;IACC;IACA;IACA;IACA;IACA;IACA;IACA;IACA;IACA;GACF,CAAC,CAAC,SAAS,OAAO,EAAE,QAAQ,IAAI,CAAC,CACrC;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,aAAa,sBAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,YAAY;GACzB,IAAI,IAAI,SAAS,SAAS,OAAO,OAAO;GACxC,MAAM,YAAY,MAAM,MACrB,MACC,EAAE,SAAS,gBAAgB,EAAE,KAAK,WAAW,KAAK,EAAE,KAAK,OAAO,QAAQ,IAAI,SAAS,CAAC,EAC1F;GACA,OAAO,YACH;IACE,cAAc;IACd,QAAQ;IACR,eAAe,UAAU;GAC3B,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,WAAW,uBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,WAAW,wBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,SAAS,OAAO,MAAM,MAAM,EAAE,SAAS,eAAe;GAC5D,OAAO,SACH;IACE,cAAc;IACd,QAAQ,sBAAsB,OAAO,QAAQ,aAAa;IAC1D,gBAAgB,OAAO;GACzB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,IAAI,OAAO,MAAM,MAAM,EAAE,SAAS,kBAAkB;GAC1D,OAAO,IACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,EAAE;GACpB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,IAAI,MAAM,MACb,MAAM,EAAE,SAAS,aAAa,OAAO,EAAE,aAAa,YAAY,EAAE,aAAa,CAClF;GACA,IAAI,CAAC,GAAG,OAAO;GACf,OAAO;IACL,cAAc;IACd,QAAQ,kBAAmB,EAAyC;IACpE,eAAe,EAAE;GACnB;EACF;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,aAAa;GAC1B,IAAI,IAAI,WAAW,WAAW,OAAO;GACrC,MAAM,aAAa,OAAO,MACvB,MACC,EAAE,SAAS,WACX,OAAO,EAAE,QAAQ,UAAU,EAAE,CAAC,CAC3B,YAAY,CAAC,CACb,SAAS,SAAS,CACzB;GACA,MAAM,QAAQ,IAAI,SAAS,SAAS,GAAA,CAAI,YAAY;GACpD,IAAI,cAAc,KAAK,SAAS,SAAS,KAAK,KAAK,SAAS,UAAU,GACpE,OAAO;IAAE,cAAc;IAAW,QAAQ;GAA0B;GAEtE,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,SAAS,MAAM;GACnD,MAAM,yBAAS,IAAI,IAAoB;GACvC,KAAK,MAAM,KAAK,OAAO;IACrB,MAAM,OAAQ,EAAsC;IACpD,MAAM,MAAM,OAAO,IAAI,IAAI,KAAK,CAAC;IACjC,IAAI,KAAK,CAAC;IACV,OAAO,IAAI,MAAM,GAAG;GACtB;GACA,KAAK,MAAM,CAAC,MAAM,QAAQ,QAAQ;IAChC,MAAM,OAAO,IAAI,QAAQ,MAAM,EAAE,WAAW,OAAO;IACnD,IAAI,KAAK,UAAU,KAAK,KAAK,WAAW,IAAI,QAC1C,OAAO;KACL,cAAc;KACd,QAAQ,GAAG,KAAK,OAAO,+BAA+B,KAAK;KAC3D,eAAe,KAAK,KAAK,SAAS,EAAE,CAAE;IACxC;GAEJ;GACA,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,YAAY;GACzB,IAAI,IAAI,SAAS,SAAS,OAAO,OAAO;GACxC,MAAM,oBAAoB,MAAM,MAC7B,MACC,EAAE,SAAS,WACV,EAAE,YAAY,mBAA0C,KAAA,KACxD,EAAE,YAAY,iBAA4B,CAC/C;GACA,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,SAAS,MAAM;GACnD,IAAI,qBAAqB,MAAM,WAAW,GACxC,OAAO;IACL,cAAc;IACd,QAAQ;GACV;GAEF,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,QAAQ,MAAM,MACjB,MACC,EAAE,SAAS,WACV,EAAuC,cAAc,YACrD,EAAuC,QAAQ,EACpD;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,eAAe,MAAM;GACvB,IACA;EACN;CACF;AACF;AAEA,SAAS,oBAAoB,SAAkC,QAAyB;CACtF,IAAI,WAAW,wBAAwB,YAAY,QAAQ,kBAAkB,CAAC,CAAC,SAAS,GACtF,OAAO;CACT,OAAO,gBAAgB,OAAO,CAAC,CAAC,MAAM,SAAS,KAAK,WAAW,MAAM;AACvE;AAEA,SAAS,iBAAiB,SAA2C;CACnE,IAAI,YAAY,QAAQ,aAAa,CAAC,CAAC,SAAS,GAAG,OAAO;CAC1D,OAAO,gBAAgB,OAAO,CAAC,CAAC,MAC7B,SAAS,MAAM,QAAQ,KAAK,aAAa,KAAK,KAAK,cAAc,SAAS,CAC7E;AACF;AAEA,SAAS,gBAAgB,SAAkE;CACzF,OAAO;EACL,GAAG,QAAQ,QAAQ,OAAO;EAC1B,GAAG,QAAQ,QAAQ,eAAe;EAClC,GAAG,QAAQ,QAAQ,KAAK;CAC1B;AACF;AAEA,SAAS,QAAQ,OAAgD;CAC/D,IAAI,CAAC,MAAM,QAAQ,KAAK,GAAG,OAAO,CAAC;CACnC,OAAO,MAAM,QACV,SACC,QAAQ,IAAI,KAAK,OAAO,SAAS,YAAY,CAAC,MAAM,QAAQ,IAAI,CACpE;AACF;AAEA,SAAS,YAAY,OAA0B;CAC7C,OAAO,MAAM,QAAQ,KAAK,IACtB,MAAM,QAAQ,SAAyB,OAAO,SAAS,QAAQ,IAC/D,CAAC;AACP;;AAGA,SAAgB,gBACd,KACA,QAAuB,eACA;CACvB,IAAI,IAAI,IAAI,SAAS,SAAS,SAAS,IAAI,IAAI,WAAW,aACxD,OAAO;EAAE,cAAc;EAAW,QAAQ;CAAqD;CAEjG,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,MAAM,KAAK,MAAM,GAAG;EAC1B,IAAI,KAAK,OAAO;CAClB;CACA,OAAO;EAAE,cAAc;EAAW,QAAQ;CAAsD;AAClG;;;;;;;;;;;ACpaA,eAAsB,sBACpB,OACA,OACA,UAA0B,CAAC,GACF;CACzB,MAAM,QAAQ,MAAM,UAAU,OAAO,KAAK;CAC1C,IAAI,MAAM,WAAW,GACnB,OAAO;EACL;EACA,YAAY;EACZ,uBAAuB;EACvB,QAAQ,CAAC;EACT,WAAW;EACX,eAAe;EACf,WAAW;CACb;CAGF,MAAM,SAAoC,CAAC;CAC3C,IAAI,cAAc;CAClB,IAAI,kBAAkB;CACtB,IAAI,wBAAwB;CAC5B,MAAM,cAAc,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;CACvE,MAAM,iCAAiB,IAAI,IAAY;CAGvC,KAAK,MAAM,KAAK,aAAa;EAC3B,OAAO,EAAE,cAAc;GACrB,OAAO;GACP,uBAAuB;GACvB,QAAQ;GACR,cAAc;GACd,YAAY;EACd;EACA,MAAM,OAAO,OAAO,EAAE;EACtB,KAAK,SAAS;EACd,IAAI,EAAE,WAAW,SAAS;GACxB,KAAK,UAAU;GACf,eAAe;EACjB;EACA,IAAI,OAAO,EAAE,cAAc,UAAU,KAAK,gBAAgB,EAAE;EAC5D,IAAI,oBAAoB,CAAC,GAAG;GAC1B,yBAAyB;GACzB,KAAK,yBAAyB;GAC9B,MAAM,MAAM,GAAG,EAAE,SAAS,GAAG,QAAQ,EAAE,IAAI;GAC3C,IAAI,eAAe,IAAI,GAAG,GAAG;IAC3B,KAAK,cAAc;IACnB,mBAAmB;GACrB;GACA,eAAe,IAAI,GAAG;EACxB;CACF;CAEA,KAAK,MAAM,QAAQ,OAAO,OAAO,MAAM,GACrC,KAAK,eAAe,KAAK,QAAQ,IAAI,KAAK,eAAe,KAAK,QAAQ;CAIxE,IAAI,qBAAqB;CACzB,IAAI,kBAAkB;CACtB,KAAK,MAAM,GAAG,QAAQ,QAAQ,cAAc,MAAM,EAAE,QAAQ,GAC1D,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAAK;EACnC,IAAI,IAAI,EAAE,CAAE,WAAW,SAAS;EAChC,sBAAsB;EACtB,IAAI,IAAI,IAAI,IAAI,mBAAmB;CACrC;CAEF,MAAM,YAAY,qBAAqB,IAAI,kBAAkB,qBAAqB;CAElF,IAAI;CACJ,IAAI,QAAQ,iBAAiB;EAC3B,MAAM,UAAU,YAAY,QAAQ,MAAM,EAAE,UAAU,QAAQ,eAAgB;EAC9E,IAAI,QAAQ,SAAS,GACnB,oBACE,QAAQ,QAAQ,MAAM,QAAQ,gBAAiB,EAAE,OAAO,CAAC,CAAC,SAAS,QAAQ;CAEjF;CAEA,OAAO;EACL;EACA,YAAY,YAAY;EACxB;EACA;EACA,WAAW,cAAc,YAAY;EACrC,eAAe,wBAAwB,IAAI,kBAAkB,wBAAwB;EACrF;EACA;CACF;AACF"}
|
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
import { c as CostLedgerHandle } from "./cost-ledger-fGS_u_O1.js";
|
|
2
|
+
import { a as RunRecord, n as RunCostProvenance, u as RunTokenUsage } from "./run-record-DcObtIGh.js";
|
|
3
|
+
import { A as ChatClient, m as JudgeInput } from "./types-Cc3qbqzj.js";
|
|
4
|
+
import { t as TraceAnalysisStore } from "./store-CxJry_cs.js";
|
|
5
|
+
//#region src/analyst/types.d.ts
|
|
6
|
+
/**
|
|
7
|
+
* Unified envelope every analyst emits. Schema-versioned so renderers
|
|
8
|
+
* and time-series diffs survive future field additions.
|
|
9
|
+
*/
|
|
10
|
+
interface AnalystFinding {
|
|
11
|
+
schema_version: '1.0.0';
|
|
12
|
+
/**
|
|
13
|
+
* Stable hash over identity-defining fields (analyst_id + canonical
|
|
14
|
+
* claim + area + optional subject). Two findings from two runs that
|
|
15
|
+
* "are the same finding" share this id — that's what `diffFindings`
|
|
16
|
+
* uses to compute appeared/disappeared sets across runs.
|
|
17
|
+
*/
|
|
18
|
+
finding_id: string;
|
|
19
|
+
analyst_id: string;
|
|
20
|
+
produced_at: string;
|
|
21
|
+
severity: AnalystSeverity;
|
|
22
|
+
/**
|
|
23
|
+
* Coarse classification. Renderers group by this. Free-form so
|
|
24
|
+
* domain-specific analysts can introduce categories without a
|
|
25
|
+
* schema change ('agent-reasoning', 'verification', 'cost',
|
|
26
|
+
* 'tool-use', 'safety', 'latency', 'data-quality', ...).
|
|
27
|
+
*/
|
|
28
|
+
area: string;
|
|
29
|
+
claim: string;
|
|
30
|
+
rationale?: string;
|
|
31
|
+
evidence_refs: EvidenceRef[];
|
|
32
|
+
recommended_action?: string;
|
|
33
|
+
validation_plan?: string;
|
|
34
|
+
/** 0..1 — the analyst's own confidence. Not calibrated across analysts. */
|
|
35
|
+
confidence: number;
|
|
36
|
+
/**
|
|
37
|
+
* Optional subject the finding is about — leaf id, agent id, request
|
|
38
|
+
* id. Included in finding_id when present so per-subject findings
|
|
39
|
+
* diff cleanly across runs.
|
|
40
|
+
*/
|
|
41
|
+
subject?: string;
|
|
42
|
+
/** True when this finding was lifted from a judge result rather than observed
|
|
43
|
+
* directly in a trace or artifact. Descriptive only: proposal access is
|
|
44
|
+
* controlled by `ProposalFinding.proposal_origin`. */
|
|
45
|
+
derived_from_judge?: boolean;
|
|
46
|
+
/** Analyst-private extras; renderers ignore unless they know the analyst. */
|
|
47
|
+
metadata?: Record<string, unknown>;
|
|
48
|
+
}
|
|
49
|
+
type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info';
|
|
50
|
+
/** Data sources that candidate generation may intentionally learn from. */
|
|
51
|
+
type ProposalFindingOrigin = 'search' | 'production';
|
|
52
|
+
/** A finding explicitly admitted as candidate-generation input. */
|
|
53
|
+
type ProposalFinding = AnalystFinding & {
|
|
54
|
+
readonly proposal_origin: ProposalFindingOrigin;
|
|
55
|
+
};
|
|
56
|
+
interface EvidenceRef {
|
|
57
|
+
/**
|
|
58
|
+
* Where the evidence lives. `span` and `event` refer to OTLP trace
|
|
59
|
+
* elements; `artifact` to a file inside the run's artifact tree;
|
|
60
|
+
* `finding` to another AnalystFinding (cross-analyst chaining);
|
|
61
|
+
* `metric` to a named scalar reading the renderer knows how to read.
|
|
62
|
+
*/
|
|
63
|
+
kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
|
|
64
|
+
uri: string;
|
|
65
|
+
excerpt?: string;
|
|
66
|
+
}
|
|
67
|
+
/**
|
|
68
|
+
* The discriminator the registry uses to pass the right input.
|
|
69
|
+
* `custom` is the escape hatch — analysts that need something else
|
|
70
|
+
* (e.g. an embedding cache, a partner SDK handle) read it from
|
|
71
|
+
* `AnalystRunInputs.custom[<analyst id>]`.
|
|
72
|
+
*/
|
|
73
|
+
type AnalystInputKind = 'trace-store' | 'artifact-dir' | 'run-record' | 'judge-input' | 'custom';
|
|
74
|
+
interface AnalystCost {
|
|
75
|
+
/** `deterministic` analysts MUST NOT call the LLM. */
|
|
76
|
+
kind: 'deterministic' | 'llm';
|
|
77
|
+
/** Optional declared upper bound; the registry can enforce a budget. */
|
|
78
|
+
est_usd_per_run?: number;
|
|
79
|
+
/** Models the analyst expects to use (informational). */
|
|
80
|
+
models?: string[];
|
|
81
|
+
/** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */
|
|
82
|
+
settlement_timeout_ms?: number;
|
|
83
|
+
}
|
|
84
|
+
interface AnalystRequirements {
|
|
85
|
+
/** Min number of shots / samples the analyst needs to produce signal. */
|
|
86
|
+
min_shots?: number;
|
|
87
|
+
/** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */
|
|
88
|
+
capabilities?: string[];
|
|
89
|
+
}
|
|
90
|
+
/**
|
|
91
|
+
* What's passed to every analyst call. The registry resolves which
|
|
92
|
+
* field the analyst's `inputKind` selects and asserts it's present.
|
|
93
|
+
*/
|
|
94
|
+
interface AnalystRunInputs {
|
|
95
|
+
traceStore?: TraceAnalysisStore;
|
|
96
|
+
artifactDir?: string;
|
|
97
|
+
runRecord?: RunRecord;
|
|
98
|
+
judgeInput?: JudgeInput;
|
|
99
|
+
/** Keyed by analyst id; populated by callers that registered custom analysts. */
|
|
100
|
+
custom?: Record<string, unknown>;
|
|
101
|
+
}
|
|
102
|
+
interface AnalystContext {
|
|
103
|
+
runId: string;
|
|
104
|
+
/** Stable correlation id so logs from a single registry.run() share a tag. */
|
|
105
|
+
correlationId: string;
|
|
106
|
+
/** Enforced wall-clock deadline (epoch ms). */
|
|
107
|
+
deadlineMs?: number;
|
|
108
|
+
/** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */
|
|
109
|
+
budgetUsd?: number;
|
|
110
|
+
/** Shared paid-call account when the analyst runs inside a larger campaign. */
|
|
111
|
+
costLedger?: CostLedgerHandle;
|
|
112
|
+
/** Attribution phase used when writing to the shared paid-call account. */
|
|
113
|
+
costPhase?: string;
|
|
114
|
+
/**
|
|
115
|
+
* Shared chat client. Analysts that call an LLM go through this so
|
|
116
|
+
* the operator picks transport (sandbox-sdk | router | cli-bridge |
|
|
117
|
+
* direct-provider | mock) at the registry boundary without touching
|
|
118
|
+
* analyst code.
|
|
119
|
+
*/
|
|
120
|
+
chat?: ChatClient;
|
|
121
|
+
/**
|
|
122
|
+
* Findings from a prior run the operator wants the analyst to see as
|
|
123
|
+
* retrieval context. Kinds that take advantage of cross-run memory
|
|
124
|
+
* (failure-mode "I saw this cluster last run", knowledge-gap "the wiki
|
|
125
|
+
* page I asked for is still missing") render these into the actor's
|
|
126
|
+
* working set. Filtering is the operator's job: pass the slice that
|
|
127
|
+
* matches the analyst's id, or pass everything and let the kind
|
|
128
|
+
* filter. Empty / absent means no cross-run context.
|
|
129
|
+
*/
|
|
130
|
+
priorFindings?: ReadonlyArray<AnalystFinding>;
|
|
131
|
+
/**
|
|
132
|
+
* Findings emitted by analysts that completed earlier in this registry run.
|
|
133
|
+
* This is separate from `priorFindings`: upstream findings are dependency
|
|
134
|
+
* context for the current pass, while prior findings are cross-run memory.
|
|
135
|
+
* The registry populates this only when `RegistryRunOpts.chainFindings` is on.
|
|
136
|
+
*/
|
|
137
|
+
upstreamFindings?: ReadonlyArray<AnalystFinding>;
|
|
138
|
+
/**
|
|
139
|
+
* Report metered work independently of findings. This keeps an empty finding
|
|
140
|
+
* set from erasing token/cost telemetry. Multiple receipts are accumulated.
|
|
141
|
+
*/
|
|
142
|
+
recordUsage?: (receipt: AnalystUsageReceipt) => void;
|
|
143
|
+
/** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */
|
|
144
|
+
tags?: Record<string, string>;
|
|
145
|
+
/** Logger callback — analysts SHOULD prefer this over console.* for testability. */
|
|
146
|
+
log?: (msg: string, fields?: Record<string, unknown>) => void;
|
|
147
|
+
/** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
|
|
148
|
+
signal?: AbortSignal;
|
|
149
|
+
}
|
|
150
|
+
/**
|
|
151
|
+
* The minimal contract. Concrete analysts can refine `TInput` so
|
|
152
|
+
* implementations stay type-safe (e.g. a trace analyst's `TInput` is
|
|
153
|
+
* `TraceAnalysisStore`); the registry passes the right field from
|
|
154
|
+
* `AnalystRunInputs` based on `inputKind`.
|
|
155
|
+
*/
|
|
156
|
+
interface Analyst<TInput = unknown> {
|
|
157
|
+
/** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */
|
|
158
|
+
readonly id: string;
|
|
159
|
+
/** Human-readable. One sentence. */
|
|
160
|
+
readonly description: string;
|
|
161
|
+
readonly inputKind: AnalystInputKind;
|
|
162
|
+
readonly cost: AnalystCost;
|
|
163
|
+
readonly requires?: AnalystRequirements;
|
|
164
|
+
/** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */
|
|
165
|
+
readonly version: string;
|
|
166
|
+
analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>;
|
|
167
|
+
}
|
|
168
|
+
/** Metered work performed by one analyst call. */
|
|
169
|
+
interface AnalystUsageReceipt {
|
|
170
|
+
/** Number of model-usage records observed at the provider boundary. */
|
|
171
|
+
calls: number | null;
|
|
172
|
+
/** Null when the provider did not return token accounting. */
|
|
173
|
+
tokens: RunTokenUsage | null;
|
|
174
|
+
/** Observed, estimated, or explicitly uncaptured dollar cost. */
|
|
175
|
+
cost: RunCostProvenance;
|
|
176
|
+
/** Known lower bound when one or more calls have uncaptured cost. */
|
|
177
|
+
knownCostUsd?: number;
|
|
178
|
+
}
|
|
179
|
+
/**
|
|
180
|
+
* Compute the stable finding_id from the identity-defining fields.
|
|
181
|
+
* Default implementation hashes {analyst_id, area, subject, normalized claim}.
|
|
182
|
+
* Analysts that emit findings whose claim text varies per run (timestamps,
|
|
183
|
+
* counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
|
|
184
|
+
* or (b) move the variable part into `rationale`/`metadata` and keep the
|
|
185
|
+
* `claim` static.
|
|
186
|
+
*/
|
|
187
|
+
declare function computeFindingId(input: {
|
|
188
|
+
analyst_id: string;
|
|
189
|
+
area: string;
|
|
190
|
+
subject?: string;
|
|
191
|
+
claim: string;
|
|
192
|
+
/** Override the claim for hashing — use when the displayed claim has run-specific bits. */
|
|
193
|
+
id_basis?: string;
|
|
194
|
+
}): string;
|
|
195
|
+
/**
|
|
196
|
+
* Convenience factory: produce a fully-formed AnalystFinding with the
|
|
197
|
+
* id computed automatically. Analyst code stays terse.
|
|
198
|
+
*/
|
|
199
|
+
declare function makeFinding(init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {
|
|
200
|
+
id_basis?: string;
|
|
201
|
+
produced_at?: string;
|
|
202
|
+
}): AnalystFinding;
|
|
203
|
+
/** Build a finding whose source is explicitly allowed during candidate generation. */
|
|
204
|
+
declare function makeProposalFinding(init: Omit<ProposalFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {
|
|
205
|
+
id_basis?: string;
|
|
206
|
+
produced_at?: string;
|
|
207
|
+
}): ProposalFinding;
|
|
208
|
+
interface AnalystRunSummary {
|
|
209
|
+
analyst_id: string;
|
|
210
|
+
status: 'ok' | 'skipped' | 'failed';
|
|
211
|
+
/** Why skipped — missing input, budget exceeded, capability unmet. */
|
|
212
|
+
reason?: string;
|
|
213
|
+
findings_count: number;
|
|
214
|
+
latency_ms: number;
|
|
215
|
+
/** Additive model usage and cost provenance for this analyst. */
|
|
216
|
+
usage: AnalystUsageReceipt;
|
|
217
|
+
/** When `status='failed'`: the error class + message, never the full stack. */
|
|
218
|
+
error?: {
|
|
219
|
+
class: string;
|
|
220
|
+
message: string;
|
|
221
|
+
};
|
|
222
|
+
}
|
|
223
|
+
interface AnalystRunResult {
|
|
224
|
+
run_id: string;
|
|
225
|
+
correlation_id: string;
|
|
226
|
+
started_at: string;
|
|
227
|
+
ended_at: string;
|
|
228
|
+
findings: AnalystFinding[];
|
|
229
|
+
per_analyst: AnalystRunSummary[];
|
|
230
|
+
/** Total LLM cost in USD across all analysts in this registry.run(). */
|
|
231
|
+
total_cost_usd: number;
|
|
232
|
+
/**
|
|
233
|
+
* Provenance for `total_cost_usd`. When uncaptured, the numeric field is only
|
|
234
|
+
* the known subtotal and must not be treated as the run's total spend.
|
|
235
|
+
*/
|
|
236
|
+
total_cost_provenance?: RunCostProvenance;
|
|
237
|
+
}
|
|
238
|
+
/**
|
|
239
|
+
* Events emitted by `AnalystRegistry.runStream(...)` in real time as
|
|
240
|
+
* the registry executes. UIs subscribe via `for await (const ev of
|
|
241
|
+
* registry.runStream(...))`; `registry.run(...)` is a thin collector
|
|
242
|
+
* over the same stream, so the two surfaces share their invariants.
|
|
243
|
+
*
|
|
244
|
+
* Per-finding events are intentionally omitted — analyzers are batch
|
|
245
|
+
* operations (an Ax actor returns the full `findings:json[]` at the
|
|
246
|
+
* end of the responder), so streaming inside one analyst would only
|
|
247
|
+
* emit partial JSON consumers can't render. The kind-completion event
|
|
248
|
+
* is the right granularity; subscribers wanting per-finding rendering
|
|
249
|
+
* iterate `event.findings` themselves.
|
|
250
|
+
*/
|
|
251
|
+
type AnalystRunEvent = {
|
|
252
|
+
type: 'run-started';
|
|
253
|
+
run_id: string;
|
|
254
|
+
correlation_id: string;
|
|
255
|
+
started_at: string;
|
|
256
|
+
/** The ordered list of analyst ids the registry will run. */
|
|
257
|
+
analyst_ids: ReadonlyArray<string>;
|
|
258
|
+
} | {
|
|
259
|
+
type: 'analyst-skipped';
|
|
260
|
+
summary: AnalystRunSummary;
|
|
261
|
+
} | {
|
|
262
|
+
type: 'analyst-started';
|
|
263
|
+
analyst_id: string;
|
|
264
|
+
started_at: string;
|
|
265
|
+
} | {
|
|
266
|
+
type: 'analyst-completed';
|
|
267
|
+
/** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */
|
|
268
|
+
summary: AnalystRunSummary;
|
|
269
|
+
findings: ReadonlyArray<AnalystFinding>;
|
|
270
|
+
} | {
|
|
271
|
+
type: 'run-completed';
|
|
272
|
+
result: AnalystRunResult;
|
|
273
|
+
};
|
|
274
|
+
//#endregion
|
|
275
|
+
export { makeFinding as _, AnalystInputKind as a, AnalystRunInputs as c, AnalystSeverity as d, AnalystUsageReceipt as f, computeFindingId as g, ProposalFindingOrigin as h, AnalystFinding as i, AnalystRunResult as l, ProposalFinding as m, AnalystContext as n, AnalystRequirements as o, EvidenceRef as p, AnalystCost as r, AnalystRunEvent as s, Analyst as t, AnalystRunSummary as u, makeProposalFinding as v };
|
|
276
|
+
//# sourceMappingURL=types-DVjczBM9.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types-DVjczBM9.d.ts","names":[],"sources":["../src/analyst/types.ts"],"mappings":";;;;;;;;;UA4BiB;EACf;;;;;;;EAOA;EACA;EACA;EACA,UAAU;;;;;;;EAOV;EACA;EACA;EACA,eAAe;EACf;EACA;;EAEA;;;;;;EAMA;;;;EAIA;;EAEA,WAAW;;KAGD;;KAGA;;KAGA,kBAAkB;WACnB,iBAAiB;;UAGX;;;;;;;EAOf;EACA;EACA;;;;;;;;KAWU;UAOK;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf;;EAEA;;;;;;UAOe;EACf,aAAa;EACb;EACA,YAAY;EACZ,aAAa;;EAEb,SAAS;;UAGM;EACf;;EAEA;;EAEA;;EAEA;;EAEA,aAAa;;EAEb;;;;;;;EAOA,OAAO;;;;;;;;;;EAUP,gBAAgB,cAAc;;;;;;;EAO9B,mBAAmB,cAAc;;;;;EAKjC,eAAe,SAAS;;EAExB,OAAO;;EAEP,OAAO,aAAa,SAAS;;EAE7B,SAAS;;;;;;;;UASM,QAAQ;;WAEd;;WAEA;WACA,WAAW;WACX,MAAM;WACN,WAAW;;WAEX;EACT,QAAQ,OAAO,QAAQ,KAAK,iBAAiB,QAAQ;;;UAItC;;EAEf;;EAEA,QAAQ;;EAER,MAAM;;EAEN;;;;;;;;;;iBAac,iBAAiB;EAC/B;EACA;EACA;EACA;;EAEA;;;;;;iBAyBc,YACd,MAAM,KAAK;EACT;EACA;IAED;;iBAiBa,oBACd,MAAM,KAAK;EACT;EACA;IAED;UAOc;EACf;EACA;;EAEA;EACA;EACA;;EAEA,OAAO;;EAEP;IAAU;IAAe;;;UAGV;EACf;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa;;EAEb;;;;;EAKA,wBAAwB;;;;;;;;;;;;;;;KAkBd;EAEN;EACA;EACA;EACA;;EAEA,aAAa;;EAGb;EACA,SAAS;;EAGT;EACA;EACA;;EAGA;;EAEA,SAAS;EACT,UAAU,cAAc;;EAGxB;EACA,QAAQ"}
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { S as PaidCallResult, T as RunPaidCallInput, a as CostChannel, c as CostLedgerHandle, f as CostLedgerSummary } from "./cost-ledger-fGS_u_O1.js";
|
|
2
2
|
import { u as RunTokenUsage } from "./run-record-DcObtIGh.js";
|
|
3
3
|
import { n as LlmCallMetadata } from "./llm-client-BiK4HW0u.js";
|
|
4
|
+
import { m as ProposalFinding } from "./types-DVjczBM9.js";
|
|
4
5
|
//#region src/campaign/types.d.ts
|
|
5
6
|
/** Stable identifier + kind tag for any scenario. Consumers
|
|
6
7
|
* extend with their per-domain payload (persona, task, requirement, ...). */
|
|
@@ -239,56 +240,38 @@ interface ScoredSurfaceOutcome {
|
|
|
239
240
|
scorableCells: number;
|
|
240
241
|
};
|
|
241
242
|
}
|
|
242
|
-
/**
|
|
243
|
-
*
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
interface ProposeContext<TFindings = unknown> {
|
|
249
|
-
currentSurface: MutableSurface;
|
|
250
|
-
history: GenerationRecord[];
|
|
251
|
-
findings: TFindings[];
|
|
243
|
+
/** Search state supplied to one candidate-generation call.
|
|
244
|
+
* Final evaluation data is not represented in this contract. */
|
|
245
|
+
interface ProposeContext<TFindings = ProposalFinding> {
|
|
246
|
+
readonly currentSurface: MutableSurface;
|
|
247
|
+
readonly history: ReadonlyArray<GenerationRecord>;
|
|
248
|
+
readonly findings: ReadonlyArray<TFindings>;
|
|
252
249
|
/** BREADTH: how many candidate surfaces to return this generation. */
|
|
253
|
-
populationSize: number;
|
|
254
|
-
generation: number;
|
|
255
|
-
signal: AbortSignal;
|
|
250
|
+
readonly populationSize: number;
|
|
251
|
+
readonly generation: number;
|
|
252
|
+
readonly signal: AbortSignal;
|
|
256
253
|
/** Present when a multi-track lineage requests this proposal. */
|
|
257
|
-
track?: ProposalTrackContext;
|
|
254
|
+
readonly track?: ProposalTrackContext;
|
|
258
255
|
/** Measured baseline for this optimization run. `runOptimization` always
|
|
259
256
|
* supplies it; optional for standalone proposer callers. */
|
|
260
|
-
baselineOutcome?: ScoredSurfaceOutcome;
|
|
257
|
+
readonly baselineOutcome?: ScoredSurfaceOutcome;
|
|
261
258
|
/** Measured result for `currentSurface`, the complete global incumbent every
|
|
262
259
|
* new candidate mutates. `runOptimization` always supplies it. */
|
|
263
|
-
incumbentOutcome?: ScoredSurfaceOutcome;
|
|
264
|
-
/** Optional analysis report produced before proposal. Opaque to the substrate:
|
|
265
|
-
* the proposer that consumes it owns the shape. */
|
|
266
|
-
report?: unknown;
|
|
267
|
-
/** Handle to all captured data — the proposer samples traces / artifacts /
|
|
268
|
-
* rewards here to ground its proposals. */
|
|
269
|
-
dataset?: LabeledScenarioStore;
|
|
260
|
+
readonly incumbentOutcome?: ScoredSurfaceOutcome;
|
|
270
261
|
/** DEPTH: max iterations the agentic generator may take per candidate.
|
|
271
262
|
* 1 = single-shot; >1 = it may iterate on its own change before handing it
|
|
272
263
|
* back to be measured. */
|
|
273
|
-
maxImprovementShots?: number;
|
|
264
|
+
readonly maxImprovementShots?: number;
|
|
274
265
|
/** GEPA Pareto frontier across ALL generations so far — the non-dominated
|
|
275
266
|
* surfaces by per-scenario objective vector. Empty/absent on generation 0
|
|
276
267
|
* (only the baseline is scored). A reflective proposer combines the
|
|
277
268
|
* complementary lessons of these parents (each excels on different
|
|
278
269
|
* scenarios) into a merged candidate. Proposers doing pure single-parent
|
|
279
270
|
* reflection may ignore it. See {@link ParetoParent}. */
|
|
280
|
-
paretoParents?: ParetoParent
|
|
271
|
+
readonly paretoParents?: ReadonlyArray<ParetoParent>;
|
|
281
272
|
/** Shared run spend account and receipt attribution phase. */
|
|
282
|
-
costLedger?: CostLedgerHandle;
|
|
283
|
-
costPhase?: string;
|
|
284
|
-
/** FIREWALL (non-negotiable): the held-out judge is write-only — its verdicts
|
|
285
|
-
* score the chosen output and gate promotion, and are NEVER an input to
|
|
286
|
-
* proposal/steering (else the optimizer games the acceptance axis = an
|
|
287
|
-
* oracle). This `never`-typed field makes that a compile-time tripwire: a
|
|
288
|
-
* proposer that tries to thread judge verdicts into the proposal will not type.
|
|
289
|
-
* Steering may consume TRACE-OBSERVABLE signals (what the agent did) via
|
|
290
|
-
* `findings`/`report`; it may NOT consume the judge's held-out verdict. */
|
|
291
|
-
judgeScores?: never;
|
|
273
|
+
readonly costLedger?: CostLedgerHandle;
|
|
274
|
+
readonly costPhase?: string;
|
|
292
275
|
}
|
|
293
276
|
/** A surface-improvement strategy. Given the current best
|
|
294
277
|
* surface, the history of what's been tried + scored, and any external
|
|
@@ -302,7 +285,7 @@ interface ProposeContext<TFindings = unknown> {
|
|
|
302
285
|
* behavior-fuzzing `MutationProposer` (`fuzz/types`), a scenario generator for
|
|
303
286
|
* a different loop.
|
|
304
287
|
*/
|
|
305
|
-
interface SurfaceProposer<TFindings =
|
|
288
|
+
interface SurfaceProposer<TFindings = ProposalFinding> {
|
|
306
289
|
kind: string;
|
|
307
290
|
/** Plan: propose N candidate surfaces for the next generation. A proposer
|
|
308
291
|
* may return bare `MutableSurface`s or `ProposedCandidate`s that carry the
|
|
@@ -312,7 +295,7 @@ interface SurfaceProposer<TFindings = unknown> {
|
|
|
312
295
|
/** Decide: stop early when the proposer judges the search converged or
|
|
313
296
|
* exhausted. Default (omitted) runs all `maxGenerations`. */
|
|
314
297
|
decide?(args: {
|
|
315
|
-
history: GenerationRecord
|
|
298
|
+
history: ReadonlyArray<GenerationRecord>;
|
|
316
299
|
}): {
|
|
317
300
|
stop: boolean;
|
|
318
301
|
reason?: string;
|
|
@@ -637,4 +620,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
|
|
|
637
620
|
}
|
|
638
621
|
//#endregion
|
|
639
622
|
export { LabeledScenarioWrite as A, ScoredSurfaceOutcome as B, JudgeDimension as C, LabeledScenarioSampleArgs as D, LabeledScenarioRecord as E, ProposeContext as F, labelTrustRank as G, SurfaceProposer as H, ProposedCandidate as I, RedactionStatus as L, OptimizerConfig as M, ParetoParent as N, LabeledScenarioSource as O, ProposalTrackContext as P, Scenario as R, JudgeConfig as S, LabelTrust as T, TraceSpan as U, SessionScript as V, isProposedCandidate as W, GateDecision as _, CampaignResult as a, GenerationRecord as b, CampaignTraceWriter as c, DispatchContext as d, DispatchFn as f, GateContribution as g, GateContext as h, CampaignCostMeter as i, MutableSurface as j, LabeledScenarioStore as k, CodeSurface as l, GateCheckStatus as m, CampaignArtifactWriter as n, CampaignScenarioIdentity as o, Gate as p, CampaignCellResult as r, CampaignTokenUsage as s, CampaignAggregates as t, ComponentSurface as u, GateResult as v, JudgeScore as w, JudgeAggregate as x, GenerationCandidate as y, ScenarioAggregate as z };
|
|
640
|
-
//# sourceMappingURL=types-
|
|
623
|
+
//# sourceMappingURL=types-DiWLru6Z.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types-DiWLru6Z.d.ts","names":[],"sources":["../src/campaign/types.ts"],"mappings":";;;;;;;UA+BiB;EACf;EACA;EACA;;;;;EAKA;;;UAIe,iCAAiC,KAAK;EACrD;;;;;UAMe;EACf;EACA;EACA;EACA;EACA,QAAQ;EACR,OAAO;EACP,WAAW;EACX,MAAM;;EAEN;;EAEA;;;;;;;;EAQA;;;;KAKU,WAAW,kBAAkB,UAAU,cACjD,UAAU,WACV,KAAK,oBACF,QAAQ;;;;UAOI,cAAc,WAAW;EACxC;EACA;EACA;;EAEA;;;EAGA,sBAAsB,UAAU,WAAW,sBAAsB,UAAU,cAAc;;UAK1E;;EAEf;;EAEA;;;;;;;;;UAUe,YAAY,WAAW,kBAAkB,WAAW;EACnE;EACA,YAAY;;;;EAIZ;;;EAGA,MAAM;IACJ,UAAU;IACV,UAAU;IACV,QAAQ;;IAER,aAAa;IACb;IACA,WAAW;MACT,aAAa,QAAQ;EACzB,aAAa,UAAU;;;;;;;;;;;UAYR;EACf,YAAY;EACZ;EACA;;EAEA,UAAU;;;;EAIV;;;EAGA;;EAEA;;EAEA,WAAW,eAAe;;;;;;;UAUX;WACN;;;WAGA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;;WAGA;aACE;aACA;aACA;;;WAGF;;;UAIM;WACN;WACA,YAAY,SAAS;;;;;;;;;;;KAYpB,0BAA0B,mBAAmB;;;;;;;UAQxC;EACf,SAAS;;EAET;;;;EAIA;;;;iBAKc,oBACd,OAAO,iBAAiB,oBACvB,SAAS;;;;;;;;;UAkBK;EACf,SAAS;EACT;;;EAGA,YAAY;;;EAGZ;;EAEA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EACA;;;;;UAMe;;;EAGf;;EAEA;EACA;EACA;EACA,YAAY;;;;;EAKZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;EAC1E;IACE;IACA;;;;;UAMa,eAAe,YAAY;WACjC,gBAAgB;WAChB,SAAS,cAAc;WACvB,UAAU,cAAc;;WAExB;WACA;WACA,QAAQ;;WAER,QAAQ;;;WAGR,kBAAkB;;;WAGlB,mBAAmB;;;;WAInB;;;;;;;WAOA,gBAAgB,cAAc;;WAE9B,aAAa;WACb;;;;;;;;;;;;;;UAeM,gBAAgB,YAAY;EAC3C;;;;;EAKA,QAAQ,KAAK,eAAe,aAAa,QAAQ,MAAM,iBAAiB;;;EAGxE,QAAQ;IAAQ,SAAS,cAAc;;IAAwB;IAAe;;;UAG/D;EACf;EACA;EACA,mBAAmB,qBAAqB;;UAGzB,wBAAwB;EACvC,UAAU;;;KAMA;;KAGA;UAEK;EACf;EACA,QAAQ;EACR;;UAGe,YAAY,WAAW,kBAAkB;EACxD,oBAAoB,YAAY;EAChC,oBAAoB,YAAY;;EAEhC,aAAa,YAAY,eAAe;;;;;EAKxC,sBAAsB,YAAY,eAAe;;;;;;;EAOjD,yBAAyB,YAAY,eAAe;;;EAGpD,uBAAuB,YAAY;EACnC,WAAW;EACX;IAAQ;IAAmB;;;EAE3B,aAAa;EACb;EACA,QAAQ;;UAGO;EACf,UAAU;EACV;EACA,mBAAmB;EACnB;;;UAIe,KAAK,qBAAqB,kBAAkB,WAAW;EACtE;EACA,OAAO,KAAK,YAAY,WAAW,aAAa,QAAQ;;;;UAOzC;EACf,KAAK,cAAc,aAAa,0BAA0B;EAC1D,SAAS;;UAGM;EACf,IAAI,aAAa;EACjB,aAAa,aAAa;;;;UAKX;EACf,MAAM,cAAc,kBAAkB,aAAa;EACnD,UAAU,cAAc,iBAAiB;;;;;;KAO/B,qBAAqB;;;;;UAMhB;;EAEf,YAAY,GACV,OAAO,KAAK,iBAAiB;IAC3B,UAAU;MAEX,QAAQ,eAAe;;;;;KAQhB;KAOA;;;;;;;;;;;;;;;KAgBA;;iBASI,eAAe,OAAO;;;;UAOrB,qBAAqB,kBAAkB,WAAW,UAAU;EAC3E,UAAU;EACV,UAAU;EACV,aAAa,eAAe;EAC5B,QAAQ;EACR;EACA;EACA,iBAAiB;;;;;EAKjB,aAAa;;EAEb;;UAGe,sBAAsB,kBAAkB,WAAW,UAAU,6BACpE,qBAAqB,WAAW;;EAExC;;;EAGA;;UAGe;EACf;;EAEA;;;;EAIA;EACA;IACE;IACA,SAAS,wBAAwB;IACjC;IACA;;;;;IAKA,WAAW;;;UAIE;EACf,QAAQ,OAAO,uBAAuB;EACtC,OAAO,MAAM,4BAA4B,QAAQ;EACjD,QAAQ;IACN;IACA;IACA,UAAU;;;IAGV,SAAS,OAAO;;;UAMH,mBAAmB;;;EAGlC;EACA;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa,eAAe;EAC5B;;EAEA;;EAEA;;;EAGA,YAAY;;;;EAIZ;EACA;EACA;EACA;;EAEA;;EAEA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA,YAAY;EACZ;;;;;;UAOe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;;EAIA;;;;EAIA;IACE;IACA;IACA,iBAAiB;MAAQ;MAAgB;;;;;EAI3C,YAAY;;;;;;;;;;EAUZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;;;EAG1E;;;;EAIA;;UAGe;EACf,SAAS,eAAe;EACxB,YAAY,eAAe;;EAE3B,MAAM;;EAEN;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe,eAAe,qBAAqB,kBAAkB,WAAW;;EAEhF;;EAEA;EACA;;EAEA;EACA;EACA;EACA;EACA,OAAO,MAAM,mBAAmB;EAChC,YAAY;EACZ;IACE,aAAa;IACb;;EAEF,OAAO;EACP;EACA;EACA,iBAAiB;;;EAGjB,WAAW,MAAM,2BAA2B,KAAK"}
|
|
@@ -407,6 +407,11 @@ Concurrent processes cannot write the same compatible run at the same time.
|
|
|
407
407
|
|
|
408
408
|
Use `SurfaceProposer` when your code or runtime owns candidate creation.
|
|
409
409
|
The proposer receives the current surface, prior campaign history, findings, generation number, requested population size, and cancellation signal.
|
|
410
|
+
Every proposal finding must declare `proposal_origin: 'search' | 'production'`.
|
|
411
|
+
`runOptimization()` rejects unclassified findings before it calls the proposer.
|
|
412
|
+
Opaque reports and capture stores are not proposal input.
|
|
413
|
+
Convert analysis into explicit findings or close over a caller-owned knowledge source in your proposer.
|
|
414
|
+
Any caller-owned source must exclude final evaluation cases and results.
|
|
410
415
|
|
|
411
416
|
```ts
|
|
412
417
|
import type { SurfaceProposer } from '@tangle-network/agent-eval/campaign'
|