@tangle-network/agent-eval 0.79.0 → 0.81.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +101 -169
- package/dist/adapters/http.d.ts +2 -2
- package/dist/adapters/langchain.d.ts +2 -2
- package/dist/adapters/otel.d.ts +4 -4
- package/dist/{agent-profile-aSEaJ9Pl.d.ts → agent-profile-D0PBIWlV.d.ts} +14 -2
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +3 -3
- package/dist/{analyst-t7zZS3TV.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/belief-state/index.d.ts +524 -0
- package/dist/belief-state/index.js +1862 -0
- package/dist/belief-state/index.js.map +1 -0
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/calibration-Cpr3WaX3.d.ts +101 -0
- package/dist/campaign/index.d.ts +40 -120
- package/dist/campaign/index.js +129 -238
- package/dist/campaign/index.js.map +1 -1
- package/dist/chunk-4DIJWVUT.js +131 -0
- package/dist/chunk-4DIJWVUT.js.map +1 -0
- package/dist/{chunk-RPLZ4OIB.js → chunk-BABOZOSN.js} +7 -4
- package/dist/{chunk-RPLZ4OIB.js.map → chunk-BABOZOSN.js.map} +1 -1
- package/dist/{chunk-IHDHUN2X.js → chunk-CVVHBFGN.js} +3 -3
- package/dist/chunk-CVVHBFGN.js.map +1 -0
- package/dist/{chunk-B26KI423.js → chunk-FZWAFVAA.js} +2 -2
- package/dist/{chunk-ITBRCT73.js → chunk-IDVBLYCY.js} +2 -10
- package/dist/chunk-IDVBLYCY.js.map +1 -0
- package/dist/{chunk-5LVWPNS5.js → chunk-L5G7OUKD.js} +4 -4
- package/dist/{chunk-5LVWPNS5.js.map → chunk-L5G7OUKD.js.map} +1 -1
- package/dist/chunk-NPCTHQIO.js +91 -0
- package/dist/chunk-NPCTHQIO.js.map +1 -0
- package/dist/{chunk-GXHLRXDI.js → chunk-OTYQPHPL.js} +4 -4
- package/dist/{chunk-6REHLN5J.js → chunk-QS3RBQPI.js} +2 -2
- package/dist/{chunk-GWGO2K6Y.js → chunk-RBNA5AZT.js} +2 -2
- package/dist/chunk-S42AWHMP.js +697 -0
- package/dist/chunk-S42AWHMP.js.map +1 -0
- package/dist/chunk-VI2UW6B6.js +162 -0
- package/dist/chunk-VI2UW6B6.js.map +1 -0
- package/dist/{chunk-CF67I6QY.js → chunk-VIDQF3F5.js} +2 -2
- package/dist/{chunk-XXNIODOM.js → chunk-WJL2NJXN.js} +3 -3
- package/dist/{chunk-LB2UOI5F.js → chunk-YGYXHNAQ.js} +47 -160
- package/dist/chunk-YGYXHNAQ.js.map +1 -0
- package/dist/{chunk-KX6F6NCG.js → chunk-Z7VFTS2J.js} +2 -2
- package/dist/{chunk-ZPSKPT3V.js → chunk-ZZ2HOPME.js} +5 -2
- package/dist/chunk-ZZ2HOPME.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/code-agent-session-BRXmavYv.d.ts +80 -0
- package/dist/contract/index.d.ts +132 -18
- package/dist/contract/index.js +139 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CehLtoET.d.ts → control-GeE8OhpN.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/governance/index.d.ts +1 -1
- package/dist/hosted/index.d.ts +4 -4
- package/dist/{index-B1RKber3.d.ts → index-DE3RXAXD.d.ts} +1 -1
- package/dist/index.d.ts +79 -288
- package/dist/index.js +87 -410
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-dlpEzQDi.d.ts → insight-report-3ADTfClO.d.ts} +1 -1
- package/dist/{kind-factory-DqV2t1Xk.d.ts → kind-factory-CVecZZG_.d.ts} +2 -2
- package/dist/{llm-client-DbjLfz-K.d.ts → llm-client-CuUg2Mn3.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +6 -99
- package/dist/meta-eval/index.js +7 -76
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/off-policy-DiwuKKg7.d.ts +132 -0
- package/dist/openapi.json +1 -1
- package/dist/{outcome-store-D6KWmYvj.d.ts → outcome-store-rnXLEqSn.d.ts} +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{provenance-CEAJI9rm.d.ts → provenance-B9Q4886D.d.ts} +4 -4
- package/dist/{registry-BmEuU94S.d.ts → registry-DrEQ3Luj.d.ts} +2 -2
- package/dist/{release-report-CXXZlR8g.d.ts → release-report-hlNtD12q.d.ts} +2 -2
- package/dist/reporting.d.ts +6 -6
- package/dist/reporting.js +3 -3
- package/dist/{researcher-rInLj9De.d.ts → researcher-BLPHBbNV.d.ts} +3 -3
- package/dist/rl.d.ts +11 -141
- package/dist/rl.js +10 -124
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-CWyWWLBg.d.ts → rubric-predictive-validity-CnEl9Jc8.d.ts} +2 -2
- package/dist/{run-campaign-OVEZF24D.js → run-campaign-4Y5V5CN3.js} +3 -3
- package/dist/{run-improvement-loop-Bgu4C59E.d.ts → run-improvement-loop-D6PZOoQL.d.ts} +2 -2
- package/dist/{run-record-sItO5ftF.d.ts → run-record-De9VarXR.d.ts} +1 -1
- package/dist/{semantic-concept-judge-Du4ZVyef.d.ts → semantic-concept-judge-DIEgr_6v.d.ts} +6 -6
- package/dist/{statistics-B7yCbi9i.d.ts → statistics-CnC1FMbx.d.ts} +3 -7
- package/dist/{store-GmBE2pZZ.d.ts → store-C1YxJDEK.d.ts} +1 -1
- package/dist/{summary-report-BTaXq1TS.d.ts → summary-report-Db0dDSWP.d.ts} +1 -1
- package/dist/traces.d.ts +5 -5
- package/dist/{types-DRvV0zRo.d.ts → types-Cu3u_x59.d.ts} +3 -3
- package/dist/{types-QHG0KnkF.d.ts → types-D7lLRYe9.d.ts} +2 -2
- package/dist/wire/index.js +2 -2
- package/dist/workflow/index.d.ts +7 -7
- package/dist/workflow/index.js +1 -1
- package/docs/concepts.md +1 -0
- package/docs/research/belief-state-agent-eval-roadmap.md +590 -0
- package/docs/research/research-roadmap.md +1 -0
- package/docs/self-improvement-map.md +111 -0
- package/package.json +7 -2
- package/dist/chunk-IHDHUN2X.js.map +0 -1
- package/dist/chunk-ITBRCT73.js.map +0 -1
- package/dist/chunk-LB2UOI5F.js.map +0 -1
- package/dist/chunk-ZPSKPT3V.js.map +0 -1
- /package/dist/{chunk-B26KI423.js.map → chunk-FZWAFVAA.js.map} +0 -0
- /package/dist/{chunk-GXHLRXDI.js.map → chunk-OTYQPHPL.js.map} +0 -0
- /package/dist/{chunk-6REHLN5J.js.map → chunk-QS3RBQPI.js.map} +0 -0
- /package/dist/{chunk-GWGO2K6Y.js.map → chunk-RBNA5AZT.js.map} +0 -0
- /package/dist/{chunk-CF67I6QY.js.map → chunk-VIDQF3F5.js.map} +0 -0
- /package/dist/{chunk-XXNIODOM.js.map → chunk-WJL2NJXN.js.map} +0 -0
- /package/dist/{chunk-KX6F6NCG.js.map → chunk-Z7VFTS2J.js.map} +0 -0
- /package/dist/{run-campaign-OVEZF24D.js.map → run-campaign-4Y5V5CN3.js.map} +0 -0
package/dist/{rubric-predictive-validity-CWyWWLBg.d.ts → rubric-predictive-validity-CnEl9Jc8.d.ts}
RENAMED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { R as RunRecord } from './run-record-
|
|
2
|
-
import {
|
|
1
|
+
import { R as RunRecord } from './run-record-De9VarXR.js';
|
|
2
|
+
import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* Rubric predictive validity — does our eval rubric predict deployment
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import {
|
|
2
2
|
runCampaign
|
|
3
|
-
} from "./chunk-
|
|
4
|
-
import "./chunk-
|
|
3
|
+
} from "./chunk-ZZ2HOPME.js";
|
|
4
|
+
import "./chunk-IDVBLYCY.js";
|
|
5
5
|
import "./chunk-3BFEG2F6.js";
|
|
6
6
|
import "./chunk-PZ5AY32C.js";
|
|
7
7
|
export {
|
|
8
8
|
runCampaign
|
|
9
9
|
};
|
|
10
|
-
//# sourceMappingURL=run-campaign-
|
|
10
|
+
//# sourceMappingURL=run-campaign-4Y5V5CN3.js.map
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { L as LlmClientOptions } from './llm-client-
|
|
2
|
-
import { I as ImprovementDriver, S as Scenario,
|
|
1
|
+
import { L as LlmClientOptions } from './llm-client-CuUg2Mn3.js';
|
|
2
|
+
import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-D7lLRYe9.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* @experimental
|
|
@@ -315,4 +315,4 @@ declare function parseRunRecordSafe(input: unknown): {
|
|
|
315
315
|
/** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */
|
|
316
316
|
declare function roundTripRunRecord(record: RunRecord): RunRecord;
|
|
317
317
|
|
|
318
|
-
export { type AgentProfileCell as A, validateAgentProfileCell as B, validateRunRecord as C, verifyAgentProfileCell as D, type JudgeScoresRecord as J, type RunRecord as R, type SandboxAgentProfileLike as S, type
|
|
318
|
+
export { type AgentProfileCell as A, validateAgentProfileCell as B, validateRunRecord as C, verifyAgentProfileCell as D, type JudgeScoresRecord as J, type RunRecord as R, type SandboxAgentProfileLike as S, type RunSplitTag as a, type RunTokenUsage as b, type RunJudgeMetadata as c, type AgentProfileCellInput as d, AGENT_PROFILE_KINDS as e, type AgentProfileCellSchemaVersion as f, AgentProfileCellValidationError as g, type AgentProfileDimensionValue as h, type AgentProfileHarness as i, type AgentProfileJson as j, type AgentProfileKind as k, type AgentProfileSource as l, type AgentProfileSourceInput as m, type RunOutcome as n, RunRecordValidationError as o, agentProfileCellHashMaterial as p, agentProfileCellKey as q, assertRunAgentProfileCell as r, buildAgentProfileCell as s, buildSandboxAgentProfileCell as t, groupRunsByAgentProfileCell as u, isRunRecord as v, parseRunRecordSafe as w, requireAgentProfileCell as x, roundTripRunRecord as y, toAgentProfileJson as z };
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import { AxAIService } from '@ax-llm/ax';
|
|
2
|
-
import { c as TraceAnalystKindSpec } from './kind-factory-
|
|
3
|
-
import { b as AnalystRegistryOptions,
|
|
2
|
+
import { c as TraceAnalystKindSpec } from './kind-factory-CVecZZG_.js';
|
|
3
|
+
import { b as AnalystRegistryOptions, a as AnalystRegistry } from './registry-DrEQ3Luj.js';
|
|
4
4
|
import { z } from 'zod';
|
|
5
|
-
import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-
|
|
6
|
-
import { a as TraceAnalystSpan } from './store-
|
|
7
|
-
import { L as LlmClientOptions } from './llm-client-
|
|
5
|
+
import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-Cu3u_x59.js';
|
|
6
|
+
import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
|
|
7
|
+
import { L as LlmClientOptions } from './llm-client-CuUg2Mn3.js';
|
|
8
8
|
import { S as Severity } from './multi-layer-verifier-DlWCXuxL.js';
|
|
9
9
|
|
|
10
10
|
interface CreateAnalystAiConfig {
|
|
@@ -615,7 +615,7 @@ declare const DEFAULT_COMPLEXITY_WEIGHTS: Record<ConceptComplexity, number>;
|
|
|
615
615
|
interface SemanticConceptJudgeOptions {
|
|
616
616
|
/** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */
|
|
617
617
|
model?: string;
|
|
618
|
-
/** Per-call timeout. Default
|
|
618
|
+
/** Per-call timeout. Default 300s. */
|
|
619
619
|
timeoutMs?: number;
|
|
620
620
|
/** Pipeline budget for the prompt (source blob truncation). Default 45000. */
|
|
621
621
|
maxSourceChars?: number;
|
|
@@ -1,13 +1,9 @@
|
|
|
1
1
|
import { C as ContinuousAgreementOptions, a as ContinuousAgreement } from './judge-calibration-DilmB3Ml.js';
|
|
2
2
|
import { J as JudgeScore } from './types-Croy5h7V.js';
|
|
3
3
|
|
|
4
|
-
/**
|
|
5
|
-
*
|
|
6
|
-
|
|
7
|
-
* already use inverted scoring in the prompt (10 = no hallucination),
|
|
8
|
-
* but this function ensures consistency if raw scores leak through.
|
|
9
|
-
*/
|
|
10
|
-
declare function normalizeScores(scores: JudgeScore[]): JudgeScore[];
|
|
4
|
+
/** Identity: dimensions already follow "higher = better" by prompt convention
|
|
5
|
+
* (inverted dims like hallucination are scored 10 = best at the source). */
|
|
6
|
+
declare const normalizeScores: (scores: JudgeScore[]) => JudgeScore[];
|
|
11
7
|
/** Weighted mean — falls back to uniform weights when omitted */
|
|
12
8
|
declare function weightedMean(scores: {
|
|
13
9
|
score: number;
|
|
@@ -245,4 +245,4 @@ interface TraceAnalysisStore {
|
|
|
245
245
|
}): Promise<SearchSpanResult>;
|
|
246
246
|
}
|
|
247
247
|
|
|
248
|
-
export { DEFAULT_TRACE_ANALYST_BUDGETS as D, type QueryTracesPage as Q, type SearchSpanResult as S, type TraceAnalysisStore as T, type ViewSpansResult as V, type TraceAnalystSpan as a, type DatasetOverview as b, type SearchTraceResult as c, type SpanMatchRecord as d, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX as e, type TraceAnalystByteBudgets as f, type TraceAnalystFilters as g, type TraceAnalystSpanKind as h, type TraceAnalystSpanStatus as i, type TraceAnalystTraceSummary as j, type ViewTraceOversized as k, type ViewTraceResult as l };
|
|
248
|
+
export { DEFAULT_TRACE_ANALYST_BUDGETS as D, type ErrorCluster as E, type QueryTracesPage as Q, type SearchSpanResult as S, type TraceAnalysisStore as T, type ViewSpansResult as V, type TraceAnalystSpan as a, type DatasetOverview as b, type SearchTraceResult as c, type SpanMatchRecord as d, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX as e, type TraceAnalystByteBudgets as f, type TraceAnalystFilters as g, type TraceAnalystSpanKind as h, type TraceAnalystSpanStatus as i, type TraceAnalystTraceSummary as j, type ViewTraceOversized as k, type ViewTraceResult as l };
|
package/dist/traces.d.ts
CHANGED
|
@@ -10,11 +10,11 @@ export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as ll
|
|
|
10
10
|
export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionReport, R as RedactionRule, r as redactString, c as redactValue } from './redact-B40YG2M_.js';
|
|
11
11
|
import { R as Run } from './schema-m0gsnbt3.js';
|
|
12
12
|
export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, d as RetrievalSpan, g as RunLayer, b as RunOutcome, f as RunStatus, e as SandboxSpan, S as Span, j as SpanBase, c as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, a as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-m0gsnbt3.js';
|
|
13
|
-
import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-
|
|
14
|
-
export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-
|
|
15
|
-
import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-
|
|
16
|
-
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-
|
|
17
|
-
import {
|
|
13
|
+
import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
|
|
14
|
+
export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
15
|
+
import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-C1YxJDEK.js';
|
|
16
|
+
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-C1YxJDEK.js';
|
|
17
|
+
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from './run-record-De9VarXR.js';
|
|
18
18
|
import { AxFunction } from '@ax-llm/ax';
|
|
19
19
|
|
|
20
20
|
/**
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { R as RunRecord } from './run-record-
|
|
2
|
-
import { T as TraceAnalysisStore } from './store-
|
|
1
|
+
import { R as RunRecord } from './run-record-De9VarXR.js';
|
|
2
|
+
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
3
|
import { a as JudgeInput } from './types-Croy5h7V.js';
|
|
4
|
-
import { b as LlmCallRequest, c as LlmCallResult } from './llm-client-
|
|
4
|
+
import { b as LlmCallRequest, c as LlmCallResult } from './llm-client-CuUg2Mn3.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* ChatClient — the single LLM abstraction analysts call.
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { b as RunTokenUsage } from './run-record-De9VarXR.js';
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
4
|
* @experimental
|
|
@@ -497,4 +497,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
|
|
|
497
497
|
scenarios: Array<Pick<TScenario, 'id' | 'kind'>>;
|
|
498
498
|
}
|
|
499
499
|
|
|
500
|
-
export { isProposedCandidate as A, labelTrustRank as B, type CampaignAggregates as C, type DispatchFn as D, type Gate as G, type ImprovementDriver as I, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizerConfig as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeConfig as a, type DispatchContext as b, type
|
|
500
|
+
export { isProposedCandidate as A, labelTrustRank as B, type CampaignAggregates as C, type DispatchFn as D, type Gate as G, type ImprovementDriver as I, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizerConfig as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeConfig as a, type DispatchContext as b, type GateDecision as c, type CampaignArtifactWriter as d, type CampaignCellResult as e, type CampaignCostMeter as f, type CampaignResult as g, type CampaignTraceWriter as h, type CodeSurface as i, type GateContext as j, type GateResult as k, type GenerationCandidate as l, type GenerationRecord as m, type JudgeDimension as n, type Mutator as o, type SessionScript as p, type ProposeContext as q, type LabeledScenarioWrite as r, type LabeledScenarioSampleArgs as s, type LabeledScenarioRecord as t, type LabelTrust as u, type LabeledScenarioSource as v, type CampaignTokenUsage as w, type JudgeAggregate as x, type ProposedCandidate as y, type ScenarioAggregate as z };
|
package/dist/wire/index.js
CHANGED
|
@@ -34,8 +34,8 @@ import {
|
|
|
34
34
|
runRpcOnce,
|
|
35
35
|
startServer,
|
|
36
36
|
startServerAsync
|
|
37
|
-
} from "../chunk-
|
|
38
|
-
import "../chunk-
|
|
37
|
+
} from "../chunk-QS3RBQPI.js";
|
|
38
|
+
import "../chunk-CVVHBFGN.js";
|
|
39
39
|
import "../chunk-PC4UYEBM.js";
|
|
40
40
|
import "../chunk-3BFEG2F6.js";
|
|
41
41
|
import "../chunk-PZ5AY32C.js";
|
package/dist/workflow/index.d.ts
CHANGED
|
@@ -1,24 +1,24 @@
|
|
|
1
1
|
import { W as WorkflowTopology } from '../harness-optimizer-EnEnQPsr.js';
|
|
2
|
-
import {
|
|
3
|
-
import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-
|
|
4
|
-
import { F as FailureClusterInsight } from '../insight-report-
|
|
2
|
+
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from '../run-record-De9VarXR.js';
|
|
3
|
+
import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-Cu3u_x59.js';
|
|
4
|
+
import { F as FailureClusterInsight } from '../insight-report-3ADTfClO.js';
|
|
5
5
|
import { a as VerificationReport, L as LayerResult } from '../multi-layer-verifier-DlWCXuxL.js';
|
|
6
6
|
import { F as FailureClusterReport } from '../failure-cluster-CL7IVgkJ.js';
|
|
7
7
|
import { R as RedactionRule, a as RedactionReport } from '../redact-B40YG2M_.js';
|
|
8
8
|
import { D as DatasetSplit } from '../dataset-B2kL-fSM.js';
|
|
9
9
|
import { a as FeedbackTrajectory } from '../feedback-trajectory-B3rErRsh.js';
|
|
10
|
-
import { a as PairedBootstrapResult } from '../statistics-
|
|
10
|
+
import { a as PairedBootstrapResult } from '../statistics-CnC1FMbx.js';
|
|
11
11
|
import '../pareto-E-pembql.js';
|
|
12
12
|
import '../run-critic-BAIjX99r.js';
|
|
13
13
|
import '../schema-m0gsnbt3.js';
|
|
14
14
|
import '../store-CKUAgsJz.js';
|
|
15
15
|
import '../errors-Dwqw-T_m.js';
|
|
16
|
-
import '../store-
|
|
16
|
+
import '../store-C1YxJDEK.js';
|
|
17
17
|
import '../types-Croy5h7V.js';
|
|
18
18
|
import '@tangle-network/tcloud';
|
|
19
|
-
import '../llm-client-
|
|
19
|
+
import '../llm-client-CuUg2Mn3.js';
|
|
20
20
|
import '../raw-provider-sink-C46HDghv.js';
|
|
21
|
-
import '../summary-report-
|
|
21
|
+
import '../summary-report-Db0dDSWP.js';
|
|
22
22
|
import '../judge-calibration-DilmB3Ml.js';
|
|
23
23
|
import '../control-runtime-DuFBYg7A.js';
|
|
24
24
|
import '../emitter-DEZwY14K.js';
|
package/dist/workflow/index.js
CHANGED
package/docs/concepts.md
CHANGED
|
@@ -182,6 +182,7 @@ release decision.
|
|
|
182
182
|
|
|
183
183
|
## Where to go next
|
|
184
184
|
|
|
185
|
+
- **Confused by "GEPA / HALO / trace analysis / drivers everywhere"?** → [self-improvement-map.md](./self-improvement-map.md) — one loop, four roles, the seven-driver catalog (production vs bench-only), and why `gepa-refine` is the same loop on a test bench.
|
|
185
186
|
- **Need the layman feature map?** → [feature-guide.md](./feature-guide.md) — what each primitive does, when to use it, integration patterns, and guardrails.
|
|
186
187
|
- **Just want to score a string against a rubric?** → [wire-protocol.md](./wire-protocol.md) — HTTP/RPC interface, pluggable from any language.
|
|
187
188
|
- **Need a reusable driver/worker/evaluator loop?** → [control-runtime.md](./control-runtime.md) — generic runtime plus coding, browser, computer-use, and research integration patterns.
|