@tangle-network/agent-eval 0.109.1 → 0.110.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -0
- package/dist/analyst/index.d.ts +10 -12
- package/dist/analyst/index.js +8 -11
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DJYpep3L.d.ts → analyze-runs-Dmz6LA9e.d.ts} +4 -4
- package/dist/{baseline-Bbid3WoO.d.ts → baseline-DsNteOgR.d.ts} +32 -2
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +7 -8
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -2
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/{calibration-BPmzuVPk.d.ts → calibration-Dz8TQV4y.d.ts} +2 -2
- package/dist/campaign/index.d.ts +62 -20
- package/dist/campaign/index.js +9 -8
- package/dist/{chunk-LOJ2QVCE.js → chunk-2IY4ILP4.js} +2 -2
- package/dist/{chunk-LIEJUH2I.js → chunk-6PL5MGDL.js} +9 -9
- package/dist/{chunk-2OGPXHOB.js → chunk-7NX6ZSBG.js} +36 -7
- package/dist/chunk-7NX6ZSBG.js.map +1 -0
- package/dist/{chunk-R6D7NEYJ.js → chunk-GBI5J5DB.js} +81 -11
- package/dist/chunk-GBI5J5DB.js.map +1 -0
- package/dist/{chunk-YEHAEDUD.js → chunk-IMWDSFUM.js} +604 -2
- package/dist/chunk-IMWDSFUM.js.map +1 -0
- package/dist/{chunk-OVPVM4JC.js → chunk-J4AKLZEV.js} +15 -4
- package/dist/{chunk-OVPVM4JC.js.map → chunk-J4AKLZEV.js.map} +1 -1
- package/dist/{chunk-JZXGWLK5.js → chunk-MHNQWM4I.js} +62 -6
- package/dist/chunk-MHNQWM4I.js.map +1 -0
- package/dist/{chunk-QRVS7MX4.js → chunk-OW47B5WA.js} +3 -5
- package/dist/{chunk-QRVS7MX4.js.map → chunk-OW47B5WA.js.map} +1 -1
- package/dist/{chunk-DBDRR6GF.js → chunk-PLOMR3HP.js} +48 -2
- package/dist/chunk-PLOMR3HP.js.map +1 -0
- package/dist/{chunk-GDZAWO2I.js → chunk-QFGTU7MT.js} +2 -2
- package/dist/{chunk-V7HNA47Z.js → chunk-RSVSSZKF.js} +5 -5
- package/dist/{chunk-5PK3626Q.js → chunk-XRGOKCMO.js} +88 -17
- package/dist/chunk-XRGOKCMO.js.map +1 -0
- package/dist/{code-agent-session-rnJKlqmT.d.ts → code-agent-session-yitf9I-F.d.ts} +1 -1
- package/dist/contract/index.d.ts +26 -28
- package/dist/contract/index.js +11 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-B8UthSBL.d.ts → control-U8LBKUES.d.ts} +5 -6
- package/dist/control.d.ts +8 -9
- package/dist/control.js +6 -8
- package/dist/{dataset-DS7ytHZU.d.ts → dataset-NENEzRgk.d.ts} +1 -1
- package/dist/{default-registry-BswHCXnU.d.ts → default-registry-Bcf1uKVI.d.ts} +1 -2
- package/dist/{emitter-C2rqGH_l.d.ts → emitter-BRchAAAx.d.ts} +2 -2
- package/dist/{failure-cluster-DH9Flgcf.d.ts → failure-cluster-C48PiReX.d.ts} +2 -2
- package/dist/feedback-trajectory-pDcz1lQ1.d.ts +348 -0
- package/dist/{gepa-B3x5Ulcv.d.ts → gepa-T8T215nw.d.ts} +149 -6
- package/dist/hosted/index.d.ts +7 -7
- package/dist/{index-pPtfoIJO.d.ts → index-Dc3VLGhp.d.ts} +2 -2
- package/dist/index.d.ts +645 -61
- package/dist/index.js +1282 -190
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-B4xrdwEK.d.ts → insight-report-D4cXFsLt.d.ts} +1 -1
- package/dist/{integrity-DqGZg3st.d.ts → integrity-qemeBAyx.d.ts} +1 -1
- package/dist/{types-D1ytG0Yg.d.ts → kind-factory-20hcaYpf.d.ts} +169 -2
- package/dist/meta-eval/index.d.ts +5 -5
- package/dist/meta-eval/index.js +1 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{multi-layer-verifier-CI4jdX-q.d.ts → multi-layer-verifier-BsqKuLyN.d.ts} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +6 -7
- package/dist/pipelines/index.js +3 -6
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-DQUXYMDm.d.ts → policy-edit-D2bBDZDf.d.ts} +2 -2
- package/dist/{pre-registration-BUhVPzE7.d.ts → pre-registration-BepVVa6P.d.ts} +3 -3
- package/dist/{provenance-DdDhf6cg.d.ts → provenance-CyxkvEi9.d.ts} +3 -5
- package/dist/{query-0aTmbmQe.d.ts → query-Ck190MOd.d.ts} +2 -2
- package/dist/{release-report-DeJpsBiA.d.ts → release-report-oBfOz8ku.d.ts} +3 -3
- package/dist/reporting.d.ts +8 -8
- package/dist/{researcher-Wc7dx6GM.d.ts → researcher-CaH0CwFC.d.ts} +6 -6
- package/dist/rl.d.ts +568 -15
- package/dist/rl.js +4 -4
- package/dist/{rubric-predictive-validity-DPnyG-CE.d.ts → rubric-predictive-validity-C-fMteAW.d.ts} +1 -1
- package/dist/{run-record-I-Z3JNvO.d.ts → run-record-DksGsfgv.d.ts} +1 -1
- package/dist/{runtime-trajectory-iW9IhV3e.d.ts → runtime-trajectory-h5i0SZUj.d.ts} +1 -1
- package/dist/{schema-m0gsnbt3.d.ts → schema-SGWcK9wa.d.ts} +1 -1
- package/dist/{semantic-concept-judge-BmNZPB_j.d.ts → semantic-concept-judge-D7z6JCLZ.d.ts} +57 -4
- package/dist/{store-BcFXE6LG.d.ts → store-BsVi7ncX.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-QMZVe3P-.d.ts → summary-report-Bz-0-t8v.d.ts} +2 -2
- package/dist/{test-graded-scenario-DeODGLra.d.ts → test-graded-scenario-mzYBKspu.d.ts} +3 -3
- package/dist/traces.d.ts +54 -11
- package/dist/traces.js +25 -27
- package/dist/{types-BdIv5dvA.d.ts → types-v--ctu-b.d.ts} +2 -2
- package/dist/wire/index.d.ts +5 -6
- package/docs/improvement-glossary.md +14 -13
- package/package.json +1 -71
- package/dist/adapters/http.d.ts +0 -142
- package/dist/adapters/http.js +0 -203
- package/dist/adapters/http.js.map +0 -1
- package/dist/adapters/langchain.d.ts +0 -95
- package/dist/adapters/langchain.js +0 -34
- package/dist/adapters/langchain.js.map +0 -1
- package/dist/adapters/otel.d.ts +0 -112
- package/dist/adapters/otel.js +0 -110
- package/dist/adapters/otel.js.map +0 -1
- package/dist/chunk-2OGPXHOB.js.map +0 -1
- package/dist/chunk-45EEMHTC.js +0 -35
- package/dist/chunk-45EEMHTC.js.map +0 -1
- package/dist/chunk-5BKGXME7.js +0 -65
- package/dist/chunk-5BKGXME7.js.map +0 -1
- package/dist/chunk-5PK3626Q.js.map +0 -1
- package/dist/chunk-6SK5VFYK.js +0 -100
- package/dist/chunk-6SK5VFYK.js.map +0 -1
- package/dist/chunk-DBDRR6GF.js.map +0 -1
- package/dist/chunk-DJWX3GVS.js +0 -81
- package/dist/chunk-DJWX3GVS.js.map +0 -1
- package/dist/chunk-FOUG2VVS.js +0 -855
- package/dist/chunk-FOUG2VVS.js.map +0 -1
- package/dist/chunk-JZXGWLK5.js.map +0 -1
- package/dist/chunk-K7QEIHHJ.js +0 -613
- package/dist/chunk-K7QEIHHJ.js.map +0 -1
- package/dist/chunk-KKHDIONI.js +0 -414
- package/dist/chunk-KKHDIONI.js.map +0 -1
- package/dist/chunk-KMPRBJK4.js +0 -74
- package/dist/chunk-KMPRBJK4.js.map +0 -1
- package/dist/chunk-Q2JRAWRI.js +0 -196
- package/dist/chunk-Q2JRAWRI.js.map +0 -1
- package/dist/chunk-R6D7NEYJ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-STGVSCDH.js +0 -202
- package/dist/chunk-STGVSCDH.js.map +0 -1
- package/dist/chunk-YEHAEDUD.js.map +0 -1
- package/dist/control-runtime-Acf9CGhw.d.ts +0 -182
- package/dist/corpus-eBVwhCp1.d.ts +0 -560
- package/dist/counterfactual-DlOz8PBx.d.ts +0 -85
- package/dist/diagnose.d.ts +0 -252
- package/dist/diagnose.js +0 -382
- package/dist/diagnose.js.map +0 -1
- package/dist/feedback-trajectory-C9KCo8ag.d.ts +0 -169
- package/dist/governance/index.d.ts +0 -135
- package/dist/governance/index.js +0 -18
- package/dist/governance/index.js.map +0 -1
- package/dist/groundedness/index.d.ts +0 -112
- package/dist/groundedness/index.js +0 -77
- package/dist/groundedness/index.js.map +0 -1
- package/dist/harness-optimizer-mOl9XX_O.d.ts +0 -106
- package/dist/kind-factory-DvIGo_cP.d.ts +0 -171
- package/dist/knowledge/index.d.ts +0 -103
- package/dist/knowledge/index.js +0 -18
- package/dist/knowledge/index.js.map +0 -1
- package/dist/pareto-E-pembql.d.ts +0 -81
- package/dist/perf/index.d.ts +0 -123
- package/dist/perf/index.js +0 -18
- package/dist/perf/index.js.map +0 -1
- package/dist/prm/index.d.ts +0 -104
- package/dist/prm/index.js +0 -265
- package/dist/prm/index.js.map +0 -1
- package/dist/product-benchmark/index.d.ts +0 -247
- package/dist/product-benchmark/index.js +0 -37
- package/dist/product-benchmark/index.js.map +0 -1
- package/dist/red-team-KmmiqBlY.d.ts +0 -63
- package/dist/redact-B40YG2M_.d.ts +0 -45
- package/dist/rubric-Cc6UHvUb.d.ts +0 -73
- package/dist/run-critic-CmMf05uV.d.ts +0 -56
- package/dist/sink-fetch-B1Yg4Til.d.ts +0 -101
- package/dist/telemetry/file.d.ts +0 -19
- package/dist/telemetry/file.js +0 -45
- package/dist/telemetry/file.js.map +0 -1
- package/dist/telemetry/index.d.ts +0 -38
- package/dist/telemetry/index.js +0 -130
- package/dist/telemetry/index.js.map +0 -1
- package/dist/testing-C21CHsq2.d.ts +0 -20
- package/dist/testing.d.ts +0 -1
- package/dist/testing.js +0 -8
- package/dist/testing.js.map +0 -1
- package/dist/trajectory-2TkpSEVh.d.ts +0 -33
- package/dist/workflow/index.d.ts +0 -496
- package/dist/workflow/index.js +0 -2178
- package/dist/workflow/index.js.map +0 -1
- /package/dist/{chunk-LOJ2QVCE.js.map → chunk-2IY4ILP4.js.map} +0 -0
- /package/dist/{chunk-LIEJUH2I.js.map → chunk-6PL5MGDL.js.map} +0 -0
- /package/dist/{chunk-GDZAWO2I.js.map → chunk-QFGTU7MT.js.map} +0 -0
- /package/dist/{chunk-V7HNA47Z.js.map → chunk-RSVSSZKF.js.map} +0 -0
|
@@ -1,106 +0,0 @@
|
|
|
1
|
-
import { O as Objective, P as ParetoResult } from './pareto-E-pembql.js';
|
|
2
|
-
import { R as RunScore, a as RunTrace, b as RunScoreWeights } from './run-critic-CmMf05uV.js';
|
|
3
|
-
|
|
4
|
-
interface SteeringRolePrompt {
|
|
5
|
-
system?: string;
|
|
6
|
-
append?: string;
|
|
7
|
-
}
|
|
8
|
-
interface SteeringBundle {
|
|
9
|
-
id: string;
|
|
10
|
-
coderPrompt?: string;
|
|
11
|
-
continuePrompt?: string;
|
|
12
|
-
reviewerPrompts?: Record<string, string>;
|
|
13
|
-
skills?: string[];
|
|
14
|
-
rolePrompts?: Record<string, SteeringRolePrompt>;
|
|
15
|
-
metadata?: Record<string, unknown>;
|
|
16
|
-
}
|
|
17
|
-
interface SteeringDelta {
|
|
18
|
-
coderPrompt?: string;
|
|
19
|
-
continuePrompt?: string;
|
|
20
|
-
reviewerPrompts?: Record<string, string>;
|
|
21
|
-
skills?: string[];
|
|
22
|
-
rolePrompts?: Record<string, SteeringRolePrompt>;
|
|
23
|
-
metadata?: Record<string, unknown>;
|
|
24
|
-
}
|
|
25
|
-
declare function mergeSteeringBundle(base: SteeringBundle, delta: SteeringDelta): SteeringBundle;
|
|
26
|
-
declare function renderSteeringText(bundle: SteeringBundle): string;
|
|
27
|
-
|
|
28
|
-
type HarnessIntervention = 'continue' | 'plan' | 'audit' | 'recover' | 'repair' | 'verify' | 'final_gate' | 'wait_for_measurement' | 'abort';
|
|
29
|
-
interface WorkflowTopology {
|
|
30
|
-
id: string;
|
|
31
|
-
interventions: HarnessIntervention[];
|
|
32
|
-
maxParallelBranches?: number;
|
|
33
|
-
metadata?: Record<string, unknown>;
|
|
34
|
-
}
|
|
35
|
-
interface MeasurementPolicy {
|
|
36
|
-
required: string[];
|
|
37
|
-
optional?: string[];
|
|
38
|
-
promoteOn?: Array<keyof RunScore | 'aggregate'>;
|
|
39
|
-
}
|
|
40
|
-
interface HarnessVariant {
|
|
41
|
-
id: string;
|
|
42
|
-
steering?: SteeringBundle;
|
|
43
|
-
topology?: WorkflowTopology;
|
|
44
|
-
measurement?: MeasurementPolicy;
|
|
45
|
-
budgets?: Record<string, number>;
|
|
46
|
-
models?: Record<string, string>;
|
|
47
|
-
reviewers?: Record<string, string>;
|
|
48
|
-
metadata?: Record<string, unknown>;
|
|
49
|
-
}
|
|
50
|
-
interface HarnessScenario {
|
|
51
|
-
id: string;
|
|
52
|
-
task: string;
|
|
53
|
-
split?: 'train' | 'validation' | 'test' | string;
|
|
54
|
-
metadata?: Record<string, unknown>;
|
|
55
|
-
}
|
|
56
|
-
interface HarnessRunRequest {
|
|
57
|
-
variant: HarnessVariant;
|
|
58
|
-
scenario: HarnessScenario;
|
|
59
|
-
trialIndex: number;
|
|
60
|
-
}
|
|
61
|
-
interface HarnessAdapter {
|
|
62
|
-
run(request: HarnessRunRequest): Promise<RunTrace>;
|
|
63
|
-
}
|
|
64
|
-
interface HarnessRunResult {
|
|
65
|
-
variant: HarnessVariant;
|
|
66
|
-
scenario: HarnessScenario;
|
|
67
|
-
trialIndex: number;
|
|
68
|
-
trace: RunTrace;
|
|
69
|
-
score: RunScore;
|
|
70
|
-
aggregate: number;
|
|
71
|
-
}
|
|
72
|
-
interface HarnessVariantReport {
|
|
73
|
-
variant: HarnessVariant;
|
|
74
|
-
runs: HarnessRunResult[];
|
|
75
|
-
aggregateMean: number;
|
|
76
|
-
passRate: number;
|
|
77
|
-
costUsdMean: number;
|
|
78
|
-
wallSecondsMean: number;
|
|
79
|
-
scoreMean: RunScore;
|
|
80
|
-
}
|
|
81
|
-
interface HarnessSelection {
|
|
82
|
-
winner: HarnessVariantReport;
|
|
83
|
-
frontier: ParetoResult<HarnessVariantReport>;
|
|
84
|
-
reports: HarnessVariantReport[];
|
|
85
|
-
}
|
|
86
|
-
interface HarnessExperimentResult {
|
|
87
|
-
results: HarnessRunResult[];
|
|
88
|
-
selection: HarnessSelection;
|
|
89
|
-
}
|
|
90
|
-
interface HarnessExperimentConfig {
|
|
91
|
-
adapter: HarnessAdapter;
|
|
92
|
-
variants: HarnessVariant[];
|
|
93
|
-
scenarios: HarnessScenario[];
|
|
94
|
-
trialsPerScenario?: number;
|
|
95
|
-
parallelism?: number;
|
|
96
|
-
weights?: Partial<RunScoreWeights>;
|
|
97
|
-
objectives?: Array<Objective<HarnessVariantReport>>;
|
|
98
|
-
score?: (trace: RunTrace, request: HarnessRunRequest) => RunScore | Promise<RunScore>;
|
|
99
|
-
onResult?: (result: HarnessRunResult) => void | Promise<void>;
|
|
100
|
-
}
|
|
101
|
-
declare const DEFAULT_HARNESS_OBJECTIVES: Array<Objective<HarnessVariantReport>>;
|
|
102
|
-
declare function runHarnessExperiment(config: HarnessExperimentConfig): Promise<HarnessExperimentResult>;
|
|
103
|
-
declare function selectHarnessVariant(results: HarnessRunResult[], objectives?: Array<Objective<HarnessVariantReport>>): HarnessSelection;
|
|
104
|
-
declare function summarizeHarnessResults(results: HarnessRunResult[]): HarnessVariantReport[];
|
|
105
|
-
|
|
106
|
-
export { DEFAULT_HARNESS_OBJECTIVES as D, type HarnessAdapter as H, type MeasurementPolicy as M, type SteeringBundle as S, type WorkflowTopology as W, type HarnessExperimentConfig as a, type HarnessExperimentResult as b, type HarnessIntervention as c, type HarnessRunRequest as d, type HarnessRunResult as e, type HarnessScenario as f, type HarnessSelection as g, type HarnessVariant as h, type HarnessVariantReport as i, type SteeringDelta as j, type SteeringRolePrompt as k, runHarnessExperiment as l, mergeSteeringBundle as m, summarizeHarnessResults as n, renderSteeringText as r, selectHarnessVariant as s };
|
|
@@ -1,171 +0,0 @@
|
|
|
1
|
-
import { AxAIService, AxFunction } from '@ax-llm/ax';
|
|
2
|
-
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
|
-
import { z } from 'zod';
|
|
4
|
-
import { g as AnalystCost, b as AnalystContext, a as Analyst } from './types-D1ytG0Yg.js';
|
|
5
|
-
|
|
6
|
-
/**
|
|
7
|
-
* Typed Ax output for analyst findings.
|
|
8
|
-
*
|
|
9
|
-
* Replaces the legacy `findings:string[]` pattern (where every bullet
|
|
10
|
-
* became a flat-severity `AnalystFinding`) with a structured object
|
|
11
|
-
* array. Ax binds the field as `findings:json[]` so the provider emits
|
|
12
|
-
* native structured output; at the kind-factory boundary we Zod-validate
|
|
13
|
-
* each emitted finding so malformed rows fail loud instead of being
|
|
14
|
-
* silently lifted with default severity.
|
|
15
|
-
*
|
|
16
|
-
* Why not `f.object().array()` directly in the signature? The Ax
|
|
17
|
-
* signature string `question:string -> findings:json[]` already lets
|
|
18
|
-
* the provider emit JSON arrays. A Zod boundary is required either
|
|
19
|
-
* way (the provider can return any JSON), and Zod gives us a single
|
|
20
|
-
* validation surface independent of which Ax version is installed.
|
|
21
|
-
*/
|
|
22
|
-
|
|
23
|
-
declare const ANALYST_SEVERITIES: readonly ["critical", "high", "medium", "low", "info"];
|
|
24
|
-
declare const RawAnalystFindingSchema: z.ZodObject<{
|
|
25
|
-
severity: z.ZodEnum<{
|
|
26
|
-
info: "info";
|
|
27
|
-
critical: "critical";
|
|
28
|
-
medium: "medium";
|
|
29
|
-
low: "low";
|
|
30
|
-
high: "high";
|
|
31
|
-
}>;
|
|
32
|
-
claim: z.ZodString;
|
|
33
|
-
subject: z.ZodOptional<z.ZodString>;
|
|
34
|
-
evidence_uri: z.ZodString;
|
|
35
|
-
evidence_excerpt: z.ZodOptional<z.ZodString>;
|
|
36
|
-
confidence: z.ZodNumber;
|
|
37
|
-
rationale: z.ZodOptional<z.ZodString>;
|
|
38
|
-
recommended_action: z.ZodOptional<z.ZodString>;
|
|
39
|
-
}, z.core.$strict>;
|
|
40
|
-
type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
|
|
41
|
-
/**
|
|
42
|
-
* Description embedded into the actor prompt so the LLM knows what
|
|
43
|
-
* shape to emit. Kept here so kinds share one source of truth rather
|
|
44
|
-
* than restating the schema in every prompt.
|
|
45
|
-
*/
|
|
46
|
-
declare const RAW_FINDING_SCHEMA_PROMPT = "Each finding MUST be a JSON object with these fields:\n - severity: one of \"critical\" | \"high\" | \"medium\" | \"low\" | \"info\"\n - claim: one-sentence statement (max 2000 chars)\n - subject?: the routing locus this finding is about. It MUST be one of the exact subject forms listed in this kind's instructions above (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`, `tool-doc:<tool>`). A free phrase, a bare noun, or any form not in that list is REJECTED at parse time and the finding is discarded \u2014 omit subject entirely rather than guess a form.\n - evidence_uri: REQUIRED, never blank. Exactly one of \"span://<trace_id>/<span_id>\" (trace evidence), \"artifact://<relative-path>\" (files), \"metric://<name>\" (named scalars) \u2014 ALWAYS cite a real id surfaced by the tools. If you have no citable id, do not emit the finding.\n - evidence_excerpt?: short quote (<=2000 chars) from the cited span/artifact\n - confidence: number 0..1 \u2014 0.9+ when backed by exact quotes, 0.6-0.8 for inferred patterns, <0.5 for speculative\n - rationale?: one or two sentences explaining the reasoning\n - recommended_action?: concrete change phrased as an imperative (\"Add ...\", \"Replace ...\", \"Stop ...\") \u2014 omit when the finding is purely descriptive\n\nEmit an empty array when the question has no findings to report. Do not fabricate evidence.";
|
|
47
|
-
/**
|
|
48
|
-
* Validate one row emitted by the LLM. Returns the typed finding on
|
|
49
|
-
* success; returns `null` and logs the reason on failure so the kind
|
|
50
|
-
* factory can skip-and-count rather than abort the whole analyst run.
|
|
51
|
-
*/
|
|
52
|
-
declare function parseRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): RawAnalystFinding | null;
|
|
53
|
-
|
|
54
|
-
/**
|
|
55
|
-
* Analyst-kind factory — the typed way to define trace analysts.
|
|
56
|
-
*
|
|
57
|
-
* A "kind" is a specialized analyst whose actor prompt, tool subset,
|
|
58
|
-
* and Ax recursion config target one failure-mode lens (failure-mode
|
|
59
|
-
* classification, knowledge gap discovery, knowledge poisoning, recursive
|
|
60
|
-
* self-improvement, ...). Kinds emit findings in the typed `RawAnalystFinding`
|
|
61
|
-
* shape via a JSON-array Ax output; the factory validates each row with
|
|
62
|
-
* Zod and lifts it into `AnalystFinding[]` with no shape guessing.
|
|
63
|
-
*
|
|
64
|
-
* Composition rules:
|
|
65
|
-
* - Each kind owns its actor description. No generic "answer this
|
|
66
|
-
* question" prompt — the prompt names the failure lens.
|
|
67
|
-
* - Each kind picks a narrow tool subset from `ANALYST_TOOL_GROUPS`.
|
|
68
|
-
* A kind that never needs full-trace dumps can drop `viewTrace` /
|
|
69
|
-
* `viewSpans` and stay cheap.
|
|
70
|
-
* - Each kind declares its recursion + parallelism budget. Discovery-
|
|
71
|
-
* heavy kinds (failure-mode) get higher `maxDepth`; lens kinds
|
|
72
|
-
* (poisoning) usually stay at 0 since they have a tighter brief.
|
|
73
|
-
*
|
|
74
|
-
* Optimizer hook: kinds may declare `goldens` — labeled examples used
|
|
75
|
-
* by `AxMiPRO` / `AxBootstrapFewShot` / `AxGEPA` to fit the actor
|
|
76
|
-
* description programmatically. Stored on the kind, not the registry,
|
|
77
|
-
* because the right metric is kind-specific.
|
|
78
|
-
*/
|
|
79
|
-
|
|
80
|
-
/**
|
|
81
|
-
* Per-kind specification. The factory turns this into a regular
|
|
82
|
-
* `Analyst<TraceAnalysisStore>` ready for `AnalystRegistry.register()`.
|
|
83
|
-
*/
|
|
84
|
-
interface TraceAnalystKindSpec {
|
|
85
|
-
/** Stable id. Appears in finding_id, telemetry, and registry exclusions. */
|
|
86
|
-
id: string;
|
|
87
|
-
/** One-sentence description shown in `registry.list()`. */
|
|
88
|
-
description: string;
|
|
89
|
-
/** Coarse classification stamped on every emitted finding (`failure-mode`, `knowledge-gap`, ...). */
|
|
90
|
-
area: string;
|
|
91
|
-
/** Bump on any breaking change to the actor prompt or output schema. */
|
|
92
|
-
version: string;
|
|
93
|
-
/** Actor system prompt. Must instruct the LLM to emit `findings` per the schema. */
|
|
94
|
-
actorDescription: string;
|
|
95
|
-
/** Responder system prompt; falls back to a minimal "format the findings" instruction. */
|
|
96
|
-
responderDescription?: string;
|
|
97
|
-
/** Tool functions the actor may call. Pick narrow subsets via `ANALYST_TOOL_GROUPS`. */
|
|
98
|
-
buildTools: (store: TraceAnalysisStore) => AxFunction[];
|
|
99
|
-
/** Recursion budget. `maxDepth: 0` disables subagents. */
|
|
100
|
-
recursion?: {
|
|
101
|
-
maxDepth: number;
|
|
102
|
-
maxParallelSubagents?: number;
|
|
103
|
-
};
|
|
104
|
-
/** Actor turn cap. Default 12. */
|
|
105
|
-
maxTurns?: number;
|
|
106
|
-
/** Runtime char cap. Default 6000. */
|
|
107
|
-
maxRuntimeChars?: number;
|
|
108
|
-
/** Cost classification surfaced in `registry.list()` and budget enforcement. */
|
|
109
|
-
cost: AnalystCost;
|
|
110
|
-
/** Per-finding-row hook — kinds may reject / rewrite before lifting. */
|
|
111
|
-
postProcess?: (row: RawAnalystFinding, ctx: AnalystContext) => RawAnalystFinding | null;
|
|
112
|
-
/** Optional optimizer hook — populated when a kind wants to fit its prompt against labeled examples. */
|
|
113
|
-
goldens?: TraceAnalystGolden[];
|
|
114
|
-
}
|
|
115
|
-
/**
|
|
116
|
-
* One labeled example consumed by Ax optimizers (MIPRO / GEPA / Bootstrap).
|
|
117
|
-
* Each input is the same `{question}` an analyst would receive; `expected`
|
|
118
|
-
* is the ground-truth finding set a fitted prompt should produce on this
|
|
119
|
-
* input. Metric: kind-specific (default: F1 on `finding_id` overlap).
|
|
120
|
-
*/
|
|
121
|
-
interface TraceAnalystGolden {
|
|
122
|
-
question: string;
|
|
123
|
-
expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
|
|
124
|
-
}
|
|
125
|
-
interface CreateTraceAnalystKindOpts {
|
|
126
|
-
/** AxAIService bound at registration time. */
|
|
127
|
-
ai: AxAIService;
|
|
128
|
-
/** Optional model override; falls back to the AI service's default. */
|
|
129
|
-
model?: string;
|
|
130
|
-
/** Override the spec's `version` (e.g. when an optimizer has fitted a new prompt). */
|
|
131
|
-
versionSuffix?: string;
|
|
132
|
-
/**
|
|
133
|
-
* Optional two-phase recovery: when the agentic harvest is empty but the
|
|
134
|
-
* actor produced a substantive free-form `report`, extract findings from that
|
|
135
|
-
* prose via a tolerant chat-completions pass (`structureFindings`) — no
|
|
136
|
-
* strict-emission contract, so it works on weak models. Omit to leave the
|
|
137
|
-
* actor's harvest as-is (the report is still surfaced fail-loud either way).
|
|
138
|
-
*/
|
|
139
|
-
recovery?: {
|
|
140
|
-
baseUrl: string;
|
|
141
|
-
apiKey?: string;
|
|
142
|
-
model?: string;
|
|
143
|
-
fetchImpl?: typeof fetch;
|
|
144
|
-
};
|
|
145
|
-
}
|
|
146
|
-
/**
|
|
147
|
-
* Build an `Analyst<TraceAnalysisStore>` from a kind spec.
|
|
148
|
-
*
|
|
149
|
-
* Lifts the Ax pipeline once at registration time so the registry
|
|
150
|
-
* gets a stateless analyst. The Ax agent is freshly constructed per
|
|
151
|
-
* `analyze()` call (the agent carries chat-log + usage state we don't
|
|
152
|
-
* want shared across analyst runs).
|
|
153
|
-
*/
|
|
154
|
-
declare function createTraceAnalystKind(spec: TraceAnalystKindSpec, opts: CreateTraceAnalystKindOpts): Analyst<TraceAnalysisStore>;
|
|
155
|
-
/**
|
|
156
|
-
* Render a compact prior-findings block the actor reads alongside its
|
|
157
|
-
* brief. Each row is one line so the actor can scan dozens cheaply.
|
|
158
|
-
* The kind's prompt instructs the actor to (a) check whether a new
|
|
159
|
-
* cluster matches a prior `finding_id` (carry the id forward via
|
|
160
|
-
* `id_basis` to keep diffs stable) and (b) raise severity / confidence
|
|
161
|
-
* when a prior finding has reappeared without remediation.
|
|
162
|
-
*
|
|
163
|
-
* Returns the empty string when there are no prior findings — most
|
|
164
|
-
* runs are "first-of-its-kind" and the prompt stays unchanged.
|
|
165
|
-
*
|
|
166
|
-
* Exported for tests + for consumers that build their own actor
|
|
167
|
-
* prompts (e.g. specialized analysts living outside the default kinds).
|
|
168
|
-
*/
|
|
169
|
-
declare function renderPriorFindings(prior: AnalystContext['priorFindings']): string;
|
|
170
|
-
|
|
171
|
-
export { ANALYST_SEVERITIES as A, type CreateTraceAnalystKindOpts as C, RAW_FINDING_SCHEMA_PROMPT as R, type TraceAnalystKindSpec as T, type RawAnalystFinding as a, RawAnalystFindingSchema as b, type TraceAnalystGolden as c, createTraceAnalystKind as d, parseRawFinding as p, renderPriorFindings as r };
|
|
@@ -1,103 +0,0 @@
|
|
|
1
|
-
import { j as ControlSeverity, C as ControlEvalResult } from '../control-runtime-Acf9CGhw.js';
|
|
2
|
-
import { T as TraceEmitter } from '../emitter-C2rqGH_l.js';
|
|
3
|
-
import '../schema-m0gsnbt3.js';
|
|
4
|
-
import '../store-BcFXE6LG.js';
|
|
5
|
-
|
|
6
|
-
type KnowledgeRequirementCategory = 'user_specific' | 'company_specific' | 'domain_specific' | 'codebase_specific' | 'market_specific' | 'regulatory' | 'tool_api' | 'credential_or_secret' | 'runtime_environment' | 'preference' | 'historical_context';
|
|
7
|
-
type KnowledgeAcquisitionMode = 'ask_user' | 'search_web' | 'query_connector' | 'inspect_repo' | 'run_command' | 'infer_low_confidence' | 'not_available';
|
|
8
|
-
type KnowledgeImportance = 'blocking' | 'high' | 'medium' | 'low';
|
|
9
|
-
type KnowledgeFreshness = 'static' | 'monthly' | 'weekly' | 'daily' | 'realtime';
|
|
10
|
-
type KnowledgeSensitivity = 'public' | 'private' | 'secret';
|
|
11
|
-
type KnowledgeFallbackPolicy = 'block' | 'ask' | 'continue_with_caveat' | 'use_default';
|
|
12
|
-
interface KnowledgeRequirement {
|
|
13
|
-
id: string;
|
|
14
|
-
description: string;
|
|
15
|
-
requiredFor: string[];
|
|
16
|
-
category: KnowledgeRequirementCategory;
|
|
17
|
-
acquisitionMode: KnowledgeAcquisitionMode;
|
|
18
|
-
importance: KnowledgeImportance;
|
|
19
|
-
freshness: KnowledgeFreshness;
|
|
20
|
-
sensitivity: KnowledgeSensitivity;
|
|
21
|
-
confidenceNeeded: number;
|
|
22
|
-
currentConfidence: number;
|
|
23
|
-
evidenceIds: string[];
|
|
24
|
-
fallbackPolicy: KnowledgeFallbackPolicy;
|
|
25
|
-
/**
|
|
26
|
-
* ISO timestamp after which this requirement must be treated as stale.
|
|
27
|
-
* Stale requirements score as missing even when they still have evidence.
|
|
28
|
-
*/
|
|
29
|
-
validUntil?: string;
|
|
30
|
-
/** ISO timestamp for the last source-grounding or human verification pass. */
|
|
31
|
-
lastVerifiedAt?: string;
|
|
32
|
-
metadata?: Record<string, unknown>;
|
|
33
|
-
}
|
|
34
|
-
interface KnowledgeBundle {
|
|
35
|
-
taskId: string;
|
|
36
|
-
requirements: KnowledgeRequirement[];
|
|
37
|
-
evidenceIds: string[];
|
|
38
|
-
claimIds: string[];
|
|
39
|
-
wikiPageIds: string[];
|
|
40
|
-
userAnswers: Record<string, string>;
|
|
41
|
-
missing: KnowledgeRequirement[];
|
|
42
|
-
readinessScore: number;
|
|
43
|
-
metadata?: Record<string, unknown>;
|
|
44
|
-
}
|
|
45
|
-
type KnowledgeRecommendedAction = 'run_agent' | 'ask_user' | 'collect_web_data' | 'query_connectors' | 'inspect_repo' | 'build_domain_wiki' | 'continue_with_caveat' | 'abort_or_rescope';
|
|
46
|
-
interface KnowledgeReadinessReport {
|
|
47
|
-
taskId: string;
|
|
48
|
-
readinessScore: number;
|
|
49
|
-
blockingMissingRequirements: KnowledgeRequirement[];
|
|
50
|
-
nonBlockingGaps: KnowledgeRequirement[];
|
|
51
|
-
recommendedAction: KnowledgeRecommendedAction;
|
|
52
|
-
bundle: KnowledgeBundle;
|
|
53
|
-
severity: ControlSeverity;
|
|
54
|
-
reason: string;
|
|
55
|
-
}
|
|
56
|
-
interface UserQuestion {
|
|
57
|
-
id: string;
|
|
58
|
-
question: string;
|
|
59
|
-
reason: string;
|
|
60
|
-
requirementId: string;
|
|
61
|
-
importance: KnowledgeImportance;
|
|
62
|
-
answerType: 'free_text' | 'select_one' | 'multi_select' | 'file_upload' | 'credential' | 'url';
|
|
63
|
-
defaultIfSkipped?: string;
|
|
64
|
-
impactIfUnknown: string;
|
|
65
|
-
options?: string[];
|
|
66
|
-
metadata?: Record<string, unknown>;
|
|
67
|
-
}
|
|
68
|
-
interface DataAcquisitionPlan {
|
|
69
|
-
id: string;
|
|
70
|
-
requirementIds: string[];
|
|
71
|
-
mode: Exclude<KnowledgeAcquisitionMode, 'not_available' | 'infer_low_confidence'> | 'build_domain_wiki';
|
|
72
|
-
description: string;
|
|
73
|
-
priority: KnowledgeImportance;
|
|
74
|
-
expectedEvidenceIds?: string[];
|
|
75
|
-
questions?: UserQuestion[];
|
|
76
|
-
metadata?: Record<string, unknown>;
|
|
77
|
-
}
|
|
78
|
-
type KnowledgeResponsibleSurface = 'knowledge-requirements' | 'data-acquisition' | 'retrieval-policy' | 'user-question-policy';
|
|
79
|
-
|
|
80
|
-
interface ScoreKnowledgeReadinessOptions {
|
|
81
|
-
taskId: string;
|
|
82
|
-
requirements: KnowledgeRequirement[];
|
|
83
|
-
evidenceIds?: string[];
|
|
84
|
-
claimIds?: string[];
|
|
85
|
-
wikiPageIds?: string[];
|
|
86
|
-
userAnswers?: Record<string, string>;
|
|
87
|
-
metadata?: Record<string, unknown>;
|
|
88
|
-
now?: Date;
|
|
89
|
-
}
|
|
90
|
-
declare function scoreKnowledgeReadiness(options: ScoreKnowledgeReadinessOptions): KnowledgeReadinessReport;
|
|
91
|
-
declare function blockingKnowledgeEval(report: KnowledgeReadinessReport, options?: {
|
|
92
|
-
id?: string;
|
|
93
|
-
minimumScore?: number;
|
|
94
|
-
emitter?: TraceEmitter;
|
|
95
|
-
}): ControlEvalResult;
|
|
96
|
-
declare function knowledgeReadinessTracePayload(report: KnowledgeReadinessReport, options?: {
|
|
97
|
-
passed?: boolean;
|
|
98
|
-
minimumScore?: number;
|
|
99
|
-
}): Record<string, unknown>;
|
|
100
|
-
declare function userQuestionsForKnowledgeGaps(gaps: KnowledgeRequirement[]): UserQuestion[];
|
|
101
|
-
declare function acquisitionPlansForKnowledgeGaps(gaps: KnowledgeRequirement[]): DataAcquisitionPlan[];
|
|
102
|
-
|
|
103
|
-
export { type DataAcquisitionPlan, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFallbackPolicy, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRecommendedAction, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeResponsibleSurface, type KnowledgeSensitivity, type ScoreKnowledgeReadinessOptions, type UserQuestion, acquisitionPlansForKnowledgeGaps, blockingKnowledgeEval, knowledgeReadinessTracePayload, scoreKnowledgeReadiness, userQuestionsForKnowledgeGaps };
|
package/dist/knowledge/index.js
DELETED
|
@@ -1,18 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
acquisitionPlansForKnowledgeGaps,
|
|
3
|
-
blockingKnowledgeEval,
|
|
4
|
-
knowledgeReadinessTracePayload,
|
|
5
|
-
scoreKnowledgeReadiness,
|
|
6
|
-
userQuestionsForKnowledgeGaps
|
|
7
|
-
} from "../chunk-Q2JRAWRI.js";
|
|
8
|
-
import "../chunk-YEHAEDUD.js";
|
|
9
|
-
import "../chunk-TVVP3ZZQ.js";
|
|
10
|
-
import "../chunk-PZ5AY32C.js";
|
|
11
|
-
export {
|
|
12
|
-
acquisitionPlansForKnowledgeGaps,
|
|
13
|
-
blockingKnowledgeEval,
|
|
14
|
-
knowledgeReadinessTracePayload,
|
|
15
|
-
scoreKnowledgeReadiness,
|
|
16
|
-
userQuestionsForKnowledgeGaps
|
|
17
|
-
};
|
|
18
|
-
//# sourceMappingURL=index.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
|
|
@@ -1,81 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Pareto frontier — multi-objective optimization over candidate runs.
|
|
3
|
-
*
|
|
4
|
-
* Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
|
|
5
|
-
* trading off (cost, latency, quality) or (passRate, tokenBudget,
|
|
6
|
-
* ttfb), you rarely have a single "winner" — you have a set of
|
|
7
|
-
* non-dominated candidates. This module exposes:
|
|
8
|
-
*
|
|
9
|
-
* - `paretoFrontier`: filter a set of candidates to the non-dominated ones
|
|
10
|
-
* - `dominates`: does A dominate B across all objectives?
|
|
11
|
-
*
|
|
12
|
-
* Each objective is declared with a direction: 'maximize' (higher=better)
|
|
13
|
-
* or 'minimize' (lower=better). Candidates are any object; pass an
|
|
14
|
-
* `objective(candidate)` accessor.
|
|
15
|
-
*/
|
|
16
|
-
type Direction = 'maximize' | 'minimize';
|
|
17
|
-
interface Objective<T> {
|
|
18
|
-
/** Stable label used in reports. */
|
|
19
|
-
name: string;
|
|
20
|
-
direction: Direction;
|
|
21
|
-
value: (candidate: T) => number;
|
|
22
|
-
}
|
|
23
|
-
interface ParetoResult<T> {
|
|
24
|
-
frontier: T[];
|
|
25
|
-
dominated: T[];
|
|
26
|
-
/** Index map: frontier[i] dominates each of dominatedBy[i]. */
|
|
27
|
-
dominanceMap: Array<{
|
|
28
|
-
dominator: T;
|
|
29
|
-
dominated: T[];
|
|
30
|
-
}>;
|
|
31
|
-
}
|
|
32
|
-
/** Does candidate A weakly dominate B — strictly better on at least one objective and no worse on any? */
|
|
33
|
-
declare function dominates<T>(a: T, b: T, objectives: Objective<T>[]): boolean;
|
|
34
|
-
/**
|
|
35
|
-
* Compute the non-dominated frontier. Candidates with NaN/Infinity on any
|
|
36
|
-
* objective are excluded (can't rank them). A candidate enters the frontier
|
|
37
|
-
* iff no other candidate dominates it.
|
|
38
|
-
*/
|
|
39
|
-
declare function paretoFrontier<T>(candidates: T[], objectives: Objective<T>[]): ParetoResult<T>;
|
|
40
|
-
/**
|
|
41
|
-
* Weighted-sum scalarisation. Use as a tie-break / single-winner selector
|
|
42
|
-
* when callers don't want to consume a frontier. Each objective contributes
|
|
43
|
-
* its normalised value (0..1 via min-max across the candidate pool) times
|
|
44
|
-
* its weight; missing weights default to 1/N.
|
|
45
|
-
*
|
|
46
|
-
* Direction is honoured automatically — `minimize` axes have their values
|
|
47
|
-
* inverted before scaling so "higher scalar = better" always holds.
|
|
48
|
-
*/
|
|
49
|
-
declare function scalarScore<T>(candidates: T[], objectives: Objective<T>[], options?: {
|
|
50
|
-
weights?: Partial<Record<string, number>>;
|
|
51
|
-
}): Array<{
|
|
52
|
-
candidate: T;
|
|
53
|
-
score: number;
|
|
54
|
-
}>;
|
|
55
|
-
/**
|
|
56
|
-
* NSGA-II crowding distance — secondary sort for ties on the frontier.
|
|
57
|
-
*
|
|
58
|
-
* When the Pareto front collapses to a single point (or many candidates tie
|
|
59
|
-
* on dominance), naive selection picks arbitrarily and the population
|
|
60
|
-
* degenerates over generations. NSGA-II preserves diversity by preferring
|
|
61
|
-
* candidates with more empty space around them on the frontier.
|
|
62
|
-
*
|
|
63
|
-
* Returns an array of `{ candidate, distance }` in the SAME order as the
|
|
64
|
-
* input. Higher distance = more isolated = should be preferred when
|
|
65
|
-
* preserving diversity.
|
|
66
|
-
*/
|
|
67
|
-
declare function crowdingDistance<T>(candidates: T[], objectives: Objective<T>[]): Array<{
|
|
68
|
-
candidate: T;
|
|
69
|
-
distance: number;
|
|
70
|
-
}>;
|
|
71
|
-
/**
|
|
72
|
-
* Pareto frontier with tie-break by crowding distance — the canonical
|
|
73
|
-
* NSGA-II selection step. Returns the frontier sorted by descending crowding
|
|
74
|
-
* distance so callers can `.slice(0, k)` to pick K diverse winners.
|
|
75
|
-
*/
|
|
76
|
-
declare function paretoFrontierWithCrowding<T>(candidates: T[], objectives: Objective<T>[]): Array<{
|
|
77
|
-
candidate: T;
|
|
78
|
-
distance: number;
|
|
79
|
-
}>;
|
|
80
|
-
|
|
81
|
-
export { type Direction as D, type Objective as O, type ParetoResult as P, paretoFrontierWithCrowding as a, crowdingDistance as c, dominates as d, paretoFrontier as p, scalarScore as s };
|
package/dist/perf/index.d.ts
DELETED
|
@@ -1,123 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Journey × axes matrix for infra performance benchmarks.
|
|
3
|
-
*
|
|
4
|
-
* A journey is one measurable user path ("provision.cold", "chat.ttft");
|
|
5
|
-
* axes are free-form scenario dimensions (driver, region, image…). The
|
|
6
|
-
* matrix expansion is pure bookkeeping — running the scenarios and
|
|
7
|
-
* recording metrics is the caller's job. This module complements the
|
|
8
|
-
* judge-panel `BenchmarkRunner` (src/benchmark.ts): that one scores
|
|
9
|
-
* QUALITY via judges, this one structures LATENCY / RELIABILITY runs
|
|
10
|
-
* over flat metric records.
|
|
11
|
-
*/
|
|
12
|
-
/** One measurable user journey (e.g. "provision.cold", "chat.ttft"). */
|
|
13
|
-
interface JourneySpec {
|
|
14
|
-
id: string;
|
|
15
|
-
description: string;
|
|
16
|
-
/** Needs a real LLM call — schedule nightly, not per-PR. */
|
|
17
|
-
requiresLLM: boolean;
|
|
18
|
-
/**
|
|
19
|
-
* Fields that MUST be non-null on a passing record of this journey.
|
|
20
|
-
* A "passing" record missing one is an integrity violation, not a pass.
|
|
21
|
-
*/
|
|
22
|
-
requiredFields: ReadonlyArray<string>;
|
|
23
|
-
/** Numeric floors, e.g. {field: 'event_count', min: 1} for streaming. */
|
|
24
|
-
minimums?: ReadonlyArray<{
|
|
25
|
-
field: string;
|
|
26
|
-
min: number;
|
|
27
|
-
}>;
|
|
28
|
-
/** Per-phase breakdown fields expected non-null (subset of requiredFields semantics, reported separately). */
|
|
29
|
-
phaseFields?: ReadonlyArray<string>;
|
|
30
|
-
}
|
|
31
|
-
interface ScenarioAxes {
|
|
32
|
-
/** e.g. driver: ['docker','firecracker'] — every key is a free-form dimension. */
|
|
33
|
-
[dimension: string]: ReadonlyArray<string>;
|
|
34
|
-
}
|
|
35
|
-
interface PerfScenario {
|
|
36
|
-
/** `${journeyId}|${dim1}=${v1}|${dim2}=${v2}` (dims sorted). */
|
|
37
|
-
key: string;
|
|
38
|
-
journey: JourneySpec;
|
|
39
|
-
axes: Record<string, string>;
|
|
40
|
-
}
|
|
41
|
-
/** Stable scenario key: journey id then `dim=value` pairs in sorted-dim order. */
|
|
42
|
-
declare function scenarioKey(journeyId: string, axes: Record<string, string>): string;
|
|
43
|
-
/** Cartesian expansion; `filter` lets callers drop invalid combos (e.g. firecracker×resume). */
|
|
44
|
-
declare function expandMatrix(journeys: ReadonlyArray<JourneySpec>, axes: ScenarioAxes, filter?: (journeyId: string, combo: Record<string, string>) => boolean): PerfScenario[];
|
|
45
|
-
|
|
46
|
-
/**
|
|
47
|
-
* Record-integrity contracts for perf metric records.
|
|
48
|
-
*
|
|
49
|
-
* A record that claims `pass === true` must actually carry the journey's
|
|
50
|
-
* required measurements — a "passing" provision run with a null
|
|
51
|
-
* `total_ms` is a lying record, not a pass. Failed records are exempt:
|
|
52
|
-
* a run that errored mid-flight legitimately has nulls.
|
|
53
|
-
*/
|
|
54
|
-
|
|
55
|
-
interface IntegrityViolation {
|
|
56
|
-
recordIndex: number;
|
|
57
|
-
journeyId: string;
|
|
58
|
-
field: string;
|
|
59
|
-
reason: 'null-required-field' | 'below-minimum';
|
|
60
|
-
detail: string;
|
|
61
|
-
}
|
|
62
|
-
interface IntegrityResult {
|
|
63
|
-
succeeded: boolean;
|
|
64
|
-
violations: IntegrityViolation[];
|
|
65
|
-
}
|
|
66
|
-
/**
|
|
67
|
-
* Validates flat metric records (Record<string, unknown> with a boolean
|
|
68
|
-
* `pass` field) against their journey contract. Only records with
|
|
69
|
-
* pass === true are checked — a failed record may legitimately have nulls.
|
|
70
|
-
* resolveJourney maps a record to its JourneySpec (or null to skip).
|
|
71
|
-
*/
|
|
72
|
-
declare function checkRecordIntegrity(records: ReadonlyArray<Record<string, unknown>>, resolveJourney: (record: Record<string, unknown>) => JourneySpec | null): IntegrityResult;
|
|
73
|
-
/** Throws an Error listing every violation when the result fails. */
|
|
74
|
-
declare function assertRecordIntegrity(records: ReadonlyArray<Record<string, unknown>>, resolveJourney: (record: Record<string, unknown>) => JourneySpec | null): void;
|
|
75
|
-
|
|
76
|
-
/**
|
|
77
|
-
* Percentile ratchet over perf metric records.
|
|
78
|
-
*
|
|
79
|
-
* `summarizeRecords` folds flat records into per-scenario p50/p90 stats;
|
|
80
|
-
* `gatePerf` compares a current summary against a committed baseline and
|
|
81
|
-
* trips on regressions beyond tolerance. Percentiles use nearest-rank on
|
|
82
|
-
* sorted values. Null / non-numeric metric values are excluded from the
|
|
83
|
-
* stat (n reflects only real samples); a field with zero samples is
|
|
84
|
-
* omitted entirely — no fake zeros.
|
|
85
|
-
*/
|
|
86
|
-
interface PerfStat {
|
|
87
|
-
p50: number;
|
|
88
|
-
p90: number;
|
|
89
|
-
n: number;
|
|
90
|
-
}
|
|
91
|
-
interface PerfBaseline {
|
|
92
|
-
version: 1;
|
|
93
|
-
/** key → metric field → stat. */
|
|
94
|
-
scenarios: Record<string, Record<string, PerfStat>>;
|
|
95
|
-
}
|
|
96
|
-
interface PerfRegression {
|
|
97
|
-
scenarioKey: string;
|
|
98
|
-
field: string;
|
|
99
|
-
baseline: PerfStat;
|
|
100
|
-
current: PerfStat;
|
|
101
|
-
/** percent over baseline p50 / p90, whichever tripped. */
|
|
102
|
-
overBy: {
|
|
103
|
-
p50Pct: number;
|
|
104
|
-
p90Pct: number;
|
|
105
|
-
};
|
|
106
|
-
}
|
|
107
|
-
interface PerfGateResult {
|
|
108
|
-
succeeded: boolean;
|
|
109
|
-
regressions: PerfRegression[];
|
|
110
|
-
/** Negative overBy: strictly better than baseline on both percentiles. */
|
|
111
|
-
improvements: PerfRegression[];
|
|
112
|
-
/** In baseline but absent (or under-sampled, n < minSamples) in current. */
|
|
113
|
-
missingScenarios: string[];
|
|
114
|
-
/** In records but absent from baseline. */
|
|
115
|
-
newScenarios: string[];
|
|
116
|
-
}
|
|
117
|
-
declare function summarizeRecords(records: ReadonlyArray<Record<string, unknown>>, keyOf: (record: Record<string, unknown>) => string | null, metricFields: ReadonlyArray<string>): PerfBaseline;
|
|
118
|
-
declare function gatePerf(current: PerfBaseline, baseline: PerfBaseline, options?: {
|
|
119
|
-
tolerancePct?: number;
|
|
120
|
-
minSamples?: number;
|
|
121
|
-
}): PerfGateResult;
|
|
122
|
-
|
|
123
|
-
export { type IntegrityResult, type IntegrityViolation, type JourneySpec, type PerfBaseline, type PerfGateResult, type PerfRegression, type PerfScenario, type PerfStat, type ScenarioAxes, assertRecordIntegrity, checkRecordIntegrity, expandMatrix, gatePerf, scenarioKey, summarizeRecords };
|
package/dist/perf/index.js
DELETED
|
@@ -1,18 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
assertRecordIntegrity,
|
|
3
|
-
checkRecordIntegrity,
|
|
4
|
-
expandMatrix,
|
|
5
|
-
gatePerf,
|
|
6
|
-
scenarioKey,
|
|
7
|
-
summarizeRecords
|
|
8
|
-
} from "../chunk-STGVSCDH.js";
|
|
9
|
-
import "../chunk-PZ5AY32C.js";
|
|
10
|
-
export {
|
|
11
|
-
assertRecordIntegrity,
|
|
12
|
-
checkRecordIntegrity,
|
|
13
|
-
expandMatrix,
|
|
14
|
-
gatePerf,
|
|
15
|
-
scenarioKey,
|
|
16
|
-
summarizeRecords
|
|
17
|
-
};
|
|
18
|
-
//# sourceMappingURL=index.js.map
|
package/dist/perf/index.js.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
|