@zanii/blackbox 0.0.0-stage → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +109 -2
  3. package/dist/agents/index.d.ts +34 -0
  4. package/dist/agents/index.js +73 -0
  5. package/dist/analysis/detectors.d.ts +36 -0
  6. package/dist/analysis/detectors.js +339 -0
  7. package/dist/analysis/faults.d.ts +9 -0
  8. package/dist/analysis/faults.js +210 -0
  9. package/dist/analysis/index.d.ts +68 -0
  10. package/dist/analysis/index.js +388 -0
  11. package/dist/analysis/landing.d.ts +25 -0
  12. package/dist/analysis/landing.js +225 -0
  13. package/dist/analysis/memory.d.ts +13 -0
  14. package/dist/analysis/memory.js +33 -0
  15. package/dist/analysis/waste.d.ts +29 -0
  16. package/dist/analysis/waste.js +79 -0
  17. package/dist/approvals/index.d.ts +11 -0
  18. package/dist/approvals/index.js +27 -0
  19. package/dist/approvals/warnings.d.ts +2 -0
  20. package/dist/approvals/warnings.js +28 -0
  21. package/dist/attest/index.d.ts +17 -0
  22. package/dist/attest/index.js +106 -0
  23. package/dist/authority/index.d.ts +24 -0
  24. package/dist/authority/index.js +77 -0
  25. package/dist/billing/index.d.ts +99 -0
  26. package/dist/billing/index.js +174 -0
  27. package/dist/cli.d.ts +2 -0
  28. package/dist/cli.js +1019 -0
  29. package/dist/client/index.d.ts +146 -0
  30. package/dist/client/index.js +210 -0
  31. package/dist/compliance/index.d.ts +41 -0
  32. package/dist/compliance/index.js +96 -0
  33. package/dist/cost/index.d.ts +133 -0
  34. package/dist/cost/index.js +293 -0
  35. package/dist/directives/index.d.ts +35 -0
  36. package/dist/directives/index.js +80 -0
  37. package/dist/drills/index.d.ts +43 -0
  38. package/dist/drills/index.js +101 -0
  39. package/dist/duty/index.d.ts +21 -0
  40. package/dist/duty/index.js +68 -0
  41. package/dist/fleet/index.d.ts +141 -0
  42. package/dist/fleet/index.js +454 -0
  43. package/dist/hooks/ai-sdk.d.ts +42 -0
  44. package/dist/hooks/ai-sdk.js +62 -0
  45. package/dist/hooks/claude-agent-sdk.d.ts +14 -0
  46. package/dist/hooks/claude-agent-sdk.js +70 -0
  47. package/dist/hooks/index.d.ts +7 -0
  48. package/dist/hooks/index.js +10 -0
  49. package/dist/hooks/langchain-agent.d.ts +69 -0
  50. package/dist/hooks/langchain-agent.js +163 -0
  51. package/dist/hooks/langchain.d.ts +41 -0
  52. package/dist/hooks/langchain.js +216 -0
  53. package/dist/hooks/langgraph-checkpoint.d.ts +12 -0
  54. package/dist/hooks/langgraph-checkpoint.js +73 -0
  55. package/dist/hooks/memory.d.ts +17 -0
  56. package/dist/hooks/memory.js +64 -0
  57. package/dist/hooks/openai-agents.d.ts +6 -0
  58. package/dist/hooks/openai-agents.js +40 -0
  59. package/dist/hooks/protect.d.ts +7 -0
  60. package/dist/hooks/protect.js +39 -0
  61. package/dist/hooks/providers.d.ts +16 -0
  62. package/dist/hooks/providers.js +149 -0
  63. package/dist/hooks/shared.d.ts +11 -0
  64. package/dist/hooks/shared.js +39 -0
  65. package/dist/index.d.ts +45 -0
  66. package/dist/index.js +47 -0
  67. package/dist/investigate/index.d.ts +66 -0
  68. package/dist/investigate/index.js +119 -0
  69. package/dist/mcp-server/index.d.ts +85 -0
  70. package/dist/mcp-server/index.js +216 -0
  71. package/dist/mcp-wrap/index.d.ts +17 -0
  72. package/dist/mcp-wrap/index.js +170 -0
  73. package/dist/money/index.d.ts +114 -0
  74. package/dist/money/index.js +622 -0
  75. package/dist/occurrence/index.d.ts +108 -0
  76. package/dist/occurrence/index.js +168 -0
  77. package/dist/ocsf/index.d.ts +22 -0
  78. package/dist/ocsf/index.js +168 -0
  79. package/dist/otlp/index.d.ts +24 -0
  80. package/dist/otlp/index.js +143 -0
  81. package/dist/packs/index.d.ts +40 -0
  82. package/dist/packs/index.js +217 -0
  83. package/dist/policy/delta.d.ts +11 -0
  84. package/dist/policy/delta.js +39 -0
  85. package/dist/policy/drafts.d.ts +34 -0
  86. package/dist/policy/drafts.js +129 -0
  87. package/dist/policy/index.d.ts +47 -0
  88. package/dist/policy/index.js +154 -0
  89. package/dist/precog/index.d.ts +96 -0
  90. package/dist/precog/index.js +167 -0
  91. package/dist/precog/intervention.d.ts +22 -0
  92. package/dist/precog/intervention.js +44 -0
  93. package/dist/precog/normal.d.ts +31 -0
  94. package/dist/precog/normal.js +89 -0
  95. package/dist/preflight/index.d.ts +11 -0
  96. package/dist/preflight/index.js +19 -0
  97. package/dist/ratings/index.d.ts +21 -0
  98. package/dist/ratings/index.js +48 -0
  99. package/dist/reconcile/claude-code.d.ts +19 -0
  100. package/dist/reconcile/claude-code.js +220 -0
  101. package/dist/reconcile/codex.d.ts +5 -0
  102. package/dist/reconcile/codex.js +191 -0
  103. package/dist/reconcile/index.d.ts +19 -0
  104. package/dist/reconcile/index.js +50 -0
  105. package/dist/reconcile/record.d.ts +49 -0
  106. package/dist/reconcile/record.js +225 -0
  107. package/dist/reconcile/shared.d.ts +65 -0
  108. package/dist/reconcile/shared.js +113 -0
  109. package/dist/replay/index.d.ts +11 -0
  110. package/dist/replay/index.js +64 -0
  111. package/dist/replay/repair.d.ts +10 -0
  112. package/dist/replay/repair.js +62 -0
  113. package/dist/session/drain.d.ts +13 -0
  114. package/dist/session/drain.js +35 -0
  115. package/dist/session/index.d.ts +256 -0
  116. package/dist/session/index.js +658 -0
  117. package/dist/undo/index.d.ts +45 -0
  118. package/dist/undo/index.js +212 -0
  119. package/dist/verify/anchor.d.ts +54 -0
  120. package/dist/verify/anchor.js +77 -0
  121. package/dist/verify/chain.d.ts +27 -0
  122. package/dist/verify/chain.js +105 -0
  123. package/dist/verify/envelope.d.ts +28 -0
  124. package/dist/verify/envelope.js +55 -0
  125. package/dist/verify/index.d.ts +3 -0
  126. package/dist/verify/index.js +3 -0
  127. package/dist/version.d.ts +1 -0
  128. package/dist/version.js +2 -0
  129. package/dist/weather/index.d.ts +24 -0
  130. package/dist/weather/index.js +45 -0
  131. package/dist/workspace-receipt/index.d.ts +15 -0
  132. package/dist/workspace-receipt/index.js +121 -0
  133. package/package.json +56 -3
@@ -0,0 +1,96 @@
1
+ import type { Finding } from "../analysis/index.ts";
2
+ import type { LoadedPrices } from "../cost/index.ts";
3
+ import type { Bodies } from "../reconcile/record.ts";
4
+ export interface PrecogModel {
5
+ version: string;
6
+ note?: string;
7
+ calls: number;
8
+ features: string[];
9
+ mean: number[];
10
+ std: number[];
11
+ weights: number[];
12
+ bias: number;
13
+ alert_ppm: number;
14
+ }
15
+ export declare const FEATURES: string[];
16
+ /** The record up to and including the `calls`-th llm.response (all of it if it has fewer). */
17
+ export declare function prefixOf(lines: readonly string[], calls: number): {
18
+ prefix: string[];
19
+ seen: number;
20
+ };
21
+ /** spec/precog.md §1, over a prefix. PRECOG_RISK findings are left out. */
22
+ export declare function features(prefix: readonly string[], bodies: Bodies, prices: LoadedPrices): number[];
23
+ export declare function predict(model: PrecogModel, x: readonly number[]): {
24
+ risk_ppm: number;
25
+ level: string;
26
+ };
27
+ /** The live alert (spec/precog.md §3): one PRECOG_RISK once the record reaches `calls` model calls. */
28
+ export declare function precogFindings(lines: readonly string[], bodies: Bodies, prices: LoadedPrices, model: PrecogModel): Finding[];
29
+ /** Batch gradient descent on the log loss, deterministic (spec/precog.md §4). */
30
+ export declare function train(examples: ReadonlyArray<{
31
+ x: readonly number[];
32
+ y: number;
33
+ }>, prior: PrecogModel, options?: {
34
+ iterations?: number;
35
+ rate?: number;
36
+ l2?: number;
37
+ }): PrecogModel;
38
+ /** Recall at ≤ 5% false alarms (spec/precog.md §5). */
39
+ export declare function calibration(scores: ReadonlyArray<{
40
+ risk_ppm: number;
41
+ y: number;
42
+ }>): {
43
+ sessions: number;
44
+ positives: number;
45
+ negatives: number;
46
+ threshold_ppm: null;
47
+ recall_ppm: null;
48
+ false_alarm_ppm: null;
49
+ } | {
50
+ sessions: number;
51
+ positives: number;
52
+ negatives: number;
53
+ threshold_ppm: number;
54
+ recall_ppm: number;
55
+ false_alarm_ppm: number;
56
+ };
57
+ /** spec/precog.md §6: which side of a reproducible split a session falls on, by its id's hash. */
58
+ export declare function splitOf(sessionId: string, testPercent: number): "train" | "test";
59
+ export declare const TARGET: {
60
+ recall_ppm: number;
61
+ false_alarm_ppm: number;
62
+ };
63
+ export declare const MIN_EACH = 30;
64
+ /** spec/precog.md §6: the model's calibration on the test side, against the roadmap target. */
65
+ export declare function precogReport(model: PrecogModel, examples: ReadonlyArray<{
66
+ session_id: string;
67
+ x: readonly number[];
68
+ y: number;
69
+ }>, testPercent: number): {
70
+ v: 1;
71
+ version: string;
72
+ test_percent: number;
73
+ in_sample: boolean;
74
+ train_sessions: number;
75
+ calibration: {
76
+ sessions: number;
77
+ positives: number;
78
+ negatives: number;
79
+ threshold_ppm: null;
80
+ recall_ppm: null;
81
+ false_alarm_ppm: null;
82
+ } | {
83
+ sessions: number;
84
+ positives: number;
85
+ negatives: number;
86
+ threshold_ppm: number;
87
+ recall_ppm: number;
88
+ false_alarm_ppm: number;
89
+ };
90
+ target: {
91
+ recall_ppm: number;
92
+ false_alarm_ppm: number;
93
+ };
94
+ min_each: number;
95
+ verdict: string;
96
+ };
@@ -0,0 +1,167 @@
1
+ // Precog v1 (spec/precog.md): early-trajectory features, a logistic model, training, calibration.
2
+ // Mirrors sdks/python/src/zanii_blackbox/precog.py.
3
+ import { createHash } from "node:crypto";
4
+ import { sessionSummary } from "../fleet/index.js";
5
+ export const FEATURES = [
6
+ "model_calls",
7
+ "tool_calls",
8
+ "distinct_tools",
9
+ "log_spend",
10
+ "warnings",
11
+ "cautions",
12
+ "near_misses",
13
+ "blocked",
14
+ ];
15
+ /** The record up to and including the `calls`-th llm.response (all of it if it has fewer). */
16
+ export function prefixOf(lines, calls) {
17
+ let seen = 0;
18
+ for (const [i, line] of lines.entries())
19
+ if (JSON.parse(line).kind === "llm.response" && ++seen === calls)
20
+ return { prefix: lines.slice(0, i + 1), seen };
21
+ return { prefix: [...lines], seen };
22
+ }
23
+ /** spec/precog.md §1, over a prefix. PRECOG_RISK findings are left out. */
24
+ export function features(prefix, bodies, prices) {
25
+ const own = prefix.filter((l) => {
26
+ const e = JSON.parse(l);
27
+ return !(e.kind === "finding" && e.meta.code === "PRECOG_RISK");
28
+ });
29
+ const s = sessionSummary(own, bodies, prices);
30
+ return [
31
+ s.model_calls,
32
+ s.tool_calls,
33
+ Object.keys(s.tools).length,
34
+ Math.log(1 + s.spend_micro_usd),
35
+ s.findings.warning,
36
+ s.findings.caution,
37
+ s.near_misses,
38
+ s.blocked_by ? 1 : 0,
39
+ ];
40
+ }
41
+ const sigmoid = (z) => 1 / (1 + Math.exp(-Math.min(500, Math.max(-500, z))));
42
+ export function predict(model, x) {
43
+ let z = model.bias;
44
+ for (const [i, v] of x.entries())
45
+ z +=
46
+ model.weights[i] * ((v - model.mean[i]) / model.std[i]);
47
+ const risk = Math.floor(sigmoid(z) * 1_000_000 + 0.5);
48
+ const level = risk >= model.alert_ppm ? "high" : risk >= Math.floor(model.alert_ppm / 2) ? "elevated" : "low";
49
+ return { risk_ppm: risk, level };
50
+ }
51
+ /** The live alert (spec/precog.md §3): one PRECOG_RISK once the record reaches `calls` model calls. */
52
+ export function precogFindings(lines, bodies, prices, model) {
53
+ const { prefix, seen } = prefixOf(lines, model.calls);
54
+ if (seen < model.calls)
55
+ return [];
56
+ const p = predict(model, features(prefix, bodies, prices));
57
+ return p.level === "high"
58
+ ? [
59
+ {
60
+ code: "PRECOG_RISK",
61
+ source: "precog",
62
+ severity: "caution",
63
+ ref: { calls: model.calls, risk_ppm: p.risk_ppm },
64
+ },
65
+ ]
66
+ : [];
67
+ }
68
+ /** Batch gradient descent on the log loss, deterministic (spec/precog.md §4). */
69
+ export function train(examples, prior, options = {}) {
70
+ const { iterations = 500, rate = 0.1, l2 = 0.01 } = options;
71
+ const n = examples.length;
72
+ const d = FEATURES.length;
73
+ const mean = [];
74
+ const std = [];
75
+ for (let j = 0; j < d; j++) {
76
+ let s = 0;
77
+ for (const e of examples)
78
+ s += e.x[j];
79
+ const m = n ? s / n : 0;
80
+ let q = 0;
81
+ for (const e of examples)
82
+ q += (e.x[j] - m) ** 2;
83
+ const v = n ? Math.sqrt(q / n) : 0;
84
+ mean.push(m);
85
+ std.push(v === 0 ? 1 : v);
86
+ }
87
+ const xs = examples.map((e) => e.x.map((v, j) => (v - mean[j]) / std[j]));
88
+ const w = new Array(d).fill(0);
89
+ let b = 0;
90
+ for (let it = 0; it < iterations && n > 0; it++) {
91
+ const gw = new Array(d).fill(0);
92
+ let gb = 0;
93
+ for (const [k, e] of examples.entries()) {
94
+ const row = xs[k];
95
+ let z = b;
96
+ for (let j = 0; j < d; j++)
97
+ z += w[j] * row[j];
98
+ const err = sigmoid(z) - e.y;
99
+ for (let j = 0; j < d; j++)
100
+ gw[j] = gw[j] + err * row[j];
101
+ gb += err;
102
+ }
103
+ for (let j = 0; j < d; j++)
104
+ w[j] = w[j] - rate * (gw[j] / n + l2 * w[j]);
105
+ b -= rate * (gb / n);
106
+ }
107
+ return {
108
+ version: `trained-${n}`,
109
+ calls: prior.calls,
110
+ features: [...FEATURES],
111
+ mean,
112
+ std,
113
+ weights: w,
114
+ bias: b,
115
+ alert_ppm: prior.alert_ppm,
116
+ };
117
+ }
118
+ /** Recall at ≤ 5% false alarms (spec/precog.md §5). */
119
+ export function calibration(scores) {
120
+ const neg = scores
121
+ .filter((s) => s.y === 0)
122
+ .map((s) => s.risk_ppm)
123
+ .sort((a, b) => a - b);
124
+ const pos = scores.filter((s) => s.y === 1).map((s) => s.risk_ppm);
125
+ const base = { sessions: scores.length, positives: pos.length, negatives: neg.length };
126
+ if (neg.length === 0 || pos.length === 0)
127
+ return { ...base, threshold_ppm: null, recall_ppm: null, false_alarm_ppm: null };
128
+ const threshold = neg[Math.max(0, Math.ceil((95 * neg.length) / 100) - 1)];
129
+ const ppm = (k, of) => Math.floor((k * 1_000_000) / of);
130
+ return {
131
+ ...base,
132
+ threshold_ppm: threshold,
133
+ recall_ppm: ppm(pos.filter((v) => v > threshold).length, pos.length),
134
+ false_alarm_ppm: ppm(neg.filter((v) => v > threshold).length, neg.length),
135
+ };
136
+ }
137
+ /** spec/precog.md §6: which side of a reproducible split a session falls on, by its id's hash. */
138
+ export function splitOf(sessionId, testPercent) {
139
+ const h = createHash("sha256").update(sessionId, "utf8").digest();
140
+ return h.readUInt32BE(0) % 100 < testPercent ? "test" : "train";
141
+ }
142
+ export const TARGET = { recall_ppm: 700_000, false_alarm_ppm: 50_000 };
143
+ export const MIN_EACH = 30;
144
+ /** spec/precog.md §6: the model's calibration on the test side, against the roadmap target. */
145
+ export function precogReport(model, examples, testPercent) {
146
+ const test = testPercent === 0
147
+ ? examples
148
+ : examples.filter((e) => splitOf(e.session_id, testPercent) === "test");
149
+ const cal = calibration(test.map((e) => ({ risk_ppm: predict(model, e.x).risk_ppm, y: e.y })));
150
+ const verdict = cal.positives < MIN_EACH || cal.negatives < MIN_EACH
151
+ ? "not enough data"
152
+ : cal.recall_ppm >= TARGET.recall_ppm &&
153
+ cal.false_alarm_ppm <= TARGET.false_alarm_ppm
154
+ ? "meets target"
155
+ : "misses target";
156
+ return {
157
+ v: 1,
158
+ version: model.version,
159
+ test_percent: testPercent,
160
+ in_sample: testPercent === 0,
161
+ train_sessions: examples.length - (testPercent === 0 ? 0 : test.length),
162
+ calibration: cal,
163
+ target: TARGET,
164
+ min_each: MIN_EACH,
165
+ verdict,
166
+ };
167
+ }
@@ -0,0 +1,22 @@
1
+ export type Outcome = "success" | "failure";
2
+ export interface InterventionEvidence {
3
+ pairs: number;
4
+ recovered: number;
5
+ not_recovered: number;
6
+ disrupted: number;
7
+ kept: number;
8
+ r_milli: number | null;
9
+ d_milli: number | null;
10
+ /** 95 % Wilson intervals. */
11
+ r_ci95_milli: [number, number] | null;
12
+ d_ci95_milli: [number, number] | null;
13
+ /** Act only above this failure chance; null without evidence on both sides. */
14
+ break_even_milli: number | null;
15
+ }
16
+ /** The evidence from (source outcome, fork outcome) pairs. */
17
+ export declare function interventionEvidence(pairs: ReadonlyArray<{
18
+ before: Outcome;
19
+ after: Outcome;
20
+ }>): InterventionEvidence;
21
+ /** The expected gain, in thousandths of a run, of acting on a run with this failure chance. */
22
+ export declare function interventionGain(e: InterventionEvidence, failureMilli: number): number | null;
@@ -0,0 +1,44 @@
1
+ // Intervention evidence (spec/precog.md §9, Stage 2 R6): is restarting a run worth it? From replay
2
+ // forks with known outcomes: recovery r (a failed run's fork succeeded) and disruption d (a good
3
+ // run's fork failed). Acting on a run with failure chance p gains p·r − (1−p)·d, so it pays only
4
+ // when p > d / (r + d) ("The Intervention Paradox", arXiv 2602.03338). Mirrors precog/intervention.py.
5
+ // Rates are in thousandths (integers, so the record stays exact).
6
+ const milli = (x) => Math.round(x * 1000);
7
+ function wilson(k, n) {
8
+ if (n === 0)
9
+ return null;
10
+ const z = 1.959963984540054;
11
+ const p = k / n;
12
+ const den = 1 + (z * z) / n;
13
+ const mid = (p + (z * z) / (2 * n)) / den;
14
+ const half = (z * Math.sqrt((p * (1 - p)) / n + (z * z) / (4 * n * n))) / den;
15
+ return [milli(Math.max(0, mid - half)), milli(Math.min(1, mid + half))];
16
+ }
17
+ /** The evidence from (source outcome, fork outcome) pairs. */
18
+ export function interventionEvidence(pairs) {
19
+ const failed = pairs.filter((p) => p.before === "failure");
20
+ const good = pairs.filter((p) => p.before === "success");
21
+ const recovered = failed.filter((p) => p.after === "success").length;
22
+ const disrupted = good.filter((p) => p.after === "failure").length;
23
+ const r = failed.length ? recovered / failed.length : null;
24
+ const d = good.length ? disrupted / good.length : null;
25
+ return {
26
+ pairs: pairs.length,
27
+ recovered,
28
+ not_recovered: failed.length - recovered,
29
+ disrupted,
30
+ kept: good.length - disrupted,
31
+ r_milli: r === null ? null : milli(r),
32
+ d_milli: d === null ? null : milli(d),
33
+ r_ci95_milli: wilson(recovered, failed.length),
34
+ d_ci95_milli: wilson(disrupted, good.length),
35
+ break_even_milli: r === null || d === null || r + d === 0 ? null : milli(d / (r + d)),
36
+ };
37
+ }
38
+ /** The expected gain, in thousandths of a run, of acting on a run with this failure chance. */
39
+ export function interventionGain(e, failureMilli) {
40
+ if (e.r_milli === null || e.d_milli === null)
41
+ return null;
42
+ const p = failureMilli / 1000;
43
+ return milli(p * (e.r_milli / 1000) - (1 - p) * (e.d_milli / 1000));
44
+ }
@@ -0,0 +1,31 @@
1
+ export interface NormalModel {
2
+ version: 1;
3
+ runs: number;
4
+ /** "a\tb" → how often b followed a, in the successful runs. */
5
+ transitions: Record<string, number>;
6
+ /** a → how often anything followed it. */
7
+ out: Record<string, number>;
8
+ symbols: number;
9
+ }
10
+ export interface NormalScore {
11
+ steps: number;
12
+ /** Mean surprise of each step, −ln p, in thousandths (integers, so the record stays exact). */
13
+ surprise_milli: number;
14
+ unseen_milli: number;
15
+ loop_milli: number;
16
+ }
17
+ /** A run's steps as symbols: each model call, and each tool by name (MCP or SDK-reported). */
18
+ export declare function activities(lines: readonly string[]): string[];
19
+ /** The map of the given successful runs (each a session's lines). */
20
+ export declare function buildNormal(runs: ReadonlyArray<readonly string[]>): NormalModel;
21
+ /** How far a run strays from the normal. Add-half smoothing; one more symbol for the unseen. */
22
+ export declare function normalScore(model: NormalModel, lines: readonly string[]): NormalScore;
23
+ /** The alarm threshold from the scores of held-out successful runs: flag a score above it. With
24
+ * probability ≥ `confidence`, at most `alpha` of normal runs are flagged (an order statistic with a
25
+ * Clopper-Pearson-style binomial bound). The lowest such threshold, for the most recall; null when
26
+ * there are too few runs for the guarantee. */
27
+ export declare function normalThreshold(scores: readonly number[], alpha?: number, confidence?: number): {
28
+ threshold: number;
29
+ above: number;
30
+ needed: number;
31
+ } | null;
@@ -0,0 +1,89 @@
1
+ // Each customer's normal (spec/precog.md §8, Stage 2 R5): a map of the paths a customer's own
2
+ // successful runs take, learned without failure labels, and a score for how far a run strays from
3
+ // it. The alarm threshold comes with a guarantee: with the chosen confidence, at most `alpha` of
4
+ // normal runs are flagged. Mirrors precog/normal.py.
5
+ /** A run's steps as symbols: each model call, and each tool by name (MCP or SDK-reported). */
6
+ export function activities(lines) {
7
+ const out = [];
8
+ for (const line of lines) {
9
+ const e = JSON.parse(line);
10
+ const m = e.meta;
11
+ if (e.kind === "llm.request")
12
+ out.push("model");
13
+ else if (e.kind === "tool.call" && m.method === "tools/call")
14
+ out.push(`tool:${String(m.server ?? "")}__${String(m.tool ?? "")}`);
15
+ else if (e.kind === "sdk.event" && m.type === "tool.call" && typeof m.name === "string")
16
+ out.push(`tool:${m.name}`);
17
+ }
18
+ return out;
19
+ }
20
+ /** The map of the given successful runs (each a session's lines). */
21
+ export function buildNormal(runs) {
22
+ const transitions = {};
23
+ const out = {};
24
+ const symbols = new Set();
25
+ for (const lines of runs) {
26
+ const acts = activities(lines);
27
+ acts.forEach((b, i) => {
28
+ const a = i === 0 ? "start" : acts[i - 1];
29
+ const key = `${a}\t${b}`;
30
+ transitions[key] = (transitions[key] ?? 0) + 1;
31
+ out[a] = (out[a] ?? 0) + 1;
32
+ symbols.add(b);
33
+ });
34
+ }
35
+ return { version: 1, runs: runs.length, transitions, out, symbols: symbols.size };
36
+ }
37
+ /** How far a run strays from the normal. Add-half smoothing; one more symbol for the unseen. */
38
+ export function normalScore(model, lines) {
39
+ const acts = activities(lines);
40
+ const k = model.symbols + 1;
41
+ let surprise = 0;
42
+ let unseen = 0;
43
+ acts.forEach((b, i) => {
44
+ const a = i === 0 ? "start" : acts[i - 1];
45
+ const n = model.transitions[`${a}\t${b}`] ?? 0;
46
+ if (n === 0)
47
+ unseen++;
48
+ surprise += -Math.log((n + 0.5) / ((model.out[a] ?? 0) + 0.5 * k));
49
+ });
50
+ const bigrams = acts.slice(1).map((b, i) => `${acts[i]}\t${b}`);
51
+ const loops = bigrams.filter((g, i) => bigrams.indexOf(g) < i).length;
52
+ const per = (x, n) => (n ? Math.round((1000 * x) / n) : 0);
53
+ return {
54
+ steps: acts.length,
55
+ surprise_milli: per(surprise, acts.length),
56
+ unseen_milli: per(unseen, acts.length),
57
+ loop_milli: per(loops, bigrams.length),
58
+ };
59
+ }
60
+ /** P(X ≥ j) for X ~ Binomial(n, p), from the pmf in log space (no underflow for large n). */
61
+ function binomialTail(n, p, j) {
62
+ let logPmf = n * Math.log(1 - p); // ln P(X = 0)
63
+ let below = 0;
64
+ for (let x = 0; x < j; x++) {
65
+ below += Math.exp(logPmf);
66
+ logPmf += Math.log((n - x) / (x + 1)) + Math.log(p / (1 - p));
67
+ }
68
+ return Math.max(0, 1 - below);
69
+ }
70
+ /** The alarm threshold from the scores of held-out successful runs: flag a score above it. With
71
+ * probability ≥ `confidence`, at most `alpha` of normal runs are flagged (an order statistic with a
72
+ * Clopper-Pearson-style binomial bound). The lowest such threshold, for the most recall; null when
73
+ * there are too few runs for the guarantee. */
74
+ export function normalThreshold(scores, alpha = 0.05, confidence = 0.95) {
75
+ const n = scores.length;
76
+ const sorted = [...scores].sort((a, b) => b - a); // highest first
77
+ const needed = Math.ceil(Math.log(1 - confidence) / Math.log(1 - alpha));
78
+ // j calibration runs may score above the threshold: the largest j the guarantee allows
79
+ let best = -1;
80
+ for (let j = 0; j < n; j++) {
81
+ if (binomialTail(n, alpha, j + 1) >= confidence)
82
+ best = j;
83
+ else
84
+ break;
85
+ }
86
+ if (best < 0)
87
+ return null;
88
+ return { threshold: sorted[best], above: best, needed };
89
+ }
@@ -0,0 +1,11 @@
1
+ export interface PreflightFinding {
2
+ code: "PREFLIGHT_DEGRADED";
3
+ source: "preflight";
4
+ severity: "advisory";
5
+ /** `degraded`: the optional items that failed, comma-separated (a ref holds scalars only). */
6
+ ref: {
7
+ seq: number;
8
+ degraded: string;
9
+ };
10
+ }
11
+ export declare function preflightFindings(lines: readonly string[]): PreflightFinding[];
@@ -0,0 +1,19 @@
1
+ // Pre-flight (spec/preflight.md §3): a session that left with equipment missing. Pure; mirrors
2
+ // sdks/python/src/zanii_blackbox/preflight.py; pinned by spec/vectors/preflight.json.
3
+ export function preflightFindings(lines) {
4
+ const first = lines[0];
5
+ if (first === undefined)
6
+ return [];
7
+ const e = JSON.parse(first);
8
+ const degraded = e.kind === "session.open" ? e.meta.preflight?.degraded : undefined;
9
+ if (!Array.isArray(degraded) || degraded.length === 0)
10
+ return [];
11
+ return [
12
+ {
13
+ code: "PREFLIGHT_DEGRADED",
14
+ source: "preflight",
15
+ severity: "advisory",
16
+ ref: { seq: e.seq, degraded: degraded.map(String).join(",") },
17
+ },
18
+ ];
19
+ }
@@ -0,0 +1,21 @@
1
+ import { type LoadedPrices } from "../cost/index.ts";
2
+ import { type AirworthinessCriteria, airworthiness, type Summary } from "../fleet/index.ts";
3
+ export type Rating = ReturnType<typeof airworthiness> & {
4
+ model: string;
5
+ };
6
+ export interface TypeUnratedFinding {
7
+ code: "TYPE_UNRATED";
8
+ source: "type_rating";
9
+ severity: "advisory" | "warning";
10
+ ref: {
11
+ seq: number;
12
+ model: string;
13
+ };
14
+ }
15
+ /** spec/type-ratings.md §1. `items` newest first. */
16
+ export declare function typeRatings(items: ReadonlyArray<{
17
+ summary: Summary;
18
+ models: readonly string[];
19
+ }>, criteria?: AirworthinessCriteria): Rating[];
20
+ /** spec/type-ratings.md §2. `severity` is "warning" when the server requires type ratings. */
21
+ export declare function typeUnrated(lines: readonly string[], prices: LoadedPrices, ratings: readonly Rating[], severity?: "advisory" | "warning"): TypeUnratedFinding[];
@@ -0,0 +1,48 @@
1
+ // Type ratings (spec/type-ratings.md): airworthiness per label and model, and flying unrated.
2
+ // Pure; mirrors sdks/python/src/zanii_blackbox/ratings.py; pinned by spec/vectors/type-ratings.json.
3
+ import { sessionCost } from "../cost/index.js";
4
+ import { airworthiness, DEFAULT_AIRWORTHINESS, } from "../fleet/index.js";
5
+ /** spec/type-ratings.md §1. `items` newest first. */
6
+ export function typeRatings(items, criteria = DEFAULT_AIRWORTHINESS) {
7
+ const pairs = new Map();
8
+ for (const i of items)
9
+ if (i.summary.closed)
10
+ for (const model of new Set(i.models)) {
11
+ const label = i.summary.label ?? "(no label)";
12
+ pairs.set(JSON.stringify([label, model]), { label, model });
13
+ }
14
+ const cmp = (a, b) => (a < b ? -1 : a > b ? 1 : 0);
15
+ return [...pairs.values()]
16
+ .sort((a, b) => cmp(a.label, b.label) || cmp(a.model, b.model))
17
+ .map(({ label, model }) => ({
18
+ ...airworthiness(label, items.filter((i) => i.models.includes(model)).map((i) => i.summary), criteria),
19
+ model,
20
+ }));
21
+ }
22
+ /** spec/type-ratings.md §2. `severity` is "warning" when the server requires type ratings. */
23
+ export function typeUnrated(lines, prices, ratings, severity = "advisory") {
24
+ const open = lines[0]
25
+ ? JSON.parse(lines[0])
26
+ : null;
27
+ if (open?.kind !== "session.open")
28
+ return [];
29
+ const label = typeof open.meta.label === "string" ? open.meta.label : "(no label)";
30
+ const mine = ratings.filter((r) => r.label === label);
31
+ if (!mine.some((r) => r.verdict === "airworthy"))
32
+ return [];
33
+ const out = [];
34
+ const seen = new Set();
35
+ for (const c of sessionCost(lines, prices).calls) {
36
+ if (!c.model || seen.has(c.model))
37
+ continue;
38
+ seen.add(c.model);
39
+ if (!mine.some((r) => r.model === c.model && r.verdict === "airworthy"))
40
+ out.push({
41
+ code: "TYPE_UNRATED",
42
+ source: "type_rating",
43
+ severity,
44
+ ref: { seq: c.seq, model: c.model },
45
+ });
46
+ }
47
+ return out;
48
+ }
@@ -0,0 +1,19 @@
1
+ import type { Call, Receipt } from "./record.ts";
2
+ import { type Finding, type LocalLog } from "./shared.ts";
3
+ export declare const CLAUDE_CODE_TYPES: ReadonlySet<string>;
4
+ export interface Outcome {
5
+ findings: Finding[];
6
+ expected: Call[];
7
+ matched: {
8
+ calls: number;
9
+ tool_results: number;
10
+ };
11
+ notes: string[];
12
+ agentSessions: string[];
13
+ /** When receipts are required (§7). */
14
+ workspace?: {
15
+ receipts: number;
16
+ changed_by: string[];
17
+ };
18
+ }
19
+ export declare function reconcileClaudeCode(calls: Call[], log: LocalLog, agentSessions: string[], allReceipts?: Receipt[], requireReceipts?: boolean): Outcome;