@zanii/blackbox 0.0.0-stage → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +109 -2
- package/dist/agents/index.d.ts +34 -0
- package/dist/agents/index.js +73 -0
- package/dist/analysis/detectors.d.ts +36 -0
- package/dist/analysis/detectors.js +339 -0
- package/dist/analysis/faults.d.ts +9 -0
- package/dist/analysis/faults.js +250 -0
- package/dist/analysis/index.d.ts +68 -0
- package/dist/analysis/index.js +388 -0
- package/dist/analysis/landing.d.ts +25 -0
- package/dist/analysis/landing.js +225 -0
- package/dist/analysis/memory.d.ts +13 -0
- package/dist/analysis/memory.js +33 -0
- package/dist/analysis/waste.d.ts +29 -0
- package/dist/analysis/waste.js +79 -0
- package/dist/approvals/index.d.ts +11 -0
- package/dist/approvals/index.js +27 -0
- package/dist/approvals/warnings.d.ts +2 -0
- package/dist/approvals/warnings.js +28 -0
- package/dist/attest/index.d.ts +17 -0
- package/dist/attest/index.js +106 -0
- package/dist/authority/index.d.ts +24 -0
- package/dist/authority/index.js +77 -0
- package/dist/billing/index.d.ts +99 -0
- package/dist/billing/index.js +174 -0
- package/dist/cli.d.ts +2 -0
- package/dist/cli.js +1057 -0
- package/dist/client/index.d.ts +146 -0
- package/dist/client/index.js +210 -0
- package/dist/compliance/index.d.ts +41 -0
- package/dist/compliance/index.js +96 -0
- package/dist/cost/index.d.ts +133 -0
- package/dist/cost/index.js +293 -0
- package/dist/data/index.d.ts +191 -0
- package/dist/data/index.js +762 -0
- package/dist/directives/index.d.ts +35 -0
- package/dist/directives/index.js +80 -0
- package/dist/drills/index.d.ts +43 -0
- package/dist/drills/index.js +101 -0
- package/dist/duty/index.d.ts +21 -0
- package/dist/duty/index.js +68 -0
- package/dist/fleet/index.d.ts +141 -0
- package/dist/fleet/index.js +454 -0
- package/dist/hooks/ai-sdk.d.ts +42 -0
- package/dist/hooks/ai-sdk.js +62 -0
- package/dist/hooks/claude-agent-sdk.d.ts +14 -0
- package/dist/hooks/claude-agent-sdk.js +70 -0
- package/dist/hooks/index.d.ts +7 -0
- package/dist/hooks/index.js +10 -0
- package/dist/hooks/langchain-agent.d.ts +69 -0
- package/dist/hooks/langchain-agent.js +163 -0
- package/dist/hooks/langchain.d.ts +41 -0
- package/dist/hooks/langchain.js +216 -0
- package/dist/hooks/langgraph-checkpoint.d.ts +12 -0
- package/dist/hooks/langgraph-checkpoint.js +73 -0
- package/dist/hooks/memory.d.ts +17 -0
- package/dist/hooks/memory.js +64 -0
- package/dist/hooks/openai-agents.d.ts +6 -0
- package/dist/hooks/openai-agents.js +40 -0
- package/dist/hooks/protect.d.ts +7 -0
- package/dist/hooks/protect.js +39 -0
- package/dist/hooks/providers.d.ts +16 -0
- package/dist/hooks/providers.js +149 -0
- package/dist/hooks/shared.d.ts +11 -0
- package/dist/hooks/shared.js +39 -0
- package/dist/index.d.ts +46 -0
- package/dist/index.js +48 -0
- package/dist/investigate/index.d.ts +66 -0
- package/dist/investigate/index.js +119 -0
- package/dist/mcp-server/index.d.ts +85 -0
- package/dist/mcp-server/index.js +216 -0
- package/dist/mcp-wrap/index.d.ts +17 -0
- package/dist/mcp-wrap/index.js +170 -0
- package/dist/money/index.d.ts +114 -0
- package/dist/money/index.js +622 -0
- package/dist/occurrence/index.d.ts +108 -0
- package/dist/occurrence/index.js +168 -0
- package/dist/ocsf/index.d.ts +22 -0
- package/dist/ocsf/index.js +168 -0
- package/dist/otlp/index.d.ts +24 -0
- package/dist/otlp/index.js +143 -0
- package/dist/packs/index.d.ts +48 -0
- package/dist/packs/index.js +343 -0
- package/dist/policy/delta.d.ts +11 -0
- package/dist/policy/delta.js +39 -0
- package/dist/policy/drafts.d.ts +34 -0
- package/dist/policy/drafts.js +129 -0
- package/dist/policy/index.d.ts +47 -0
- package/dist/policy/index.js +154 -0
- package/dist/precog/index.d.ts +96 -0
- package/dist/precog/index.js +167 -0
- package/dist/precog/intervention.d.ts +22 -0
- package/dist/precog/intervention.js +44 -0
- package/dist/precog/normal.d.ts +31 -0
- package/dist/precog/normal.js +89 -0
- package/dist/preflight/index.d.ts +11 -0
- package/dist/preflight/index.js +19 -0
- package/dist/ratings/index.d.ts +21 -0
- package/dist/ratings/index.js +48 -0
- package/dist/reconcile/claude-code.d.ts +19 -0
- package/dist/reconcile/claude-code.js +220 -0
- package/dist/reconcile/codex.d.ts +5 -0
- package/dist/reconcile/codex.js +191 -0
- package/dist/reconcile/index.d.ts +19 -0
- package/dist/reconcile/index.js +50 -0
- package/dist/reconcile/record.d.ts +49 -0
- package/dist/reconcile/record.js +225 -0
- package/dist/reconcile/shared.d.ts +65 -0
- package/dist/reconcile/shared.js +113 -0
- package/dist/replay/index.d.ts +11 -0
- package/dist/replay/index.js +64 -0
- package/dist/replay/repair.d.ts +10 -0
- package/dist/replay/repair.js +62 -0
- package/dist/session/drain.d.ts +13 -0
- package/dist/session/drain.js +35 -0
- package/dist/session/index.d.ts +275 -0
- package/dist/session/index.js +681 -0
- package/dist/undo/index.d.ts +45 -0
- package/dist/undo/index.js +212 -0
- package/dist/verify/anchor.d.ts +54 -0
- package/dist/verify/anchor.js +77 -0
- package/dist/verify/chain.d.ts +27 -0
- package/dist/verify/chain.js +105 -0
- package/dist/verify/envelope.d.ts +28 -0
- package/dist/verify/envelope.js +55 -0
- package/dist/verify/index.d.ts +3 -0
- package/dist/verify/index.js +3 -0
- package/dist/version.d.ts +1 -0
- package/dist/version.js +2 -0
- package/dist/weather/index.d.ts +24 -0
- package/dist/weather/index.js +45 -0
- package/dist/workspace-receipt/index.d.ts +15 -0
- package/dist/workspace-receipt/index.js +121 -0
- package/package.json +56 -3
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
// Tool policy (spec/policy.md): rules checked in order, the first match decides. Mirrors policy.py.
|
|
2
|
+
import { callsOf } from "../reconcile/record.js";
|
|
3
|
+
import { canonical } from "../reconcile/shared.js";
|
|
4
|
+
import { checkCompensations } from "../undo/index.js";
|
|
5
|
+
const MAX_ARGS = 64 * 1024;
|
|
6
|
+
/** Escapes a string for use inside a regular expression. */
|
|
7
|
+
const escapeRe = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
8
|
+
/** A tool pattern: `*` is any run of characters, the rest is literal, anchored at both ends. */
|
|
9
|
+
const glob = (pattern) => new RegExp(`^${pattern.split("*").map(escapeRe).join(".*")}$`);
|
|
10
|
+
/** Parses a policy file; throws on anything invalid (the server refuses to start). */
|
|
11
|
+
export function loadPolicy(bytes) {
|
|
12
|
+
const doc = JSON.parse(new TextDecoder().decode(bytes));
|
|
13
|
+
if (doc.version !== 1 || !Array.isArray(doc.rules))
|
|
14
|
+
throw new Error("not a v1 policy file");
|
|
15
|
+
for (const r of doc.rules) {
|
|
16
|
+
if (typeof r.id !== "string" || !/^[a-z0-9-]{1,64}$/.test(r.id))
|
|
17
|
+
throw new Error(`policy: rule id "${r.id}" must match ^[a-z0-9-]{1,64}$`);
|
|
18
|
+
if (typeof r.tool !== "string" || r.tool === "")
|
|
19
|
+
throw new Error(`policy ${r.id}: tool is required`);
|
|
20
|
+
if (r.action !== "deny" && r.action !== "allow" && r.action !== "require_approval")
|
|
21
|
+
throw new Error(`policy ${r.id}: action must be deny, allow or require_approval`);
|
|
22
|
+
if (r.args_match !== undefined) {
|
|
23
|
+
if (typeof r.args_match !== "string" || r.args_match.length > 200)
|
|
24
|
+
throw new Error(`policy ${r.id}: args_match must be a regular expression of ≤ 200 characters`);
|
|
25
|
+
new RegExp(r.args_match); // throws on bad syntax
|
|
26
|
+
}
|
|
27
|
+
if (r.audit !== undefined && typeof r.audit !== "boolean")
|
|
28
|
+
throw new Error(`policy ${r.id}: audit must be true or false`);
|
|
29
|
+
}
|
|
30
|
+
const ids = doc.rules.map((r) => r.id);
|
|
31
|
+
const twice = ids.find((id, i) => ids.indexOf(id) !== i);
|
|
32
|
+
if (twice !== undefined)
|
|
33
|
+
throw new Error(`policy: rule id "${twice}" is used twice`);
|
|
34
|
+
checkShadowed(doc.rules);
|
|
35
|
+
checkCompensations(doc.compensations);
|
|
36
|
+
return doc;
|
|
37
|
+
}
|
|
38
|
+
/** N3 (idea S5): a rule an earlier one always matches first, with another action, is a mistake. An
|
|
39
|
+
* earlier tool pattern covers a later one when it matches the later pattern's own text (its `*`
|
|
40
|
+
* can absorb the later `*`). Audit rules never decide, so they never shadow and are never dead. */
|
|
41
|
+
function checkShadowed(rules) {
|
|
42
|
+
const enforced = rules.filter((r) => r.audit !== true);
|
|
43
|
+
enforced.forEach((later, j) => {
|
|
44
|
+
const first = enforced
|
|
45
|
+
.slice(0, j)
|
|
46
|
+
.find((e) => glob(e.tool).test(later.tool) &&
|
|
47
|
+
(e.args_match === undefined ||
|
|
48
|
+
(e.args_match === later.args_match && !!e.ignore_case === !!later.ignore_case)));
|
|
49
|
+
if (first && first.action !== later.action)
|
|
50
|
+
throw new Error(`policy ${later.id}: never applies: ${first.id} matches the same calls first (${first.action})`);
|
|
51
|
+
});
|
|
52
|
+
}
|
|
53
|
+
export function compilePolicy(policy) {
|
|
54
|
+
return policy.rules.map((rule) => ({
|
|
55
|
+
rule,
|
|
56
|
+
tool: glob(rule.tool),
|
|
57
|
+
args: rule.args_match === undefined
|
|
58
|
+
? null
|
|
59
|
+
: new RegExp(rule.args_match, rule.ignore_case ? "i" : ""),
|
|
60
|
+
}));
|
|
61
|
+
}
|
|
62
|
+
/** The first rule that matches this call, or null (allowed). `names`: the tool's names (an MCP tool has two). */
|
|
63
|
+
export function decide(compiled, names, args) {
|
|
64
|
+
let text = null;
|
|
65
|
+
for (const c of compiled) {
|
|
66
|
+
if (c.rule.audit === true || !names.some((n) => c.tool.test(n)))
|
|
67
|
+
continue;
|
|
68
|
+
if (c.args) {
|
|
69
|
+
text ??= canonical(args ?? null).slice(0, MAX_ARGS);
|
|
70
|
+
if (!c.args.test(text))
|
|
71
|
+
continue;
|
|
72
|
+
}
|
|
73
|
+
return {
|
|
74
|
+
rule: c.rule.id,
|
|
75
|
+
action: c.rule.action,
|
|
76
|
+
...(c.rule.reason ? { reason: c.rule.reason } : {}),
|
|
77
|
+
};
|
|
78
|
+
}
|
|
79
|
+
return null;
|
|
80
|
+
}
|
|
81
|
+
/** N3 (idea C5): every audit rule this call matches, and what it would do if it were enforced. */
|
|
82
|
+
export function audited(compiled, names, args) {
|
|
83
|
+
const text = canonical(args ?? null).slice(0, MAX_ARGS);
|
|
84
|
+
return compiled
|
|
85
|
+
.filter((c) => c.rule.audit === true && names.some((n) => c.tool.test(n)))
|
|
86
|
+
.filter((c) => !c.args || c.args.test(text))
|
|
87
|
+
.map((c) => ({ rule: c.rule.id, would: c.rule.action }));
|
|
88
|
+
}
|
|
89
|
+
/** N3 (idea C5): what a record-only rule would have done. */
|
|
90
|
+
function auditFinding(rule, would, at, tool) {
|
|
91
|
+
return {
|
|
92
|
+
code: "POLICY_AUDIT",
|
|
93
|
+
source: "policy",
|
|
94
|
+
severity: "advisory",
|
|
95
|
+
ref: { rule, would, ...at },
|
|
96
|
+
detail: `Policy ${rule} (audit only) would ${would === "deny" ? "have refused" : "have held"} ${String(tool ?? "a tool call")}.`,
|
|
97
|
+
};
|
|
98
|
+
}
|
|
99
|
+
/**
|
|
100
|
+
* spec/policy.md §2 as findings: POLICY_DENIED for MCP calls the gateway refused (tool.call meta.policy),
|
|
101
|
+
* POLICY_VIOLATION for tools a model response asked for that a deny rule matches.
|
|
102
|
+
*/
|
|
103
|
+
export function policyFindings(lines, bodies, compiled) {
|
|
104
|
+
const out = [];
|
|
105
|
+
for (const line of lines) {
|
|
106
|
+
const e = JSON.parse(line);
|
|
107
|
+
const p = e.meta.policy;
|
|
108
|
+
if (e.kind === "tool.call" && p?.action === "deny" && typeof p.rule === "string")
|
|
109
|
+
out.push({
|
|
110
|
+
code: "POLICY_DENIED",
|
|
111
|
+
source: "policy",
|
|
112
|
+
severity: "advisory",
|
|
113
|
+
ref: { rule: p.rule, seq: e.seq },
|
|
114
|
+
detail: `The gateway refused ${String(e.meta.tool ?? "a tool call")} (policy ${p.rule}).`,
|
|
115
|
+
});
|
|
116
|
+
const hits = e.kind === "tool.call" ? e.meta.policy_audit : undefined;
|
|
117
|
+
for (const a of Array.isArray(hits) ? hits : [])
|
|
118
|
+
if (typeof a.rule === "string" && a.would !== "allow")
|
|
119
|
+
out.push(auditFinding(a.rule, String(a.would), { seq: e.seq }, e.meta.tool));
|
|
120
|
+
}
|
|
121
|
+
const calls = [...callsOf(lines, bodies)].sort((a, b) => a.endSeq - b.endSeq);
|
|
122
|
+
const violation = (tool, id, args) => {
|
|
123
|
+
for (const a of audited(compiled, [tool], args))
|
|
124
|
+
if (a.would !== "allow")
|
|
125
|
+
out.push(auditFinding(a.rule, a.would, { tool_use_id: id }, tool));
|
|
126
|
+
const d = decide(compiled, [tool], args);
|
|
127
|
+
if (d?.action === "deny")
|
|
128
|
+
out.push({
|
|
129
|
+
code: "POLICY_VIOLATION",
|
|
130
|
+
source: "policy",
|
|
131
|
+
severity: "warning",
|
|
132
|
+
ref: { rule: d.rule, tool_use_id: id },
|
|
133
|
+
detail: `The model asked for ${tool}, which policy ${d.rule} denies.`,
|
|
134
|
+
});
|
|
135
|
+
};
|
|
136
|
+
for (const c of calls) {
|
|
137
|
+
for (const [, b] of [...c.blocks].sort(([x], [y]) => x - y))
|
|
138
|
+
if (b.type === "tool_use" && typeof b.name === "string" && typeof b.id === "string")
|
|
139
|
+
violation(b.name, b.id, b.input ?? {});
|
|
140
|
+
for (const item of c.items)
|
|
141
|
+
if (item.type === "function_call" &&
|
|
142
|
+
typeof item.name === "string" &&
|
|
143
|
+
typeof item.call_id === "string") {
|
|
144
|
+
let args = item.arguments;
|
|
145
|
+
if (typeof args === "string")
|
|
146
|
+
try {
|
|
147
|
+
args = JSON.parse(args);
|
|
148
|
+
}
|
|
149
|
+
catch { }
|
|
150
|
+
violation(item.name, item.call_id, args);
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
return out;
|
|
154
|
+
}
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
import type { Finding } from "../analysis/index.ts";
|
|
2
|
+
import type { LoadedPrices } from "../cost/index.ts";
|
|
3
|
+
import type { Bodies } from "../reconcile/record.ts";
|
|
4
|
+
export interface PrecogModel {
|
|
5
|
+
version: string;
|
|
6
|
+
note?: string;
|
|
7
|
+
calls: number;
|
|
8
|
+
features: string[];
|
|
9
|
+
mean: number[];
|
|
10
|
+
std: number[];
|
|
11
|
+
weights: number[];
|
|
12
|
+
bias: number;
|
|
13
|
+
alert_ppm: number;
|
|
14
|
+
}
|
|
15
|
+
export declare const FEATURES: string[];
|
|
16
|
+
/** The record up to and including the `calls`-th llm.response (all of it if it has fewer). */
|
|
17
|
+
export declare function prefixOf(lines: readonly string[], calls: number): {
|
|
18
|
+
prefix: string[];
|
|
19
|
+
seen: number;
|
|
20
|
+
};
|
|
21
|
+
/** spec/precog.md §1, over a prefix. PRECOG_RISK findings are left out. */
|
|
22
|
+
export declare function features(prefix: readonly string[], bodies: Bodies, prices: LoadedPrices): number[];
|
|
23
|
+
export declare function predict(model: PrecogModel, x: readonly number[]): {
|
|
24
|
+
risk_ppm: number;
|
|
25
|
+
level: string;
|
|
26
|
+
};
|
|
27
|
+
/** The live alert (spec/precog.md §3): one PRECOG_RISK once the record reaches `calls` model calls. */
|
|
28
|
+
export declare function precogFindings(lines: readonly string[], bodies: Bodies, prices: LoadedPrices, model: PrecogModel): Finding[];
|
|
29
|
+
/** Batch gradient descent on the log loss, deterministic (spec/precog.md §4). */
|
|
30
|
+
export declare function train(examples: ReadonlyArray<{
|
|
31
|
+
x: readonly number[];
|
|
32
|
+
y: number;
|
|
33
|
+
}>, prior: PrecogModel, options?: {
|
|
34
|
+
iterations?: number;
|
|
35
|
+
rate?: number;
|
|
36
|
+
l2?: number;
|
|
37
|
+
}): PrecogModel;
|
|
38
|
+
/** Recall at ≤ 5% false alarms (spec/precog.md §5). */
|
|
39
|
+
export declare function calibration(scores: ReadonlyArray<{
|
|
40
|
+
risk_ppm: number;
|
|
41
|
+
y: number;
|
|
42
|
+
}>): {
|
|
43
|
+
sessions: number;
|
|
44
|
+
positives: number;
|
|
45
|
+
negatives: number;
|
|
46
|
+
threshold_ppm: null;
|
|
47
|
+
recall_ppm: null;
|
|
48
|
+
false_alarm_ppm: null;
|
|
49
|
+
} | {
|
|
50
|
+
sessions: number;
|
|
51
|
+
positives: number;
|
|
52
|
+
negatives: number;
|
|
53
|
+
threshold_ppm: number;
|
|
54
|
+
recall_ppm: number;
|
|
55
|
+
false_alarm_ppm: number;
|
|
56
|
+
};
|
|
57
|
+
/** spec/precog.md §6: which side of a reproducible split a session falls on, by its id's hash. */
|
|
58
|
+
export declare function splitOf(sessionId: string, testPercent: number): "train" | "test";
|
|
59
|
+
export declare const TARGET: {
|
|
60
|
+
recall_ppm: number;
|
|
61
|
+
false_alarm_ppm: number;
|
|
62
|
+
};
|
|
63
|
+
export declare const MIN_EACH = 30;
|
|
64
|
+
/** spec/precog.md §6: the model's calibration on the test side, against the roadmap target. */
|
|
65
|
+
export declare function precogReport(model: PrecogModel, examples: ReadonlyArray<{
|
|
66
|
+
session_id: string;
|
|
67
|
+
x: readonly number[];
|
|
68
|
+
y: number;
|
|
69
|
+
}>, testPercent: number): {
|
|
70
|
+
v: 1;
|
|
71
|
+
version: string;
|
|
72
|
+
test_percent: number;
|
|
73
|
+
in_sample: boolean;
|
|
74
|
+
train_sessions: number;
|
|
75
|
+
calibration: {
|
|
76
|
+
sessions: number;
|
|
77
|
+
positives: number;
|
|
78
|
+
negatives: number;
|
|
79
|
+
threshold_ppm: null;
|
|
80
|
+
recall_ppm: null;
|
|
81
|
+
false_alarm_ppm: null;
|
|
82
|
+
} | {
|
|
83
|
+
sessions: number;
|
|
84
|
+
positives: number;
|
|
85
|
+
negatives: number;
|
|
86
|
+
threshold_ppm: number;
|
|
87
|
+
recall_ppm: number;
|
|
88
|
+
false_alarm_ppm: number;
|
|
89
|
+
};
|
|
90
|
+
target: {
|
|
91
|
+
recall_ppm: number;
|
|
92
|
+
false_alarm_ppm: number;
|
|
93
|
+
};
|
|
94
|
+
min_each: number;
|
|
95
|
+
verdict: string;
|
|
96
|
+
};
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
// Precog v1 (spec/precog.md): early-trajectory features, a logistic model, training, calibration.
|
|
2
|
+
// Mirrors sdks/python/src/zanii_blackbox/precog.py.
|
|
3
|
+
import { createHash } from "node:crypto";
|
|
4
|
+
import { sessionSummary } from "../fleet/index.js";
|
|
5
|
+
export const FEATURES = [
|
|
6
|
+
"model_calls",
|
|
7
|
+
"tool_calls",
|
|
8
|
+
"distinct_tools",
|
|
9
|
+
"log_spend",
|
|
10
|
+
"warnings",
|
|
11
|
+
"cautions",
|
|
12
|
+
"near_misses",
|
|
13
|
+
"blocked",
|
|
14
|
+
];
|
|
15
|
+
/** The record up to and including the `calls`-th llm.response (all of it if it has fewer). */
|
|
16
|
+
export function prefixOf(lines, calls) {
|
|
17
|
+
let seen = 0;
|
|
18
|
+
for (const [i, line] of lines.entries())
|
|
19
|
+
if (JSON.parse(line).kind === "llm.response" && ++seen === calls)
|
|
20
|
+
return { prefix: lines.slice(0, i + 1), seen };
|
|
21
|
+
return { prefix: [...lines], seen };
|
|
22
|
+
}
|
|
23
|
+
/** spec/precog.md §1, over a prefix. PRECOG_RISK findings are left out. */
|
|
24
|
+
export function features(prefix, bodies, prices) {
|
|
25
|
+
const own = prefix.filter((l) => {
|
|
26
|
+
const e = JSON.parse(l);
|
|
27
|
+
return !(e.kind === "finding" && e.meta.code === "PRECOG_RISK");
|
|
28
|
+
});
|
|
29
|
+
const s = sessionSummary(own, bodies, prices);
|
|
30
|
+
return [
|
|
31
|
+
s.model_calls,
|
|
32
|
+
s.tool_calls,
|
|
33
|
+
Object.keys(s.tools).length,
|
|
34
|
+
Math.log(1 + s.spend_micro_usd),
|
|
35
|
+
s.findings.warning,
|
|
36
|
+
s.findings.caution,
|
|
37
|
+
s.near_misses,
|
|
38
|
+
s.blocked_by ? 1 : 0,
|
|
39
|
+
];
|
|
40
|
+
}
|
|
41
|
+
const sigmoid = (z) => 1 / (1 + Math.exp(-Math.min(500, Math.max(-500, z))));
|
|
42
|
+
export function predict(model, x) {
|
|
43
|
+
let z = model.bias;
|
|
44
|
+
for (const [i, v] of x.entries())
|
|
45
|
+
z +=
|
|
46
|
+
model.weights[i] * ((v - model.mean[i]) / model.std[i]);
|
|
47
|
+
const risk = Math.floor(sigmoid(z) * 1_000_000 + 0.5);
|
|
48
|
+
const level = risk >= model.alert_ppm ? "high" : risk >= Math.floor(model.alert_ppm / 2) ? "elevated" : "low";
|
|
49
|
+
return { risk_ppm: risk, level };
|
|
50
|
+
}
|
|
51
|
+
/** The live alert (spec/precog.md §3): one PRECOG_RISK once the record reaches `calls` model calls. */
|
|
52
|
+
export function precogFindings(lines, bodies, prices, model) {
|
|
53
|
+
const { prefix, seen } = prefixOf(lines, model.calls);
|
|
54
|
+
if (seen < model.calls)
|
|
55
|
+
return [];
|
|
56
|
+
const p = predict(model, features(prefix, bodies, prices));
|
|
57
|
+
return p.level === "high"
|
|
58
|
+
? [
|
|
59
|
+
{
|
|
60
|
+
code: "PRECOG_RISK",
|
|
61
|
+
source: "precog",
|
|
62
|
+
severity: "caution",
|
|
63
|
+
ref: { calls: model.calls, risk_ppm: p.risk_ppm },
|
|
64
|
+
},
|
|
65
|
+
]
|
|
66
|
+
: [];
|
|
67
|
+
}
|
|
68
|
+
/** Batch gradient descent on the log loss, deterministic (spec/precog.md §4). */
|
|
69
|
+
export function train(examples, prior, options = {}) {
|
|
70
|
+
const { iterations = 500, rate = 0.1, l2 = 0.01 } = options;
|
|
71
|
+
const n = examples.length;
|
|
72
|
+
const d = FEATURES.length;
|
|
73
|
+
const mean = [];
|
|
74
|
+
const std = [];
|
|
75
|
+
for (let j = 0; j < d; j++) {
|
|
76
|
+
let s = 0;
|
|
77
|
+
for (const e of examples)
|
|
78
|
+
s += e.x[j];
|
|
79
|
+
const m = n ? s / n : 0;
|
|
80
|
+
let q = 0;
|
|
81
|
+
for (const e of examples)
|
|
82
|
+
q += (e.x[j] - m) ** 2;
|
|
83
|
+
const v = n ? Math.sqrt(q / n) : 0;
|
|
84
|
+
mean.push(m);
|
|
85
|
+
std.push(v === 0 ? 1 : v);
|
|
86
|
+
}
|
|
87
|
+
const xs = examples.map((e) => e.x.map((v, j) => (v - mean[j]) / std[j]));
|
|
88
|
+
const w = new Array(d).fill(0);
|
|
89
|
+
let b = 0;
|
|
90
|
+
for (let it = 0; it < iterations && n > 0; it++) {
|
|
91
|
+
const gw = new Array(d).fill(0);
|
|
92
|
+
let gb = 0;
|
|
93
|
+
for (const [k, e] of examples.entries()) {
|
|
94
|
+
const row = xs[k];
|
|
95
|
+
let z = b;
|
|
96
|
+
for (let j = 0; j < d; j++)
|
|
97
|
+
z += w[j] * row[j];
|
|
98
|
+
const err = sigmoid(z) - e.y;
|
|
99
|
+
for (let j = 0; j < d; j++)
|
|
100
|
+
gw[j] = gw[j] + err * row[j];
|
|
101
|
+
gb += err;
|
|
102
|
+
}
|
|
103
|
+
for (let j = 0; j < d; j++)
|
|
104
|
+
w[j] = w[j] - rate * (gw[j] / n + l2 * w[j]);
|
|
105
|
+
b -= rate * (gb / n);
|
|
106
|
+
}
|
|
107
|
+
return {
|
|
108
|
+
version: `trained-${n}`,
|
|
109
|
+
calls: prior.calls,
|
|
110
|
+
features: [...FEATURES],
|
|
111
|
+
mean,
|
|
112
|
+
std,
|
|
113
|
+
weights: w,
|
|
114
|
+
bias: b,
|
|
115
|
+
alert_ppm: prior.alert_ppm,
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
/** Recall at ≤ 5% false alarms (spec/precog.md §5). */
|
|
119
|
+
export function calibration(scores) {
|
|
120
|
+
const neg = scores
|
|
121
|
+
.filter((s) => s.y === 0)
|
|
122
|
+
.map((s) => s.risk_ppm)
|
|
123
|
+
.sort((a, b) => a - b);
|
|
124
|
+
const pos = scores.filter((s) => s.y === 1).map((s) => s.risk_ppm);
|
|
125
|
+
const base = { sessions: scores.length, positives: pos.length, negatives: neg.length };
|
|
126
|
+
if (neg.length === 0 || pos.length === 0)
|
|
127
|
+
return { ...base, threshold_ppm: null, recall_ppm: null, false_alarm_ppm: null };
|
|
128
|
+
const threshold = neg[Math.max(0, Math.ceil((95 * neg.length) / 100) - 1)];
|
|
129
|
+
const ppm = (k, of) => Math.floor((k * 1_000_000) / of);
|
|
130
|
+
return {
|
|
131
|
+
...base,
|
|
132
|
+
threshold_ppm: threshold,
|
|
133
|
+
recall_ppm: ppm(pos.filter((v) => v > threshold).length, pos.length),
|
|
134
|
+
false_alarm_ppm: ppm(neg.filter((v) => v > threshold).length, neg.length),
|
|
135
|
+
};
|
|
136
|
+
}
|
|
137
|
+
/** spec/precog.md §6: which side of a reproducible split a session falls on, by its id's hash. */
|
|
138
|
+
export function splitOf(sessionId, testPercent) {
|
|
139
|
+
const h = createHash("sha256").update(sessionId, "utf8").digest();
|
|
140
|
+
return h.readUInt32BE(0) % 100 < testPercent ? "test" : "train";
|
|
141
|
+
}
|
|
142
|
+
export const TARGET = { recall_ppm: 700_000, false_alarm_ppm: 50_000 };
|
|
143
|
+
export const MIN_EACH = 30;
|
|
144
|
+
/** spec/precog.md §6: the model's calibration on the test side, against the roadmap target. */
|
|
145
|
+
export function precogReport(model, examples, testPercent) {
|
|
146
|
+
const test = testPercent === 0
|
|
147
|
+
? examples
|
|
148
|
+
: examples.filter((e) => splitOf(e.session_id, testPercent) === "test");
|
|
149
|
+
const cal = calibration(test.map((e) => ({ risk_ppm: predict(model, e.x).risk_ppm, y: e.y })));
|
|
150
|
+
const verdict = cal.positives < MIN_EACH || cal.negatives < MIN_EACH
|
|
151
|
+
? "not enough data"
|
|
152
|
+
: cal.recall_ppm >= TARGET.recall_ppm &&
|
|
153
|
+
cal.false_alarm_ppm <= TARGET.false_alarm_ppm
|
|
154
|
+
? "meets target"
|
|
155
|
+
: "misses target";
|
|
156
|
+
return {
|
|
157
|
+
v: 1,
|
|
158
|
+
version: model.version,
|
|
159
|
+
test_percent: testPercent,
|
|
160
|
+
in_sample: testPercent === 0,
|
|
161
|
+
train_sessions: examples.length - (testPercent === 0 ? 0 : test.length),
|
|
162
|
+
calibration: cal,
|
|
163
|
+
target: TARGET,
|
|
164
|
+
min_each: MIN_EACH,
|
|
165
|
+
verdict,
|
|
166
|
+
};
|
|
167
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
export type Outcome = "success" | "failure";
|
|
2
|
+
export interface InterventionEvidence {
|
|
3
|
+
pairs: number;
|
|
4
|
+
recovered: number;
|
|
5
|
+
not_recovered: number;
|
|
6
|
+
disrupted: number;
|
|
7
|
+
kept: number;
|
|
8
|
+
r_milli: number | null;
|
|
9
|
+
d_milli: number | null;
|
|
10
|
+
/** 95 % Wilson intervals. */
|
|
11
|
+
r_ci95_milli: [number, number] | null;
|
|
12
|
+
d_ci95_milli: [number, number] | null;
|
|
13
|
+
/** Act only above this failure chance; null without evidence on both sides. */
|
|
14
|
+
break_even_milli: number | null;
|
|
15
|
+
}
|
|
16
|
+
/** The evidence from (source outcome, fork outcome) pairs. */
|
|
17
|
+
export declare function interventionEvidence(pairs: ReadonlyArray<{
|
|
18
|
+
before: Outcome;
|
|
19
|
+
after: Outcome;
|
|
20
|
+
}>): InterventionEvidence;
|
|
21
|
+
/** The expected gain, in thousandths of a run, of acting on a run with this failure chance. */
|
|
22
|
+
export declare function interventionGain(e: InterventionEvidence, failureMilli: number): number | null;
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
// Intervention evidence (spec/precog.md §9, Stage 2 R6): is restarting a run worth it? From replay
|
|
2
|
+
// forks with known outcomes: recovery r (a failed run's fork succeeded) and disruption d (a good
|
|
3
|
+
// run's fork failed). Acting on a run with failure chance p gains p·r − (1−p)·d, so it pays only
|
|
4
|
+
// when p > d / (r + d) ("The Intervention Paradox", arXiv 2602.03338). Mirrors precog/intervention.py.
|
|
5
|
+
// Rates are in thousandths (integers, so the record stays exact).
|
|
6
|
+
const milli = (x) => Math.round(x * 1000);
|
|
7
|
+
function wilson(k, n) {
|
|
8
|
+
if (n === 0)
|
|
9
|
+
return null;
|
|
10
|
+
const z = 1.959963984540054;
|
|
11
|
+
const p = k / n;
|
|
12
|
+
const den = 1 + (z * z) / n;
|
|
13
|
+
const mid = (p + (z * z) / (2 * n)) / den;
|
|
14
|
+
const half = (z * Math.sqrt((p * (1 - p)) / n + (z * z) / (4 * n * n))) / den;
|
|
15
|
+
return [milli(Math.max(0, mid - half)), milli(Math.min(1, mid + half))];
|
|
16
|
+
}
|
|
17
|
+
/** The evidence from (source outcome, fork outcome) pairs. */
|
|
18
|
+
export function interventionEvidence(pairs) {
|
|
19
|
+
const failed = pairs.filter((p) => p.before === "failure");
|
|
20
|
+
const good = pairs.filter((p) => p.before === "success");
|
|
21
|
+
const recovered = failed.filter((p) => p.after === "success").length;
|
|
22
|
+
const disrupted = good.filter((p) => p.after === "failure").length;
|
|
23
|
+
const r = failed.length ? recovered / failed.length : null;
|
|
24
|
+
const d = good.length ? disrupted / good.length : null;
|
|
25
|
+
return {
|
|
26
|
+
pairs: pairs.length,
|
|
27
|
+
recovered,
|
|
28
|
+
not_recovered: failed.length - recovered,
|
|
29
|
+
disrupted,
|
|
30
|
+
kept: good.length - disrupted,
|
|
31
|
+
r_milli: r === null ? null : milli(r),
|
|
32
|
+
d_milli: d === null ? null : milli(d),
|
|
33
|
+
r_ci95_milli: wilson(recovered, failed.length),
|
|
34
|
+
d_ci95_milli: wilson(disrupted, good.length),
|
|
35
|
+
break_even_milli: r === null || d === null || r + d === 0 ? null : milli(d / (r + d)),
|
|
36
|
+
};
|
|
37
|
+
}
|
|
38
|
+
/** The expected gain, in thousandths of a run, of acting on a run with this failure chance. */
|
|
39
|
+
export function interventionGain(e, failureMilli) {
|
|
40
|
+
if (e.r_milli === null || e.d_milli === null)
|
|
41
|
+
return null;
|
|
42
|
+
const p = failureMilli / 1000;
|
|
43
|
+
return milli(p * (e.r_milli / 1000) - (1 - p) * (e.d_milli / 1000));
|
|
44
|
+
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
export interface NormalModel {
|
|
2
|
+
version: 1;
|
|
3
|
+
runs: number;
|
|
4
|
+
/** "a\tb" → how often b followed a, in the successful runs. */
|
|
5
|
+
transitions: Record<string, number>;
|
|
6
|
+
/** a → how often anything followed it. */
|
|
7
|
+
out: Record<string, number>;
|
|
8
|
+
symbols: number;
|
|
9
|
+
}
|
|
10
|
+
export interface NormalScore {
|
|
11
|
+
steps: number;
|
|
12
|
+
/** Mean surprise of each step, −ln p, in thousandths (integers, so the record stays exact). */
|
|
13
|
+
surprise_milli: number;
|
|
14
|
+
unseen_milli: number;
|
|
15
|
+
loop_milli: number;
|
|
16
|
+
}
|
|
17
|
+
/** A run's steps as symbols: each model call, and each tool by name (MCP or SDK-reported). */
|
|
18
|
+
export declare function activities(lines: readonly string[]): string[];
|
|
19
|
+
/** The map of the given successful runs (each a session's lines). */
|
|
20
|
+
export declare function buildNormal(runs: ReadonlyArray<readonly string[]>): NormalModel;
|
|
21
|
+
/** How far a run strays from the normal. Add-half smoothing; one more symbol for the unseen. */
|
|
22
|
+
export declare function normalScore(model: NormalModel, lines: readonly string[]): NormalScore;
|
|
23
|
+
/** The alarm threshold from the scores of held-out successful runs: flag a score above it. With
|
|
24
|
+
* probability ≥ `confidence`, at most `alpha` of normal runs are flagged (an order statistic with a
|
|
25
|
+
* Clopper-Pearson-style binomial bound). The lowest such threshold, for the most recall; null when
|
|
26
|
+
* there are too few runs for the guarantee. */
|
|
27
|
+
export declare function normalThreshold(scores: readonly number[], alpha?: number, confidence?: number): {
|
|
28
|
+
threshold: number;
|
|
29
|
+
above: number;
|
|
30
|
+
needed: number;
|
|
31
|
+
} | null;
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
// Each customer's normal (spec/precog.md §8, Stage 2 R5): a map of the paths a customer's own
|
|
2
|
+
// successful runs take, learned without failure labels, and a score for how far a run strays from
|
|
3
|
+
// it. The alarm threshold comes with a guarantee: with the chosen confidence, at most `alpha` of
|
|
4
|
+
// normal runs are flagged. Mirrors precog/normal.py.
|
|
5
|
+
/** A run's steps as symbols: each model call, and each tool by name (MCP or SDK-reported). */
|
|
6
|
+
export function activities(lines) {
|
|
7
|
+
const out = [];
|
|
8
|
+
for (const line of lines) {
|
|
9
|
+
const e = JSON.parse(line);
|
|
10
|
+
const m = e.meta;
|
|
11
|
+
if (e.kind === "llm.request")
|
|
12
|
+
out.push("model");
|
|
13
|
+
else if (e.kind === "tool.call" && m.method === "tools/call")
|
|
14
|
+
out.push(`tool:${String(m.server ?? "")}__${String(m.tool ?? "")}`);
|
|
15
|
+
else if (e.kind === "sdk.event" && m.type === "tool.call" && typeof m.name === "string")
|
|
16
|
+
out.push(`tool:${m.name}`);
|
|
17
|
+
}
|
|
18
|
+
return out;
|
|
19
|
+
}
|
|
20
|
+
/** The map of the given successful runs (each a session's lines). */
|
|
21
|
+
export function buildNormal(runs) {
|
|
22
|
+
const transitions = {};
|
|
23
|
+
const out = {};
|
|
24
|
+
const symbols = new Set();
|
|
25
|
+
for (const lines of runs) {
|
|
26
|
+
const acts = activities(lines);
|
|
27
|
+
acts.forEach((b, i) => {
|
|
28
|
+
const a = i === 0 ? "start" : acts[i - 1];
|
|
29
|
+
const key = `${a}\t${b}`;
|
|
30
|
+
transitions[key] = (transitions[key] ?? 0) + 1;
|
|
31
|
+
out[a] = (out[a] ?? 0) + 1;
|
|
32
|
+
symbols.add(b);
|
|
33
|
+
});
|
|
34
|
+
}
|
|
35
|
+
return { version: 1, runs: runs.length, transitions, out, symbols: symbols.size };
|
|
36
|
+
}
|
|
37
|
+
/** How far a run strays from the normal. Add-half smoothing; one more symbol for the unseen. */
|
|
38
|
+
export function normalScore(model, lines) {
|
|
39
|
+
const acts = activities(lines);
|
|
40
|
+
const k = model.symbols + 1;
|
|
41
|
+
let surprise = 0;
|
|
42
|
+
let unseen = 0;
|
|
43
|
+
acts.forEach((b, i) => {
|
|
44
|
+
const a = i === 0 ? "start" : acts[i - 1];
|
|
45
|
+
const n = model.transitions[`${a}\t${b}`] ?? 0;
|
|
46
|
+
if (n === 0)
|
|
47
|
+
unseen++;
|
|
48
|
+
surprise += -Math.log((n + 0.5) / ((model.out[a] ?? 0) + 0.5 * k));
|
|
49
|
+
});
|
|
50
|
+
const bigrams = acts.slice(1).map((b, i) => `${acts[i]}\t${b}`);
|
|
51
|
+
const loops = bigrams.filter((g, i) => bigrams.indexOf(g) < i).length;
|
|
52
|
+
const per = (x, n) => (n ? Math.round((1000 * x) / n) : 0);
|
|
53
|
+
return {
|
|
54
|
+
steps: acts.length,
|
|
55
|
+
surprise_milli: per(surprise, acts.length),
|
|
56
|
+
unseen_milli: per(unseen, acts.length),
|
|
57
|
+
loop_milli: per(loops, bigrams.length),
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
/** P(X ≥ j) for X ~ Binomial(n, p), from the pmf in log space (no underflow for large n). */
|
|
61
|
+
function binomialTail(n, p, j) {
|
|
62
|
+
let logPmf = n * Math.log(1 - p); // ln P(X = 0)
|
|
63
|
+
let below = 0;
|
|
64
|
+
for (let x = 0; x < j; x++) {
|
|
65
|
+
below += Math.exp(logPmf);
|
|
66
|
+
logPmf += Math.log((n - x) / (x + 1)) + Math.log(p / (1 - p));
|
|
67
|
+
}
|
|
68
|
+
return Math.max(0, 1 - below);
|
|
69
|
+
}
|
|
70
|
+
/** The alarm threshold from the scores of held-out successful runs: flag a score above it. With
|
|
71
|
+
* probability ≥ `confidence`, at most `alpha` of normal runs are flagged (an order statistic with a
|
|
72
|
+
* Clopper-Pearson-style binomial bound). The lowest such threshold, for the most recall; null when
|
|
73
|
+
* there are too few runs for the guarantee. */
|
|
74
|
+
export function normalThreshold(scores, alpha = 0.05, confidence = 0.95) {
|
|
75
|
+
const n = scores.length;
|
|
76
|
+
const sorted = [...scores].sort((a, b) => b - a); // highest first
|
|
77
|
+
const needed = Math.ceil(Math.log(1 - confidence) / Math.log(1 - alpha));
|
|
78
|
+
// j calibration runs may score above the threshold: the largest j the guarantee allows
|
|
79
|
+
let best = -1;
|
|
80
|
+
for (let j = 0; j < n; j++) {
|
|
81
|
+
if (binomialTail(n, alpha, j + 1) >= confidence)
|
|
82
|
+
best = j;
|
|
83
|
+
else
|
|
84
|
+
break;
|
|
85
|
+
}
|
|
86
|
+
if (best < 0)
|
|
87
|
+
return null;
|
|
88
|
+
return { threshold: sorted[best], above: best, needed };
|
|
89
|
+
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
export interface PreflightFinding {
|
|
2
|
+
code: "PREFLIGHT_DEGRADED";
|
|
3
|
+
source: "preflight";
|
|
4
|
+
severity: "advisory";
|
|
5
|
+
/** `degraded`: the optional items that failed, comma-separated (a ref holds scalars only). */
|
|
6
|
+
ref: {
|
|
7
|
+
seq: number;
|
|
8
|
+
degraded: string;
|
|
9
|
+
};
|
|
10
|
+
}
|
|
11
|
+
export declare function preflightFindings(lines: readonly string[]): PreflightFinding[];
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
// Pre-flight (spec/preflight.md §3): a session that left with equipment missing. Pure; mirrors
|
|
2
|
+
// sdks/python/src/zanii_blackbox/preflight.py; pinned by spec/vectors/preflight.json.
|
|
3
|
+
export function preflightFindings(lines) {
|
|
4
|
+
const first = lines[0];
|
|
5
|
+
if (first === undefined)
|
|
6
|
+
return [];
|
|
7
|
+
const e = JSON.parse(first);
|
|
8
|
+
const degraded = e.kind === "session.open" ? e.meta.preflight?.degraded : undefined;
|
|
9
|
+
if (!Array.isArray(degraded) || degraded.length === 0)
|
|
10
|
+
return [];
|
|
11
|
+
return [
|
|
12
|
+
{
|
|
13
|
+
code: "PREFLIGHT_DEGRADED",
|
|
14
|
+
source: "preflight",
|
|
15
|
+
severity: "advisory",
|
|
16
|
+
ref: { seq: e.seq, degraded: degraded.map(String).join(",") },
|
|
17
|
+
},
|
|
18
|
+
];
|
|
19
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import { type LoadedPrices } from "../cost/index.ts";
|
|
2
|
+
import { type AirworthinessCriteria, airworthiness, type Summary } from "../fleet/index.ts";
|
|
3
|
+
export type Rating = ReturnType<typeof airworthiness> & {
|
|
4
|
+
model: string;
|
|
5
|
+
};
|
|
6
|
+
export interface TypeUnratedFinding {
|
|
7
|
+
code: "TYPE_UNRATED";
|
|
8
|
+
source: "type_rating";
|
|
9
|
+
severity: "advisory" | "warning";
|
|
10
|
+
ref: {
|
|
11
|
+
seq: number;
|
|
12
|
+
model: string;
|
|
13
|
+
};
|
|
14
|
+
}
|
|
15
|
+
/** spec/type-ratings.md §1. `items` newest first. */
|
|
16
|
+
export declare function typeRatings(items: ReadonlyArray<{
|
|
17
|
+
summary: Summary;
|
|
18
|
+
models: readonly string[];
|
|
19
|
+
}>, criteria?: AirworthinessCriteria): Rating[];
|
|
20
|
+
/** spec/type-ratings.md §2. `severity` is "warning" when the server requires type ratings. */
|
|
21
|
+
export declare function typeUnrated(lines: readonly string[], prices: LoadedPrices, ratings: readonly Rating[], severity?: "advisory" | "warning"): TypeUnratedFinding[];
|