@zanii/blackbox 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -1
- package/dist/a2a/index.d.ts +27 -0
- package/dist/a2a/index.js +104 -1
- package/dist/agents/index.d.ts +8 -0
- package/dist/analysis/accuracy.d.ts +24 -0
- package/dist/analysis/accuracy.js +45 -0
- package/dist/analysis/credential.d.ts +101 -0
- package/dist/analysis/credential.js +142 -0
- package/dist/analysis/faults.js +115 -0
- package/dist/analysis/grounding.d.ts +122 -0
- package/dist/analysis/grounding.js +445 -0
- package/dist/analysis/hallucination.d.ts +32 -0
- package/dist/analysis/hallucination.js +357 -0
- package/dist/analysis/index.d.ts +23 -0
- package/dist/analysis/index.js +93 -0
- package/dist/analysis/memory.d.ts +8 -0
- package/dist/analysis/memory.js +35 -8
- package/dist/analysis/reference.d.ts +49 -0
- package/dist/analysis/reference.js +164 -0
- package/dist/analysis/taxonomy.js +1 -0
- package/dist/approvals/index.d.ts +23 -0
- package/dist/approvals/index.js +48 -0
- package/dist/archive/parquet.d.ts +2 -0
- package/dist/archive/parquet.js +185 -0
- package/dist/badge/index.d.ts +16 -0
- package/dist/badge/index.js +48 -0
- package/dist/bom/index.js +20 -0
- package/dist/cli.js +114 -10
- package/dist/compliance/art12.js +36 -9
- package/dist/compliance/index.d.ts +36 -2
- package/dist/compliance/index.js +78 -11
- package/dist/compliance/zanii.d.ts +29 -0
- package/dist/compliance/zanii.js +84 -0
- package/dist/constitution/index.d.ts +57 -0
- package/dist/constitution/index.js +131 -0
- package/dist/cv/index.d.ts +39 -0
- package/dist/cv/index.js +108 -0
- package/dist/disclosure/index.d.ts +31 -0
- package/dist/disclosure/index.js +113 -0
- package/dist/encryption/index.d.ts +9 -0
- package/dist/encryption/index.js +31 -0
- package/dist/evidence/index.d.ts +60 -0
- package/dist/evidence/index.js +151 -0
- package/dist/federation/index.d.ts +35 -0
- package/dist/federation/index.js +102 -0
- package/dist/finance/index.d.ts +126 -0
- package/dist/finance/index.js +320 -0
- package/dist/fleet/index.js +9 -0
- package/dist/gov/index.d.ts +108 -0
- package/dist/gov/index.js +225 -0
- package/dist/health/index.d.ts +120 -0
- package/dist/health/index.js +233 -0
- package/dist/index.d.ts +28 -6
- package/dist/index.js +28 -6
- package/dist/memory/index.d.ts +36 -0
- package/dist/memory/index.js +85 -0
- package/dist/occurrence/index.d.ts +11 -0
- package/dist/occurrence/index.js +18 -0
- package/dist/otlp/index.js +28 -1
- package/dist/packs/index.js +44 -4
- package/dist/policy/delta.js +7 -1
- package/dist/policy/index.d.ts +23 -6
- package/dist/policy/index.js +151 -8
- package/dist/policy/zanii.d.ts +31 -0
- package/dist/policy/zanii.js +87 -0
- package/dist/pq/index.d.ts +23 -0
- package/dist/pq/index.js +104 -0
- package/dist/search/index.d.ts +23 -0
- package/dist/search/index.js +69 -0
- package/dist/session/index.d.ts +89 -1
- package/dist/session/index.js +143 -11
- package/dist/sla/index.d.ts +61 -0
- package/dist/sla/index.js +197 -0
- package/dist/succession/index.d.ts +50 -0
- package/dist/succession/index.js +123 -0
- package/dist/tokens/index.d.ts +6 -0
- package/dist/tokens/index.js +46 -0
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/dist/walls/index.d.ts +31 -0
- package/dist/walls/index.js +119 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
The TypeScript SDK for [Zanii Blackbox](https://zanii.agency), the flight recorder for AI agents.
|
|
4
4
|
|
|
5
|
-
Blackbox records every model call and tool call your agent makes, from a gateway that runs outside the agent's process. The records are hash-chained, so an edit or a deletion shows, and anchored as receipts that anyone can verify offline. This SDK is optional. It adds what only the agent knows (steps, its own tools, its checks, the outcome) as a second, independent record, and it reads and verifies what the gateway recorded.
|
|
5
|
+
Blackbox records every model call and tool call your agent makes, for any industry and any organisation, from a gateway that runs outside the agent's process. The records are hash-chained, so an edit or a deletion shows, and anchored as receipts that anyone can verify offline. This SDK is optional. It adds what only the agent knows (steps, its own tools, its checks, the outcome) as a second, independent record, and it reads and verifies what the gateway recorded.
|
|
6
6
|
|
|
7
7
|
```sh
|
|
8
8
|
npm install @zanii/blackbox
|
|
@@ -65,6 +65,31 @@ await s.close("success");
|
|
|
65
65
|
- **It stays out of the way.** Recording is non-blocking with bounded memory. A recorded call costs about 0.3 ms (measured over 20,000 calls on a laptop); shipping happens in the background.
|
|
66
66
|
- **It has no other runtime dependencies** besides the Zanii core libraries.
|
|
67
67
|
|
|
68
|
+
## Hallucination controls
|
|
69
|
+
|
|
70
|
+
The gateway checks every record, with no model needed: an invented tool or ID, arguments a tool's schema refuses, "done" right after an error, and amounts, doses, dates and IDs in answers that nothing the agent was given holds. With your organisation's own **reference packs** (prices, fees, rules, limits) it also catches answers that contradict them, and your own grounding model and judge can plug in. A risky answer can be held for a person before your customer sees it.
|
|
71
|
+
|
|
72
|
+
From the SDK, pin what the agent read, then check an answer's signed credential anywhere:
|
|
73
|
+
|
|
74
|
+
```ts
|
|
75
|
+
import { verifyAnswerCredential, verifyCertificate } from "@zanii/blackbox";
|
|
76
|
+
|
|
77
|
+
s.retrieved([{ id: "returns-policy", version: "2026-10", content: policyText }]); // hashed here, never sent
|
|
78
|
+
|
|
79
|
+
// later, from GET /v1/sessions/:id/answers/:seq/credential and the session's record:
|
|
80
|
+
verifyCertificate(credential).ok; // signed by the gateway
|
|
81
|
+
verifyAnswerCredential(credential, lines, bodies, packs); // { ok, problems }: the record backs every claim
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
The same checks run offline as pure functions: `toolHallucinations`, `groundingFindings`, `factsIn`, `loadReferencePacks`, `answerRisk`, `detectorAccuracy`.
|
|
85
|
+
|
|
86
|
+
## New in 0.4.0
|
|
87
|
+
|
|
88
|
+
- Hallucination controls: the checks above, reference packs, answer credentials (`answerClaims`, `answerCredential`, `verifyAnswerCredential`), `session.retrieved()`, accuracy from people's labels (`detectorAccuracy`, Wilson bounds), the grounding and judge contracts, `answerRisk` for held answers.
|
|
89
|
+
- Keyed redaction tokens (`tokenFor`, `resolveTokens`).
|
|
90
|
+
- Evidence for payments, health and public services (`payment`, `healthAccess`, `decision` and their verifiers); provable agent memory (`memoryChain`); record badges, agent CVs, SLAs, ledger co-signing and the post-quantum binding.
|
|
91
|
+
- Compliance reports in Arabic, and the hallucination-controls report (`hallucinationFacts`).
|
|
92
|
+
|
|
68
93
|
## Frameworks
|
|
69
94
|
|
|
70
95
|
| Framework | Use |
|
package/dist/a2a/index.d.ts
CHANGED
|
@@ -47,4 +47,31 @@ type Jwk = {
|
|
|
47
47
|
* least one signature verifies.
|
|
48
48
|
*/
|
|
49
49
|
export declare function verifyAgentCard(card: unknown, keys: readonly Jwk[]): AgentCardReport;
|
|
50
|
+
/**
|
|
51
|
+
* spec/a2a.md §3a (L3): the card as the gateway serves it, so a client that follows it stays on the
|
|
52
|
+
* record: every JSON-RPC interface points at `gatewayUrl`, other bindings (which the gateway can't
|
|
53
|
+
* capture) are dropped, and the publisher's signatures, which no longer hold, are replaced by the
|
|
54
|
+
* gateway's own EdDSA signature (kid: its did:key).
|
|
55
|
+
*/
|
|
56
|
+
export declare function rewriteAgentCard(card: Record<string, unknown>, o: {
|
|
57
|
+
gatewayUrl: string;
|
|
58
|
+
signer: {
|
|
59
|
+
kid: string;
|
|
60
|
+
seed: Uint8Array;
|
|
61
|
+
};
|
|
62
|
+
}): Record<string, unknown>;
|
|
63
|
+
/** The JWK of a gateway's Ed25519 did:key, to check a card it rewrote with verifyAgentCard. */
|
|
64
|
+
export declare function gatewayCardKey(did: string): Jwk;
|
|
65
|
+
/** The A2A method and task a REST call is, or undefined for a path that isn't one of A2A's. */
|
|
66
|
+
export declare function a2aRestMethod(httpMethod: string, path: string): {
|
|
67
|
+
method: string;
|
|
68
|
+
taskId?: string;
|
|
69
|
+
} | undefined;
|
|
70
|
+
/** §2a: a REST call's meta: the same fields as a JSON-RPC call's, with `binding: "http+json"`. */
|
|
71
|
+
export declare function a2aRestRequestMeta(httpMethod: string, path: string, body: Uint8Array, headers?: {
|
|
72
|
+
version?: string | null;
|
|
73
|
+
extensions?: string | null;
|
|
74
|
+
}): Meta;
|
|
75
|
+
/** §2a: a REST answer's meta (a bare Task, a `{task | message}`, or one SSE event's StreamResponse). */
|
|
76
|
+
export declare function a2aRestResponseMeta(bytes: Uint8Array, contentType: string | undefined): Meta;
|
|
50
77
|
export {};
|
package/dist/a2a/index.js
CHANGED
|
@@ -3,7 +3,8 @@
|
|
|
3
3
|
// an Agent Card's JWS signatures (A2A §8.4: detached, over the JCS card without `signatures`).
|
|
4
4
|
// Mirrors sdks/python/src/zanii_blackbox/a2a.py.
|
|
5
5
|
import { createHash, createPublicKey, verify } from "node:crypto";
|
|
6
|
-
import { canonicalBytes } from "@zanii/core";
|
|
6
|
+
import { canonicalBytes, publicKeyFromDid } from "@zanii/core";
|
|
7
|
+
import { ed25519Sign } from "../transparency/index.js";
|
|
7
8
|
import { assertSafeIntegers } from "../verify/envelope.js";
|
|
8
9
|
/** v0.3 method names → v1.0 (A2A §9.4). */
|
|
9
10
|
const LEGACY = {
|
|
@@ -200,3 +201,105 @@ export function verifyAgentCard(card, keys) {
|
|
|
200
201
|
: "no signature is by a trusted key");
|
|
201
202
|
return report;
|
|
202
203
|
}
|
|
204
|
+
const JSONRPC = (t) => typeof t === "string" && t.toUpperCase().replace(/[-_]/g, "") === "JSONRPC";
|
|
205
|
+
/**
|
|
206
|
+
* spec/a2a.md §3a (L3): the card as the gateway serves it, so a client that follows it stays on the
|
|
207
|
+
* record: every JSON-RPC interface points at `gatewayUrl`, other bindings (which the gateway can't
|
|
208
|
+
* capture) are dropped, and the publisher's signatures, which no longer hold, are replaced by the
|
|
209
|
+
* gateway's own EdDSA signature (kid: its did:key).
|
|
210
|
+
*/
|
|
211
|
+
export function rewriteAgentCard(card, o) {
|
|
212
|
+
const { signatures: _drop, ...rest } = card;
|
|
213
|
+
const out = { ...rest };
|
|
214
|
+
if (Array.isArray(card.supportedInterfaces))
|
|
215
|
+
out.supportedInterfaces = card.supportedInterfaces
|
|
216
|
+
.filter((i) => i && typeof i === "object" && JSONRPC(i.protocolBinding ?? i.transport))
|
|
217
|
+
.map((i) => ({ ...i, url: o.gatewayUrl }));
|
|
218
|
+
if (typeof card.url === "string")
|
|
219
|
+
out.url = o.gatewayUrl;
|
|
220
|
+
if (card.preferredTransport !== undefined)
|
|
221
|
+
out.preferredTransport = "JSONRPC";
|
|
222
|
+
if (Array.isArray(card.additionalInterfaces))
|
|
223
|
+
out.additionalInterfaces = card.additionalInterfaces
|
|
224
|
+
.filter((i) => i && typeof i === "object" && JSONRPC(i.transport ?? i.protocolBinding))
|
|
225
|
+
.map((i) => ({ ...i, url: o.gatewayUrl }));
|
|
226
|
+
const header = b64u(new TextEncoder().encode(JSON.stringify({ alg: "EdDSA", kid: o.signer.kid, typ: "JOSE" })));
|
|
227
|
+
const input = new TextEncoder().encode(`${header}.${agentCardPayload(out)}`);
|
|
228
|
+
const signature = b64u(ed25519Sign(o.signer.seed, input));
|
|
229
|
+
return { ...out, signatures: [{ protected: header, signature }] };
|
|
230
|
+
}
|
|
231
|
+
/** The JWK of a gateway's Ed25519 did:key, to check a card it rewrote with verifyAgentCard. */
|
|
232
|
+
export function gatewayCardKey(did) {
|
|
233
|
+
const pub = publicKeyFromDid(did);
|
|
234
|
+
if (!pub)
|
|
235
|
+
throw new Error("not an Ed25519 did:key");
|
|
236
|
+
return { kty: "OKP", crv: "Ed25519", x: b64u(pub), kid: did };
|
|
237
|
+
}
|
|
238
|
+
// ---------------------------------------------------------------- the HTTP+JSON binding (§2a, L8)
|
|
239
|
+
/** A2A §11's REST routes, by HTTP method and path (after the agent's base, an optional `/v1`). */
|
|
240
|
+
const REST = [
|
|
241
|
+
["POST", /^\/message:send$/, "SendMessage"],
|
|
242
|
+
["POST", /^\/message:stream$/, "SendStreamingMessage"],
|
|
243
|
+
["GET", /^\/tasks$/, "ListTasks"],
|
|
244
|
+
["GET", /^\/tasks\/([^/:]+)$/, "GetTask"],
|
|
245
|
+
["POST", /^\/tasks\/([^/:]+):cancel$/, "CancelTask"],
|
|
246
|
+
["GET", /^\/tasks\/([^/:]+):subscribe$/, "SubscribeToTask"],
|
|
247
|
+
["POST", /^\/tasks\/([^/:]+):subscribe$/, "SubscribeToTask"],
|
|
248
|
+
["POST", /^\/tasks\/([^/:]+)\/pushNotificationConfigs$/, "CreateTaskPushNotificationConfig"],
|
|
249
|
+
["GET", /^\/tasks\/([^/:]+)\/pushNotificationConfigs$/, "ListTaskPushNotificationConfigs"],
|
|
250
|
+
["GET", /^\/tasks\/([^/:]+)\/pushNotificationConfigs\/[^/]+$/, "GetTaskPushNotificationConfig"],
|
|
251
|
+
[
|
|
252
|
+
"DELETE",
|
|
253
|
+
/^\/tasks\/([^/:]+)\/pushNotificationConfigs\/[^/]+$/,
|
|
254
|
+
"DeleteTaskPushNotificationConfig",
|
|
255
|
+
],
|
|
256
|
+
["GET", /^\/extendedAgentCard$/, "GetExtendedAgentCard"],
|
|
257
|
+
];
|
|
258
|
+
/** The A2A method and task a REST call is, or undefined for a path that isn't one of A2A's. */
|
|
259
|
+
export function a2aRestMethod(httpMethod, path) {
|
|
260
|
+
const p = (path.split("?")[0] ?? "").replace(/^\/v1(?=\/)/, "");
|
|
261
|
+
for (const [verb, re, method] of REST) {
|
|
262
|
+
if (verb !== httpMethod.toUpperCase())
|
|
263
|
+
continue;
|
|
264
|
+
const m = re.exec(p);
|
|
265
|
+
if (m)
|
|
266
|
+
return m[1] ? { method, taskId: decodeURIComponent(m[1]) } : { method };
|
|
267
|
+
}
|
|
268
|
+
return undefined;
|
|
269
|
+
}
|
|
270
|
+
/** §2a: a REST call's meta: the same fields as a JSON-RPC call's, with `binding: "http+json"`. */
|
|
271
|
+
export function a2aRestRequestMeta(httpMethod, path, body, headers = {}) {
|
|
272
|
+
const r = a2aRestMethod(httpMethod, path);
|
|
273
|
+
let params = {};
|
|
274
|
+
try {
|
|
275
|
+
params = obj(JSON.parse(Buffer.from(body).toString("utf8")));
|
|
276
|
+
}
|
|
277
|
+
catch { }
|
|
278
|
+
if (r?.taskId !== undefined)
|
|
279
|
+
params = { ...params, id: r.taskId };
|
|
280
|
+
const rpc = r ? { method: r.method, params } : { params };
|
|
281
|
+
return {
|
|
282
|
+
...a2aRequestMeta(new TextEncoder().encode(JSON.stringify(rpc)), headers),
|
|
283
|
+
binding: "http+json",
|
|
284
|
+
};
|
|
285
|
+
}
|
|
286
|
+
/** §2a: a REST answer's meta (a bare Task, a `{task | message}`, or one SSE event's StreamResponse). */
|
|
287
|
+
export function a2aRestResponseMeta(bytes, contentType) {
|
|
288
|
+
let text = Buffer.from(bytes).toString("utf8");
|
|
289
|
+
if (contentType?.includes("text/event-stream"))
|
|
290
|
+
text = text
|
|
291
|
+
.split(/\r?\n/)
|
|
292
|
+
.filter((l) => l.startsWith("data:"))
|
|
293
|
+
.map((l) => l.slice(5).trim())
|
|
294
|
+
.join("\n");
|
|
295
|
+
let value;
|
|
296
|
+
try {
|
|
297
|
+
value = obj(JSON.parse(text));
|
|
298
|
+
}
|
|
299
|
+
catch {
|
|
300
|
+
return {};
|
|
301
|
+
}
|
|
302
|
+
const oneOf = ["task", "message", "statusUpdate", "artifactUpdate"].some((k) => k in value);
|
|
303
|
+
const result = oneOf ? value : "id" in value && "status" in value ? { task: value } : value;
|
|
304
|
+
return responseOf({ result });
|
|
305
|
+
}
|
package/dist/agents/index.d.ts
CHANGED
|
@@ -8,6 +8,14 @@ export declare function memoryXray(lines: readonly string[], bodies: Bodies, rev
|
|
|
8
8
|
seq: number;
|
|
9
9
|
memory_id: string;
|
|
10
10
|
}[];
|
|
11
|
+
chain?: {
|
|
12
|
+
ok: boolean;
|
|
13
|
+
length: number;
|
|
14
|
+
broken: Array<{
|
|
15
|
+
seq: number;
|
|
16
|
+
reason: string;
|
|
17
|
+
}>;
|
|
18
|
+
};
|
|
11
19
|
};
|
|
12
20
|
/** spec/agents.md §3. */
|
|
13
21
|
export declare function causality(records: ReadonlyArray<readonly string[]>): {
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
export type Label = {
|
|
2
|
+
code: string;
|
|
3
|
+
verdict: "right" | "wrong" | "missed";
|
|
4
|
+
};
|
|
5
|
+
export interface Accuracy {
|
|
6
|
+
code: string;
|
|
7
|
+
right: number;
|
|
8
|
+
wrong: number;
|
|
9
|
+
missed: number;
|
|
10
|
+
/** right / (right + wrong): how often a finding was real. null with no right or wrong labels. */
|
|
11
|
+
precision: number | null;
|
|
12
|
+
/** right / (right + missed): how often a real case was found. null with neither. */
|
|
13
|
+
recall: number | null;
|
|
14
|
+
/** The Wilson 95% lower bounds of the two. */
|
|
15
|
+
precision_low: number | null;
|
|
16
|
+
recall_low: number | null;
|
|
17
|
+
}
|
|
18
|
+
/** The Wilson score interval's lower bound for `k` of `n`, at 95%; null when n = 0. */
|
|
19
|
+
export declare function wilsonLow(k: number, n: number): number | null;
|
|
20
|
+
/** §12: each code's accuracy, by code, and all codes together. */
|
|
21
|
+
export declare function detectorAccuracy(labels: readonly Label[]): {
|
|
22
|
+
codes: Accuracy[];
|
|
23
|
+
overall: Accuracy;
|
|
24
|
+
};
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
// How right each check is (spec/findings.md §12, H5): people mark findings right or wrong, and
|
|
2
|
+
// report hallucinations no check caught; from those labels, each code's precision and recall, with
|
|
3
|
+
// a 95% lower bound an auditor can quote. Mirrors sdks/python/src/zanii_blackbox/analysis/
|
|
4
|
+
// accuracy.py; pinned by spec/vectors/accuracy.json.
|
|
5
|
+
const Z = 1.96;
|
|
6
|
+
const round4 = (x) => Math.floor(x * 10_000 + 0.5) / 10_000;
|
|
7
|
+
/** The Wilson score interval's lower bound for `k` of `n`, at 95%; null when n = 0. */
|
|
8
|
+
export function wilsonLow(k, n) {
|
|
9
|
+
if (n === 0)
|
|
10
|
+
return null;
|
|
11
|
+
const p = k / n;
|
|
12
|
+
const z2 = Z * Z;
|
|
13
|
+
const centre = p + z2 / (2 * n);
|
|
14
|
+
const spread = Z * Math.sqrt((p * (1 - p)) / n + z2 / (4 * n * n));
|
|
15
|
+
return round4(Math.max(0, (centre - spread) / (1 + z2 / n)));
|
|
16
|
+
}
|
|
17
|
+
function row(code, right, wrong, missed) {
|
|
18
|
+
const judged = right + wrong;
|
|
19
|
+
const real = right + missed;
|
|
20
|
+
return {
|
|
21
|
+
code,
|
|
22
|
+
right,
|
|
23
|
+
wrong,
|
|
24
|
+
missed,
|
|
25
|
+
precision: judged ? round4(right / judged) : null,
|
|
26
|
+
recall: real ? round4(right / real) : null,
|
|
27
|
+
precision_low: wilsonLow(right, judged),
|
|
28
|
+
recall_low: wilsonLow(right, real),
|
|
29
|
+
};
|
|
30
|
+
}
|
|
31
|
+
/** §12: each code's accuracy, by code, and all codes together. */
|
|
32
|
+
export function detectorAccuracy(labels) {
|
|
33
|
+
const counts = new Map();
|
|
34
|
+
for (const l of labels) {
|
|
35
|
+
const c = counts.get(l.code) ?? { right: 0, wrong: 0, missed: 0 };
|
|
36
|
+
c[l.verdict]++;
|
|
37
|
+
counts.set(l.code, c);
|
|
38
|
+
}
|
|
39
|
+
const codes = [...counts.keys()].sort().map((code) => {
|
|
40
|
+
const c = counts.get(code);
|
|
41
|
+
return row(code, c.right, c.wrong, c.missed);
|
|
42
|
+
});
|
|
43
|
+
const sum = (k) => codes.reduce((n, c) => n + c[k], 0);
|
|
44
|
+
return { codes, overall: row("*", sum("right"), sum("wrong"), sum("missed")) };
|
|
45
|
+
}
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
import type { Bodies } from "../reconcile/record.ts";
|
|
2
|
+
import { type ReferencePack } from "./reference.ts";
|
|
3
|
+
export interface ClaimEntry {
|
|
4
|
+
index: number;
|
|
5
|
+
kind: string;
|
|
6
|
+
value_sha256: string;
|
|
7
|
+
status: "grounded" | "ungrounded" | "contradicted";
|
|
8
|
+
via: "pack" | "exact";
|
|
9
|
+
evidence: Array<{
|
|
10
|
+
seq: number;
|
|
11
|
+
line_hash: string;
|
|
12
|
+
}>;
|
|
13
|
+
pack?: {
|
|
14
|
+
id: string;
|
|
15
|
+
version: string;
|
|
16
|
+
hash: string;
|
|
17
|
+
fact: string;
|
|
18
|
+
};
|
|
19
|
+
}
|
|
20
|
+
export interface AnswerClaims {
|
|
21
|
+
seq: number;
|
|
22
|
+
line_hash: string;
|
|
23
|
+
text_sha256: string;
|
|
24
|
+
claims: ClaimEntry[];
|
|
25
|
+
service?: {
|
|
26
|
+
seq: number;
|
|
27
|
+
line_hash: string;
|
|
28
|
+
outage: true;
|
|
29
|
+
} | {
|
|
30
|
+
seq: number;
|
|
31
|
+
line_hash: string;
|
|
32
|
+
model?: string;
|
|
33
|
+
version?: string;
|
|
34
|
+
claims: Array<{
|
|
35
|
+
index: number;
|
|
36
|
+
status: string;
|
|
37
|
+
text_sha256: string;
|
|
38
|
+
score?: number;
|
|
39
|
+
}>;
|
|
40
|
+
};
|
|
41
|
+
sources: Array<{
|
|
42
|
+
seq: number;
|
|
43
|
+
id: string;
|
|
44
|
+
uri?: string;
|
|
45
|
+
version?: string;
|
|
46
|
+
sha256: string;
|
|
47
|
+
}>;
|
|
48
|
+
}
|
|
49
|
+
/** spec/findings.md §11: the claims of every final answer on a record. */
|
|
50
|
+
export declare function answerClaims(lines: readonly string[], bodies: Bodies, packs?: readonly ReferencePack[]): AnswerClaims[];
|
|
51
|
+
/** spec/findings.md §11: the credential for one answer, unsigned; null when `seq` isn't a final
|
|
52
|
+
* answer. The server signs it (`signCertificate`). */
|
|
53
|
+
export declare function answerCredential(lines: readonly string[], bodies: Bodies, seq: number, o: {
|
|
54
|
+
sessionId: string;
|
|
55
|
+
packs?: readonly ReferencePack[];
|
|
56
|
+
}): {
|
|
57
|
+
v: number;
|
|
58
|
+
type: string;
|
|
59
|
+
session_id: string;
|
|
60
|
+
answer: {
|
|
61
|
+
seq: number;
|
|
62
|
+
line_hash: string;
|
|
63
|
+
text_sha256: string;
|
|
64
|
+
};
|
|
65
|
+
claims: ClaimEntry[];
|
|
66
|
+
service?: {
|
|
67
|
+
seq: number;
|
|
68
|
+
line_hash: string;
|
|
69
|
+
outage: true;
|
|
70
|
+
} | {
|
|
71
|
+
seq: number;
|
|
72
|
+
line_hash: string;
|
|
73
|
+
model?: string;
|
|
74
|
+
version?: string;
|
|
75
|
+
claims: Array<{
|
|
76
|
+
index: number;
|
|
77
|
+
status: string;
|
|
78
|
+
text_sha256: string;
|
|
79
|
+
score?: number;
|
|
80
|
+
}>;
|
|
81
|
+
};
|
|
82
|
+
sources: {
|
|
83
|
+
seq: number;
|
|
84
|
+
id: string;
|
|
85
|
+
uri?: string;
|
|
86
|
+
version?: string;
|
|
87
|
+
sha256: string;
|
|
88
|
+
}[];
|
|
89
|
+
packs: {
|
|
90
|
+
id: string;
|
|
91
|
+
version: string;
|
|
92
|
+
hash: string;
|
|
93
|
+
}[];
|
|
94
|
+
issued_at: string;
|
|
95
|
+
} | null;
|
|
96
|
+
/** §11: a credential checked against the record it names: recomputed, then compared field by field
|
|
97
|
+
* (the signature is `verifyCertificate`'s job). */
|
|
98
|
+
export declare function verifyAnswerCredential(credential: Record<string, unknown>, lines: readonly string[], bodies: Bodies, packs?: readonly ReferencePack[]): {
|
|
99
|
+
ok: boolean;
|
|
100
|
+
problems: string[];
|
|
101
|
+
};
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
// The claim record and answer credentials (spec/findings.md §11, H3): for each final answer, every
|
|
2
|
+
// fact in it, how it was checked (a pack, the record, the grounding service) and the record lines
|
|
3
|
+
// that hold its evidence; and a credential for one answer that anyone holding the record can check.
|
|
4
|
+
// Mirrors sdks/python/src/zanii_blackbox/analysis/credential.py; pinned by
|
|
5
|
+
// spec/vectors/credential.json.
|
|
6
|
+
import { createHash } from "node:crypto";
|
|
7
|
+
import { canonical } from "../reconcile/shared.js";
|
|
8
|
+
import { hashLine } from "../verify/envelope.js";
|
|
9
|
+
import { answersOf, factsIn, parseGroundingAnswer } from "./grounding.js";
|
|
10
|
+
import { checkAgainstPacks } from "./reference.js";
|
|
11
|
+
const sha = (s) => `sha256:${createHash("sha256").update(s).digest("hex")}`;
|
|
12
|
+
const decoder = new TextDecoder();
|
|
13
|
+
/** spec/findings.md §11: the claims of every final answer on a record. */
|
|
14
|
+
export function answerClaims(lines, bodies, packs = []) {
|
|
15
|
+
const hashes = new Map();
|
|
16
|
+
const verdicts = new Map();
|
|
17
|
+
const retrieved = [];
|
|
18
|
+
for (const l of lines) {
|
|
19
|
+
const e = JSON.parse(l);
|
|
20
|
+
hashes.set(e.seq, hashLine(l));
|
|
21
|
+
if (e.kind === "control" &&
|
|
22
|
+
e.meta.action === "grounding" &&
|
|
23
|
+
typeof e.meta.answer_seq === "number")
|
|
24
|
+
verdicts.set(e.meta.answer_seq, e); // the latest verdict for an answer is its verdict
|
|
25
|
+
if (e.kind === "sdk.event" && e.meta.type === "retrieved") {
|
|
26
|
+
const b = bodies(e.body_hash);
|
|
27
|
+
try {
|
|
28
|
+
const d = b ? JSON.parse(decoder.decode(b)) : {};
|
|
29
|
+
if (Array.isArray(d.docs))
|
|
30
|
+
retrieved.push({ seq: e.seq, docs: d.docs });
|
|
31
|
+
}
|
|
32
|
+
catch { }
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
const at = (seq) => ({ seq, line_hash: hashes.get(seq) });
|
|
36
|
+
return answersOf(lines, bodies).map((a) => {
|
|
37
|
+
const claims = factsIn(a.text).map((f, index) => {
|
|
38
|
+
const base = { index, kind: f.kind, value_sha256: sha(f.value) };
|
|
39
|
+
const said = checkAgainstPacks(a.text, f, packs);
|
|
40
|
+
if (said) {
|
|
41
|
+
const pack = {
|
|
42
|
+
id: said.pack.id,
|
|
43
|
+
version: said.pack.version,
|
|
44
|
+
hash: said.pack.hash,
|
|
45
|
+
fact: said.fact.id,
|
|
46
|
+
};
|
|
47
|
+
return { ...base, status: said.status, via: "pack", evidence: [], pack };
|
|
48
|
+
}
|
|
49
|
+
const seqs = a.evidence.support(f);
|
|
50
|
+
return seqs
|
|
51
|
+
? { ...base, status: "grounded", via: "exact", evidence: seqs.map(at) }
|
|
52
|
+
: { ...base, status: "ungrounded", via: "exact", evidence: [] };
|
|
53
|
+
});
|
|
54
|
+
const v = verdicts.get(a.seq);
|
|
55
|
+
let service;
|
|
56
|
+
if (v?.meta.outage === true)
|
|
57
|
+
service = { ...at(v.seq), outage: true };
|
|
58
|
+
else if (v) {
|
|
59
|
+
const b = bodies(v.body_hash);
|
|
60
|
+
let parsed = null;
|
|
61
|
+
try {
|
|
62
|
+
parsed = b ? parseGroundingAnswer(JSON.parse(decoder.decode(b))) : null;
|
|
63
|
+
}
|
|
64
|
+
catch { }
|
|
65
|
+
if (parsed)
|
|
66
|
+
service = {
|
|
67
|
+
...at(v.seq),
|
|
68
|
+
...(parsed.model ? { model: parsed.model } : {}),
|
|
69
|
+
...(parsed.version ? { version: parsed.version } : {}),
|
|
70
|
+
claims: parsed.claims.map((c, index) => ({
|
|
71
|
+
index,
|
|
72
|
+
status: c.status,
|
|
73
|
+
text_sha256: sha(c.text),
|
|
74
|
+
...(c.score !== undefined ? { score: c.score } : {}),
|
|
75
|
+
})),
|
|
76
|
+
};
|
|
77
|
+
}
|
|
78
|
+
const sources = retrieved
|
|
79
|
+
.filter((r) => r.seq < a.seq)
|
|
80
|
+
.flatMap((r) => r.docs.flatMap((d) => {
|
|
81
|
+
const o = d;
|
|
82
|
+
if (typeof o.id !== "string" || typeof o.sha256 !== "string")
|
|
83
|
+
return [];
|
|
84
|
+
return [
|
|
85
|
+
{
|
|
86
|
+
seq: r.seq,
|
|
87
|
+
id: o.id,
|
|
88
|
+
...(typeof o.uri === "string" ? { uri: o.uri } : {}),
|
|
89
|
+
...(typeof o.version === "string" ? { version: o.version } : {}),
|
|
90
|
+
sha256: o.sha256,
|
|
91
|
+
},
|
|
92
|
+
];
|
|
93
|
+
}));
|
|
94
|
+
return {
|
|
95
|
+
...at(a.seq),
|
|
96
|
+
text_sha256: sha(a.text),
|
|
97
|
+
claims,
|
|
98
|
+
...(service ? { service } : {}),
|
|
99
|
+
sources,
|
|
100
|
+
};
|
|
101
|
+
});
|
|
102
|
+
}
|
|
103
|
+
/** spec/findings.md §11: the credential for one answer, unsigned; null when `seq` isn't a final
|
|
104
|
+
* answer. The server signs it (`signCertificate`). */
|
|
105
|
+
export function answerCredential(lines, bodies, seq, o) {
|
|
106
|
+
const packs = o.packs ?? [];
|
|
107
|
+
const entry = answerClaims(lines, bodies, packs).find((x) => x.seq === seq);
|
|
108
|
+
if (!entry)
|
|
109
|
+
return null;
|
|
110
|
+
const ts = lines.map((l) => JSON.parse(l)).find((e) => e.seq === seq)?.ts;
|
|
111
|
+
return {
|
|
112
|
+
v: 1,
|
|
113
|
+
type: "answer_credential",
|
|
114
|
+
session_id: o.sessionId,
|
|
115
|
+
answer: { seq: entry.seq, line_hash: entry.line_hash, text_sha256: entry.text_sha256 },
|
|
116
|
+
claims: entry.claims,
|
|
117
|
+
...(entry.service ? { service: entry.service } : {}),
|
|
118
|
+
sources: entry.sources,
|
|
119
|
+
packs: packs.map((p) => ({ id: p.id, version: p.version, hash: p.hash })),
|
|
120
|
+
issued_at: ts,
|
|
121
|
+
};
|
|
122
|
+
}
|
|
123
|
+
/** §11: a credential checked against the record it names: recomputed, then compared field by field
|
|
124
|
+
* (the signature is `verifyCertificate`'s job). */
|
|
125
|
+
export function verifyAnswerCredential(credential, lines, bodies, packs = []) {
|
|
126
|
+
const answer = credential.answer;
|
|
127
|
+
if (typeof answer?.seq !== "number" || typeof credential.session_id !== "string")
|
|
128
|
+
return { ok: false, problems: ["malformed"] };
|
|
129
|
+
const again = answerCredential(lines, bodies, answer.seq, {
|
|
130
|
+
sessionId: credential.session_id,
|
|
131
|
+
packs,
|
|
132
|
+
});
|
|
133
|
+
if (!again)
|
|
134
|
+
return { ok: false, problems: ["the record has no such answer"] };
|
|
135
|
+
const problems = [];
|
|
136
|
+
const fields = new Set([...Object.keys(again), ...Object.keys(credential)]);
|
|
137
|
+
fields.delete("signed");
|
|
138
|
+
for (const k of [...fields].sort())
|
|
139
|
+
if (canonical(again[k] ?? null) !== canonical(credential[k] ?? null))
|
|
140
|
+
problems.push(`${k} differs from the record`);
|
|
141
|
+
return { ok: problems.length === 0, problems };
|
|
142
|
+
}
|
package/dist/analysis/faults.js
CHANGED
|
@@ -28,6 +28,26 @@ export const FAULTS = {
|
|
|
28
28
|
en: "File change outside any tool call",
|
|
29
29
|
ar: "تغيير في الملفات خارج أي استدعاء أداة",
|
|
30
30
|
},
|
|
31
|
+
INTEGRITY_FAILED: {
|
|
32
|
+
fault: "BBX-1107",
|
|
33
|
+
en: "A stored record failed the self-check",
|
|
34
|
+
ar: "فشل سجل محفوظ في الفحص الذاتي",
|
|
35
|
+
},
|
|
36
|
+
ANCHOR_STALE: {
|
|
37
|
+
fault: "BBX-1108",
|
|
38
|
+
en: "Anchoring has stalled",
|
|
39
|
+
ar: "توقّف تثبيت السجلات",
|
|
40
|
+
},
|
|
41
|
+
LEDGER_VIOLATION: {
|
|
42
|
+
fault: "BBX-1109",
|
|
43
|
+
en: "The ledger's log was rewritten",
|
|
44
|
+
ar: "أُعيدت كتابة سجل دفتر الأستاذ",
|
|
45
|
+
},
|
|
46
|
+
ANCHOR_REJECTED: {
|
|
47
|
+
fault: "BBX-1110",
|
|
48
|
+
en: "The ledger rejected an anchor",
|
|
49
|
+
ar: "رفض دفتر الأستاذ تثبيتًا",
|
|
50
|
+
},
|
|
31
51
|
SDK_SILENT: {
|
|
32
52
|
fault: "BBX-1201",
|
|
33
53
|
en: "The SDK stopped reporting",
|
|
@@ -102,6 +122,46 @@ export const FAULTS = {
|
|
|
102
122
|
en: "A check that passed now fails",
|
|
103
123
|
ar: "فحص نجح سابقًا ثم فشل",
|
|
104
124
|
},
|
|
125
|
+
UNKNOWN_TOOL: {
|
|
126
|
+
fault: "BBX-4301",
|
|
127
|
+
en: "A call to a tool that doesn't exist",
|
|
128
|
+
ar: "استدعاء أداة غير موجودة",
|
|
129
|
+
},
|
|
130
|
+
TOOL_ARGS_INVALID: {
|
|
131
|
+
fault: "BBX-4302",
|
|
132
|
+
en: "Tool arguments its schema refuses",
|
|
133
|
+
ar: "وسائط أداة يرفضها مخططها",
|
|
134
|
+
},
|
|
135
|
+
FABRICATED_VALUE: {
|
|
136
|
+
fault: "BBX-4303",
|
|
137
|
+
en: "A value the agent never saw",
|
|
138
|
+
ar: "قيمة لم يطّلع عليها الوكيل",
|
|
139
|
+
},
|
|
140
|
+
SUCCESS_AFTER_ERROR: {
|
|
141
|
+
fault: "BBX-4304",
|
|
142
|
+
en: "Success claimed right after an error",
|
|
143
|
+
ar: "نجاح مُعلن بعد خطأ مباشرة",
|
|
144
|
+
},
|
|
145
|
+
PHANTOM_RESULT: {
|
|
146
|
+
fault: "BBX-4305",
|
|
147
|
+
en: "A result quoted from a tool never called",
|
|
148
|
+
ar: "نتيجة منسوبة إلى أداة لم تُستدعَ",
|
|
149
|
+
},
|
|
150
|
+
UNGROUNDED_CLAIM: {
|
|
151
|
+
fault: "BBX-4401",
|
|
152
|
+
en: "A fact in an answer nothing supports",
|
|
153
|
+
ar: "معلومة في إجابة لا يدعمها شيء",
|
|
154
|
+
},
|
|
155
|
+
CONTRADICTED_CLAIM: {
|
|
156
|
+
fault: "BBX-4402",
|
|
157
|
+
en: "A fact in an answer the evidence contradicts",
|
|
158
|
+
ar: "معلومة في إجابة يناقضها الدليل",
|
|
159
|
+
},
|
|
160
|
+
JUDGE_FAILED: {
|
|
161
|
+
fault: "BBX-4501",
|
|
162
|
+
en: "A judge's rubric step failed",
|
|
163
|
+
ar: "خطوة من معايير الحكم لم تتحقق",
|
|
164
|
+
},
|
|
105
165
|
CLAIM_UNVERIFIED: {
|
|
106
166
|
fault: "BBX-4202",
|
|
107
167
|
en: "Completion claim can't be verified",
|
|
@@ -117,6 +177,41 @@ export const FAULTS = {
|
|
|
117
177
|
en: "Tool use against policy",
|
|
118
178
|
ar: "استخدام أداة مخالف للسياسة",
|
|
119
179
|
},
|
|
180
|
+
WALL_CROSSED: {
|
|
181
|
+
fault: "BBX-5104",
|
|
182
|
+
en: "An answer crossed a regulator's wall",
|
|
183
|
+
ar: "تجاوزت إجابة حدود جهة رقابية",
|
|
184
|
+
},
|
|
185
|
+
UNSCREENED_COUNTERPARTY: {
|
|
186
|
+
fault: "BBX-9401",
|
|
187
|
+
en: "A payment to an unscreened counterparty",
|
|
188
|
+
ar: "دفعة إلى طرف لم يُفحص",
|
|
189
|
+
},
|
|
190
|
+
SCREENING_HIT_PAID: {
|
|
191
|
+
fault: "BBX-9402",
|
|
192
|
+
en: "A flagged counterparty was paid",
|
|
193
|
+
ar: "دُفع لطرف عليه إنذار في الفحص",
|
|
194
|
+
},
|
|
195
|
+
FTA_NO_HANDOFF: {
|
|
196
|
+
fault: "BBX-9403",
|
|
197
|
+
en: "A tax filing never reached a Tax Agent",
|
|
198
|
+
ar: "لم يصل إقرار ضريبي إلى وكيل ضريبي",
|
|
199
|
+
},
|
|
200
|
+
BREAK_GLASS: {
|
|
201
|
+
fault: "BBX-9501",
|
|
202
|
+
en: "Emergency access to a record (break-glass)",
|
|
203
|
+
ar: "وصول طارئ إلى سجل (كسر الزجاج)",
|
|
204
|
+
},
|
|
205
|
+
HEALTH_SIGNATURE_INVALID: {
|
|
206
|
+
fault: "BBX-9502",
|
|
207
|
+
en: "A clinician signature doesn't check",
|
|
208
|
+
ar: "توقيع طبيب لا يصح",
|
|
209
|
+
},
|
|
210
|
+
UNCONFIRMED_RECOMMENDATION: {
|
|
211
|
+
fault: "BBX-9503",
|
|
212
|
+
en: "No clinician confirmed a recommendation",
|
|
213
|
+
ar: "لم يؤكد أي طبيب توصية",
|
|
214
|
+
},
|
|
120
215
|
POLICY_AUDIT: {
|
|
121
216
|
fault: "BBX-5103",
|
|
122
217
|
en: "An audit-only policy rule would have acted",
|
|
@@ -132,11 +227,21 @@ export const FAULTS = {
|
|
|
132
227
|
en: "The session token was sent onward",
|
|
133
228
|
ar: "أُرسل رمز الجلسة إلى جهة أخرى",
|
|
134
229
|
},
|
|
230
|
+
TOOL_CHANGED: {
|
|
231
|
+
fault: "BBX-7402",
|
|
232
|
+
en: "A tool's definition changed mid-session",
|
|
233
|
+
ar: "تغيّر تعريف أداة أثناء الجلسة",
|
|
234
|
+
},
|
|
135
235
|
REVOKED_MEMORY_READ: {
|
|
136
236
|
fault: "BBX-7101",
|
|
137
237
|
en: "A revoked memory was used",
|
|
138
238
|
ar: "استُخدمت ذاكرة ملغاة",
|
|
139
239
|
},
|
|
240
|
+
MEMORY_CHAIN_BROKEN: {
|
|
241
|
+
fault: "BBX-7102",
|
|
242
|
+
en: "A memory entry was changed or is out of order",
|
|
243
|
+
ar: "عُدّل إدخال ذاكرة أو جاء خارج الترتيب",
|
|
244
|
+
},
|
|
140
245
|
TOOL_OUTPUT_MISMATCH: {
|
|
141
246
|
fault: "BBX-7201",
|
|
142
247
|
en: "Tool output doesn't match a re-run",
|
|
@@ -188,6 +293,11 @@ export const FAULTS = {
|
|
|
188
293
|
en: "An action waits for a second person",
|
|
189
294
|
ar: "إجراء بانتظار شخص ثانٍ",
|
|
190
295
|
},
|
|
296
|
+
APPROVAL_ARGS_CHANGED: {
|
|
297
|
+
fault: "BBX-8404",
|
|
298
|
+
en: "An approved tool ran with other arguments",
|
|
299
|
+
ar: "شُغّلت أداة معتمدة بمعطيات أخرى",
|
|
300
|
+
},
|
|
191
301
|
PREFLIGHT_DEGRADED: {
|
|
192
302
|
fault: "BBX-8501",
|
|
193
303
|
en: "Started with equipment missing",
|
|
@@ -238,6 +348,11 @@ export const FAULTS = {
|
|
|
238
348
|
en: "A tool's signature doesn't verify",
|
|
239
349
|
ar: "توقيع الأداة غير صالح",
|
|
240
350
|
},
|
|
351
|
+
UNPROVEN_SIDE_EFFECT: {
|
|
352
|
+
fault: "BBX-9203",
|
|
353
|
+
en: "A change the tool didn't prove",
|
|
354
|
+
ar: "تغيير لم تُثبته الأداة",
|
|
355
|
+
},
|
|
241
356
|
EGRESS_OPEN: {
|
|
242
357
|
fault: "BBX-9301",
|
|
243
358
|
en: "The agent can reach the internet around the gateway",
|