omk-adaptorch-wpl 0.91.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +49 -0
- package/dist/adaptorch-client.d.ts +184 -0
- package/dist/adaptorch-client.d.ts.map +1 -0
- package/dist/adaptorch-client.js +109 -0
- package/dist/adaptorch-client.js.map +1 -0
- package/dist/adjudicator-registry.d.ts +93 -0
- package/dist/adjudicator-registry.d.ts.map +1 -0
- package/dist/adjudicator-registry.js +89 -0
- package/dist/adjudicator-registry.js.map +1 -0
- package/dist/adjudicator.d.ts +73 -0
- package/dist/adjudicator.d.ts.map +1 -0
- package/dist/adjudicator.js +356 -0
- package/dist/adjudicator.js.map +1 -0
- package/dist/b2c-mapper.d.ts +25 -0
- package/dist/b2c-mapper.d.ts.map +1 -0
- package/dist/b2c-mapper.js +186 -0
- package/dist/b2c-mapper.js.map +1 -0
- package/dist/b2c-verdict.d.ts +66 -0
- package/dist/b2c-verdict.d.ts.map +1 -0
- package/dist/b2c-verdict.js +5 -0
- package/dist/b2c-verdict.js.map +1 -0
- package/dist/deep-wall.d.ts +80 -0
- package/dist/deep-wall.d.ts.map +1 -0
- package/dist/deep-wall.js +97 -0
- package/dist/deep-wall.js.map +1 -0
- package/dist/evaluate-correctness-wall.d.ts +35 -0
- package/dist/evaluate-correctness-wall.d.ts.map +1 -0
- package/dist/evaluate-correctness-wall.js +128 -0
- package/dist/evaluate-correctness-wall.js.map +1 -0
- package/dist/in-memory-adaptorch.d.ts +12 -0
- package/dist/in-memory-adaptorch.d.ts.map +1 -0
- package/dist/in-memory-adaptorch.js +26 -0
- package/dist/in-memory-adaptorch.js.map +1 -0
- package/dist/index.d.ts +46 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +28 -0
- package/dist/index.js.map +1 -0
- package/dist/live-adaptorch-transport.d.ts +14 -0
- package/dist/live-adaptorch-transport.d.ts.map +1 -0
- package/dist/live-adaptorch-transport.js +19 -0
- package/dist/live-adaptorch-transport.js.map +1 -0
- package/dist/loop.d.ts +149 -0
- package/dist/loop.d.ts.map +1 -0
- package/dist/loop.js +219 -0
- package/dist/loop.js.map +1 -0
- package/dist/mcp-introspection-transport.d.ts +18 -0
- package/dist/mcp-introspection-transport.d.ts.map +1 -0
- package/dist/mcp-introspection-transport.js +29 -0
- package/dist/mcp-introspection-transport.js.map +1 -0
- package/dist/policy-wall.d.ts +34 -0
- package/dist/policy-wall.d.ts.map +1 -0
- package/dist/policy-wall.js +125 -0
- package/dist/policy-wall.js.map +1 -0
- package/dist/receipt-signature.d.ts +28 -0
- package/dist/receipt-signature.d.ts.map +1 -0
- package/dist/receipt-signature.js +30 -0
- package/dist/receipt-signature.js.map +1 -0
- package/dist/regenerate-packet.d.ts +25 -0
- package/dist/regenerate-packet.d.ts.map +1 -0
- package/dist/regenerate-packet.js +29 -0
- package/dist/regenerate-packet.js.map +1 -0
- package/dist/repair-budget.d.ts +24 -0
- package/dist/repair-budget.d.ts.map +1 -0
- package/dist/repair-budget.js +44 -0
- package/dist/repair-budget.js.map +1 -0
- package/dist/repair-loop.d.ts +18 -0
- package/dist/repair-loop.d.ts.map +1 -0
- package/dist/repair-loop.js +54 -0
- package/dist/repair-loop.js.map +1 -0
- package/dist/retry-backoff.d.ts +56 -0
- package/dist/retry-backoff.d.ts.map +1 -0
- package/dist/retry-backoff.js +75 -0
- package/dist/retry-backoff.js.map +1 -0
- package/dist/signed-receipt.d.ts +29 -0
- package/dist/signed-receipt.d.ts.map +1 -0
- package/dist/signed-receipt.js +25 -0
- package/dist/signed-receipt.js.map +1 -0
- package/dist/state-machine.d.ts +40 -0
- package/dist/state-machine.d.ts.map +1 -0
- package/dist/state-machine.js +88 -0
- package/dist/state-machine.js.map +1 -0
- package/dist/types.d.ts +149 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +16 -0
- package/dist/types.js.map +1 -0
- package/dist/wall-meta.d.ts +3 -0
- package/dist/wall-meta.d.ts.map +1 -0
- package/dist/wall-meta.js +3 -0
- package/dist/wall-meta.js.map +1 -0
- package/package.json +46 -0
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Outcome Adjudicator (Part 2 sections 2, 3, 5) - a wrapper-side verification layer that
|
|
3
|
+
* turns AdaptOrch's own terminal-status report into a corroborated 5-state verdict, using
|
|
4
|
+
* only `getRun`/`getArtifacts`/`getTraces` introspection.
|
|
5
|
+
*
|
|
6
|
+
* Out of scope for this file (not covered by the input contract available here): the
|
|
7
|
+
* scope check and freshness check from Part 2 section 2.3, and the retry-count threshold
|
|
8
|
+
* from section 2.3, since none of `AdjudicationRequest`, `VerifierRegistryEntry`, or the
|
|
9
|
+
* client return shapes carry lane-scope, artifact-timestamp, or run-start-time data to
|
|
10
|
+
* check them against. What is implemented here: the non-empty check (honoring
|
|
11
|
+
* `allow_zero_artifacts`), the section 3 ambiguous-data fork, `content_check`,
|
|
12
|
+
* `trace_check`, an error-span heuristic scan, and the `expected_min_actions` count.
|
|
13
|
+
*/
|
|
14
|
+
import type { AdaptOrchClient } from "./adaptorch-client.ts";
|
|
15
|
+
import type { AdjudicationReasonCode, VerdictState, VerifierRegistryEntry } from "./adjudicator-registry.ts";
|
|
16
|
+
/**
|
|
17
|
+
* Adjudication input (Part 2 section 2.1), sourced from the Dispatch Record and Work
|
|
18
|
+
* Packet on the core-algorithm side. Both fields are mandatory; `run_ids` must be
|
|
19
|
+
* non-empty (an empty list is treated as a malformed request, section 2.1/2.4).
|
|
20
|
+
*/
|
|
21
|
+
export interface AdjudicationRequest {
|
|
22
|
+
dispatch_record_id: string;
|
|
23
|
+
kind: string;
|
|
24
|
+
run_ids: string[];
|
|
25
|
+
}
|
|
26
|
+
/**
|
|
27
|
+
* A single `run_id`'s own verdict, reason, and evidence references (Part 2 section 5).
|
|
28
|
+
* Preserved verbatim inside the record-level {@link AdjudicationResult}; never summarized
|
|
29
|
+
* away, even when the request has only one `run_id`.
|
|
30
|
+
*/
|
|
31
|
+
export interface PerRunVerdict {
|
|
32
|
+
run_id: string;
|
|
33
|
+
verdict: VerdictState;
|
|
34
|
+
/** Machine-readable classification; disposition logic branches on this, never on `reason`. */
|
|
35
|
+
reason_code: AdjudicationReasonCode;
|
|
36
|
+
/** Human-readable explanation, for logs and review UIs only. */
|
|
37
|
+
reason: string;
|
|
38
|
+
evidence_refs: unknown;
|
|
39
|
+
}
|
|
40
|
+
/**
|
|
41
|
+
* Record-level adjudication output (Part 2 section 2.5 aggregation; section 5 record
|
|
42
|
+
* shape). `augmented_payload` is present only when the looked-up registry entry defines
|
|
43
|
+
* `build_augmented_payload` and the record-level verdict is not CONFIRMED (section 4).
|
|
44
|
+
*/
|
|
45
|
+
export interface AdjudicationResult {
|
|
46
|
+
verdict: VerdictState;
|
|
47
|
+
/**
|
|
48
|
+
* Machine-readable classification, reduced worst-wins from the `per_run` entries that
|
|
49
|
+
* share the record-level verdict. `projectVerdictToDisposition` branches on this code
|
|
50
|
+
* exclusively; `reason` is never string-matched.
|
|
51
|
+
*/
|
|
52
|
+
reason_code: AdjudicationReasonCode;
|
|
53
|
+
/** Human-readable explanation, for logs and review UIs only. */
|
|
54
|
+
reason: string;
|
|
55
|
+
per_run: PerRunVerdict[];
|
|
56
|
+
augmented_payload?: unknown;
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* Adjudicates an `AdjudicationRequest` end to end (Part 2 sections 2.1-2.5, 3, 5).
|
|
60
|
+
*
|
|
61
|
+
* 1. Rejects malformed requests (missing `kind` or empty `run_ids`) as a record-level
|
|
62
|
+
* VERIFIER-ERROR with no per-`run_id` results (section 2.1/2.4).
|
|
63
|
+
* 2. Resolves the per-kind verifier once (section 4) and applies it identically to every
|
|
64
|
+
* `run_id`, independently (section 2.2-2.4, 3).
|
|
65
|
+
* 3. Reduces the per-`run_id` verdicts to one record-level verdict, worst-wins (section
|
|
66
|
+
* 2.5), and never discards the per-`run_id` detail (section 5).
|
|
67
|
+
* 4. Invokes `build_augmented_payload`, if defined, when the record-level verdict is not
|
|
68
|
+
* CONFIRMED (section 4).
|
|
69
|
+
*/
|
|
70
|
+
export declare function adjudicate(request: AdjudicationRequest, client: AdaptOrchClient, registry: {
|
|
71
|
+
get(kind: string): VerifierRegistryEntry;
|
|
72
|
+
}): Promise<AdjudicationResult>;
|
|
73
|
+
//# sourceMappingURL=adjudicator.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"adjudicator.d.ts","sourceRoot":"","sources":["../src/adjudicator.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,uBAAuB,CAAC;AAC7D,OAAO,KAAK,EACX,sBAAsB,EAEtB,YAAY,EACZ,qBAAqB,EACrB,MAAM,2BAA2B,CAAC;AAUnC;;;;GAIG;AACH,MAAM,WAAW,mBAAmB;IACnC,kBAAkB,EAAE,MAAM,CAAC;IAC3B,IAAI,EAAE,MAAM,CAAC;IACb,OAAO,EAAE,MAAM,EAAE,CAAC;CAClB;AAED;;;;GAIG;AACH,MAAM,WAAW,aAAa;IAC7B,MAAM,EAAE,MAAM,CAAC;IACf,OAAO,EAAE,YAAY,CAAC;IACtB,8FAA8F;IAC9F,WAAW,EAAE,sBAAsB,CAAC;IACpC,gEAAgE;IAChE,MAAM,EAAE,MAAM,CAAC;IACf,aAAa,EAAE,OAAO,CAAC;CACvB;AAED;;;;GAIG;AACH,MAAM,WAAW,kBAAkB;IAClC,OAAO,EAAE,YAAY,CAAC;IACtB;;;;OAIG;IACH,WAAW,EAAE,sBAAsB,CAAC;IACpC,gEAAgE;IAChE,MAAM,EAAE,MAAM,CAAC;IACf,OAAO,EAAE,aAAa,EAAE,CAAC;IACzB,iBAAiB,CAAC,EAAE,OAAO,CAAC;CAC5B;AAyTD;;;;;;;;;;;GAWG;AACH,wBAAsB,UAAU,CAC/B,OAAO,EAAE,mBAAmB,EAC5B,MAAM,EAAE,eAAe,EACvB,QAAQ,EAAE;IAAE,GAAG,CAAC,IAAI,EAAE,MAAM,GAAG,qBAAqB,CAAA;CAAE,GACpD,OAAO,CAAC,kBAAkB,CAAC,CA0C7B","sourcesContent":["/**\n * Outcome Adjudicator (Part 2 sections 2, 3, 5) - a wrapper-side verification layer that\n * turns AdaptOrch's own terminal-status report into a corroborated 5-state verdict, using\n * only `getRun`/`getArtifacts`/`getTraces` introspection.\n *\n * Out of scope for this file (not covered by the input contract available here): the\n * scope check and freshness check from Part 2 section 2.3, and the retry-count threshold\n * from section 2.3, since none of `AdjudicationRequest`, `VerifierRegistryEntry`, or the\n * client return shapes carry lane-scope, artifact-timestamp, or run-start-time data to\n * check them against. What is implemented here: the non-empty check (honoring\n * `allow_zero_artifacts`), the section 3 ambiguous-data fork, `content_check`,\n * `trace_check`, an error-span heuristic scan, and the `expected_min_actions` count.\n */\n\nimport type { AdaptOrchClient } from \"./adaptorch-client.ts\";\nimport type {\n\tAdjudicationReasonCode,\n\tCheckResult,\n\tVerdictState,\n\tVerifierRegistryEntry,\n} from \"./adjudicator-registry.ts\";\nimport { reduceReasonCodes } from \"./adjudicator-registry.ts\";\n\n/** Raw payload shape returned by `AdaptOrchClient.getRun`, inferred rather than duplicated. */\ntype RunPayload = Awaited<ReturnType<AdaptOrchClient[\"getRun\"]>>;\n/** Raw payload shape returned by `AdaptOrchClient.getArtifacts`, inferred rather than duplicated. */\ntype ArtifactsPayload = Awaited<ReturnType<AdaptOrchClient[\"getArtifacts\"]>>;\n/** Raw payload shape returned by `AdaptOrchClient.getTraces`, inferred rather than duplicated. */\ntype TracesPayload = Awaited<ReturnType<AdaptOrchClient[\"getTraces\"]>>;\n\n/**\n * Adjudication input (Part 2 section 2.1), sourced from the Dispatch Record and Work\n * Packet on the core-algorithm side. Both fields are mandatory; `run_ids` must be\n * non-empty (an empty list is treated as a malformed request, section 2.1/2.4).\n */\nexport interface AdjudicationRequest {\n\tdispatch_record_id: string;\n\tkind: string;\n\trun_ids: string[];\n}\n\n/**\n * A single `run_id`'s own verdict, reason, and evidence references (Part 2 section 5).\n * Preserved verbatim inside the record-level {@link AdjudicationResult}; never summarized\n * away, even when the request has only one `run_id`.\n */\nexport interface PerRunVerdict {\n\trun_id: string;\n\tverdict: VerdictState;\n\t/** Machine-readable classification; disposition logic branches on this, never on `reason`. */\n\treason_code: AdjudicationReasonCode;\n\t/** Human-readable explanation, for logs and review UIs only. */\n\treason: string;\n\tevidence_refs: unknown;\n}\n\n/**\n * Record-level adjudication output (Part 2 section 2.5 aggregation; section 5 record\n * shape). `augmented_payload` is present only when the looked-up registry entry defines\n * `build_augmented_payload` and the record-level verdict is not CONFIRMED (section 4).\n */\nexport interface AdjudicationResult {\n\tverdict: VerdictState;\n\t/**\n\t * Machine-readable classification, reduced worst-wins from the `per_run` entries that\n\t * share the record-level verdict. `projectVerdictToDisposition` branches on this code\n\t * exclusively; `reason` is never string-matched.\n\t */\n\treason_code: AdjudicationReasonCode;\n\t/** Human-readable explanation, for logs and review UIs only. */\n\treason: string;\n\tper_run: PerRunVerdict[];\n\taugmented_payload?: unknown;\n}\n\n/** Snapshot of the raw payloads a single `run_id`'s verdict was computed from (section 5). */\ninterface RunEvidence {\n\trun?: RunPayload;\n\tartifacts?: ArtifactsPayload;\n\ttraces?: TracesPayload;\n}\n\ninterface RunOutcome {\n\trun_id: string;\n\tverdict: VerdictState;\n\treason_code: AdjudicationReasonCode;\n\treason: string;\n\tevidence: RunEvidence;\n}\n\nconst NON_TERMINAL_STATUSES = new Set([\n\t\"pending\",\n\t\"queued\",\n\t\"running\",\n\t\"in_progress\",\n\t\"in-progress\",\n\t\"scheduled\",\n\t\"dispatched\",\n\t\"starting\",\n\t\"initializing\",\n]);\nconst SUCCESS_STATUSES = new Set([\"completed\", \"success\", \"succeeded\", \"done\", \"ok\", \"finished\"]);\nconst FAILURE_STATUSES = new Set([\n\t\"failed\",\n\t\"failure\",\n\t\"error\",\n\t\"errored\",\n\t\"cancelled\",\n\t\"canceled\",\n\t\"timeout\",\n\t\"timed_out\",\n\t\"aborted\",\n]);\n\ntype RunStatusBranch = \"non-terminal\" | \"success\" | \"failure\" | \"unparseable\";\n\nfunction isRecord(value: unknown): value is Record<string, unknown> {\n\treturn typeof value === \"object\" && value !== null;\n}\n\nfunction readStringField(value: unknown, keys: string[]): string | undefined {\n\tif (!isRecord(value)) return undefined;\n\tfor (const key of keys) {\n\t\tconst field = value[key];\n\t\tif (typeof field === \"string\") return field;\n\t}\n\treturn undefined;\n}\n\n/**\n * Interprets a `getRun` payload into the branch the OA cares about (Part 2 section 2.2).\n * A payload with no recognizable string status field is treated as unparseable - a\n * checker failure, not a target-task failure (section 2.4, VERIFIER-ERROR).\n */\nfunction interpretRunStatus(run: unknown): RunStatusBranch {\n\tconst status = readStringField(run, [\"status\", \"state\"]);\n\tif (status === undefined) return \"unparseable\";\n\tconst normalized = status.toLowerCase();\n\tif (NON_TERMINAL_STATUSES.has(normalized)) return \"non-terminal\";\n\tif (SUCCESS_STATUSES.has(normalized)) return \"success\";\n\tif (FAILURE_STATUSES.has(normalized)) return \"failure\";\n\treturn \"unparseable\";\n}\n\n/**\n * Coerces an unknown `getArtifacts`/`getTraces` payload into a flat list without assuming\n * a specific shape (the concrete return type belongs to the concurrently-authored\n * `adaptorch-client.ts`). Falls back to common pagination-style container keys, then to\n * treating a single non-null value as a one-item list.\n */\nfunction asList(value: unknown): unknown[] {\n\tif (value === null || value === undefined) return [];\n\tif (Array.isArray(value)) return value;\n\tif (isRecord(value)) {\n\t\tfor (const key of [\"items\", \"artifacts\", \"traces\", \"spans\", \"entries\", \"data\", \"results\"]) {\n\t\t\tconst field = value[key];\n\t\t\tif (Array.isArray(field)) return field;\n\t\t}\n\t}\n\treturn [value];\n}\n\n/** True unless the item is a recognizably empty/whitespace-only value (Part 2 section 2.3). */\nfunction hasSubstance(item: unknown): boolean {\n\tif (typeof item === \"string\") return item.trim().length > 0;\n\tif (isRecord(item)) {\n\t\tconst size = item.size ?? item.byteLength ?? item.length;\n\t\tif (typeof size === \"number\") return size > 0;\n\t\tconst text = readStringField(item, [\"content\", \"text\", \"body\"]);\n\t\tif (text !== undefined) return text.trim().length > 0;\n\t}\n\treturn true;\n}\n\n/** Heuristic ERROR-severity span scan (Part 2 section 2.3), tolerant of unknown span shapes. */\nfunction countErrorSpans(traces: unknown[]): number {\n\tlet count = 0;\n\tfor (const span of traces) {\n\t\tif (!isRecord(span)) continue;\n\t\tconst level = readStringField(span, [\"level\", \"severity\", \"status\"]);\n\t\tif (level !== undefined && level.toLowerCase() === \"error\") {\n\t\t\tcount += 1;\n\t\t\tcontinue;\n\t\t}\n\t\tif (span.error === true || span.isError === true) count += 1;\n\t}\n\treturn count;\n}\n\n/**\n * Heuristic count of \"action\" spans for the `expected_min_actions` check (Part 2 sections\n * 2.3/4). Recognizes a handful of common marker fields; if none of the spans carry any of\n * them, falls back to the total span count so the check degrades to a coarse presence\n * signal instead of always failing.\n */\nfunction countActionSpans(traces: unknown[]): number {\n\tlet recognized = 0;\n\tlet matched = 0;\n\tfor (const span of traces) {\n\t\tif (!isRecord(span)) continue;\n\t\tconst marker = readStringField(span, [\"action\", \"tool_call\", \"toolCall\", \"kind\", \"type\"]);\n\t\tif (marker !== undefined) {\n\t\t\trecognized += 1;\n\t\t\tif (/action|tool|write|edit|call/i.test(marker)) matched += 1;\n\t\t}\n\t}\n\treturn recognized > 0 ? matched : traces.length;\n}\n\nfunction describeError(error: unknown): string {\n\treturn error instanceof Error ? error.message : String(error);\n}\n\n/**\n * Adjudicates a single `run_id` against the resolved registry entry (Part 2 sections\n * 2.2-2.4 and 3). Robust to a misbehaving fetch: any thrown error from the client during\n * this run's own fetch/check sequence becomes that run_id's own VERIFIER-ERROR rather\n * than propagating out of {@link adjudicate}.\n */\nasync function adjudicateRun(\n\trunId: string,\n\tentry: VerifierRegistryEntry,\n\tclient: AdaptOrchClient,\n): Promise<RunOutcome> {\n\tconst evidence: RunEvidence = {};\n\ttry {\n\t\tconst run = await client.getRun(runId);\n\t\tevidence.run = run;\n\n\t\tconst branch = interpretRunStatus(run);\n\t\tif (branch === \"unparseable\") {\n\t\t\treturn {\n\t\t\t\trun_id: runId,\n\t\t\t\tverdict: \"VERIFIER-ERROR\",\n\t\t\t\treason_code: \"RUN_STATUS_UNPARSEABLE\",\n\t\t\t\treason: \"run-status-unparseable\",\n\t\t\t\tevidence,\n\t\t\t};\n\t\t}\n\t\tif (branch === \"non-terminal\") {\n\t\t\treturn {\n\t\t\t\trun_id: runId,\n\t\t\t\tverdict: \"INDETERMINATE\",\n\t\t\t\treason_code: \"RUN_NOT_TERMINAL\",\n\t\t\t\treason: \"run-not-terminal\",\n\t\t\t\tevidence,\n\t\t\t};\n\t\t}\n\n\t\tconst artifacts = await client.getArtifacts(runId);\n\t\tevidence.artifacts = artifacts;\n\t\tconst traces = await client.getTraces(runId);\n\t\tevidence.traces = traces;\n\n\t\tconst artifactsList = asList(artifacts);\n\t\tconst tracesList = asList(traces);\n\t\tconst artifactsEmpty = artifactsList.length === 0;\n\t\tconst tracesEmpty = tracesList.length === 0;\n\t\tconst isSuccess = branch === \"success\";\n\n\t\t// Part 2 section 3: absence of contrary evidence is never read as confirming evidence.\n\t\tif ((artifactsEmpty && !entry.allow_zero_artifacts) || tracesEmpty) {\n\t\t\tif (isSuccess && artifactsEmpty && tracesEmpty && entry.no_evidence_on_success_is_contradiction) {\n\t\t\t\treturn {\n\t\t\t\t\trun_id: runId,\n\t\t\t\t\tverdict: \"CONTRADICTED\",\n\t\t\t\t\treason_code: \"NO_EVIDENCE_ON_SUCCESS\",\n\t\t\t\t\treason: \"no-evidence-on-success-is-contradiction\",\n\t\t\t\t\tevidence,\n\t\t\t\t};\n\t\t\t}\n\t\t\tconst reasons: string[] = [];\n\t\t\tif (artifactsEmpty && !entry.allow_zero_artifacts) reasons.push(\"artifacts-empty-unexpected\");\n\t\t\tif (tracesEmpty) reasons.push(\"traces-empty-unexpected\");\n\t\t\treturn {\n\t\t\t\trun_id: runId,\n\t\t\t\tverdict: \"INDETERMINATE\",\n\t\t\t\treason_code: \"EVIDENCE_EMPTY\",\n\t\t\t\treason: reasons.join(\",\"),\n\t\t\t\tevidence,\n\t\t\t};\n\t\t}\n\n\t\tif (!isSuccess) {\n\t\t\t// CORROBORATED-FAILURE: reported failure, and the ambiguous-data fork above\n\t\t\t// already ruled out the case where evidence was too thin to say anything.\n\t\t\treturn {\n\t\t\t\trun_id: runId,\n\t\t\t\tverdict: \"CORROBORATED-FAILURE\",\n\t\t\t\treason_code: \"FAILURE_REPORTED\",\n\t\t\t\treason: \"failure-reported\",\n\t\t\t\tevidence,\n\t\t\t};\n\t\t}\n\n\t\tconst problems: { code: AdjudicationReasonCode; message: string }[] = [];\n\t\tif (artifactsList.some((artifact) => !hasSubstance(artifact))) {\n\t\t\tproblems.push({ code: \"EMPTY_ARTIFACT_CONTENT\", message: \"artifact-empty-or-whitespace\" });\n\t\t}\n\t\tif (entry.content_check) {\n\t\t\tfor (const artifact of artifactsList) {\n\t\t\t\tconst result: CheckResult = entry.content_check(artifact);\n\t\t\t\tif (!result.ok) {\n\t\t\t\t\tproblems.push({\n\t\t\t\t\t\tcode: result.code ?? \"CONTENT_CHECK_FAILED\",\n\t\t\t\t\t\tmessage: `content-check-failed${result.reason ? `: ${result.reason}` : \"\"}`,\n\t\t\t\t\t});\n\t\t\t\t}\n\t\t\t}\n\t\t}\n\t\tif (entry.trace_check) {\n\t\t\tconst result: CheckResult = entry.trace_check(traces);\n\t\t\tif (!result.ok) {\n\t\t\t\tproblems.push({\n\t\t\t\t\tcode: result.code ?? \"TRACE_CHECK_FAILED\",\n\t\t\t\t\tmessage: `trace-check-failed${result.reason ? `: ${result.reason}` : \"\"}`,\n\t\t\t\t});\n\t\t\t}\n\t\t}\n\t\tconst errorSpanCount = countErrorSpans(tracesList);\n\t\tif (errorSpanCount > 0) {\n\t\t\tproblems.push({ code: \"TRACE_ERROR_SPAN\", message: `error-span-scan: ${errorSpanCount} error span(s) found` });\n\t\t}\n\t\tif (typeof entry.expected_min_actions === \"number\" && entry.expected_min_actions > 0) {\n\t\t\tconst actionCount = countActionSpans(tracesList);\n\t\t\tif (actionCount < entry.expected_min_actions) {\n\t\t\t\tproblems.push({\n\t\t\t\t\tcode: \"MIN_ACTIONS_UNMET\",\n\t\t\t\t\tmessage: `expected-min-actions-not-met: found ${actionCount}, needed ${entry.expected_min_actions}`,\n\t\t\t\t});\n\t\t\t}\n\t\t}\n\n\t\tif (problems.length > 0) {\n\t\t\treturn {\n\t\t\t\trun_id: runId,\n\t\t\t\tverdict: \"CONTRADICTED\",\n\t\t\t\treason_code: reduceReasonCodes(problems.map((problem) => problem.code)),\n\t\t\t\treason: problems.map((problem) => problem.message).join(\"; \"),\n\t\t\t\tevidence,\n\t\t\t};\n\t\t}\n\t\treturn {\n\t\t\trun_id: runId,\n\t\t\tverdict: \"CONFIRMED\",\n\t\t\treason_code: \"ALL_CHECKS_PASSED\",\n\t\t\treason: \"all-checks-passed\",\n\t\t\tevidence,\n\t\t};\n\t} catch (error) {\n\t\treturn {\n\t\t\trun_id: runId,\n\t\t\tverdict: \"VERIFIER-ERROR\",\n\t\t\treason_code: \"RUN_FETCH_FAILED\",\n\t\t\treason: `run-adjudication-failed: ${describeError(error)}`,\n\t\t\tevidence,\n\t\t};\n\t}\n}\n\n/** Worst-wins record-level reduction over per-`run_id` verdicts (Part 2 section 2.5). */\nfunction reduceVerdicts(perRun: PerRunVerdict[]): VerdictState {\n\tconst priority: VerdictState[] = [\n\t\t\"VERIFIER-ERROR\",\n\t\t\"CONTRADICTED\",\n\t\t\"CORROBORATED-FAILURE\",\n\t\t\"INDETERMINATE\",\n\t\t\"CONFIRMED\",\n\t];\n\tfor (const state of priority) {\n\t\tif (perRun.some((r) => r.verdict === state)) return state;\n\t}\n\treturn \"CONFIRMED\";\n}\n\nfunction buildRecordReason(verdict: VerdictState, perRun: PerRunVerdict[]): string {\n\tif (perRun.length === 1) return perRun[0].reason;\n\tconst contributing = perRun.filter((r) => r.verdict === verdict).map((r) => `${r.run_id}: ${r.reason}`);\n\treturn `${verdict} via ${contributing.join(\"; \")}`;\n}\n\n/** Worst-wins record-level reason code, reduced over the runs sharing the record verdict. */\nfunction buildRecordReasonCode(verdict: VerdictState, perRun: PerRunVerdict[]): AdjudicationReasonCode {\n\treturn reduceReasonCodes(perRun.filter((r) => r.verdict === verdict).map((r) => r.reason_code));\n}\n\n/**\n * Adjudicates an `AdjudicationRequest` end to end (Part 2 sections 2.1-2.5, 3, 5).\n *\n * 1. Rejects malformed requests (missing `kind` or empty `run_ids`) as a record-level\n * VERIFIER-ERROR with no per-`run_id` results (section 2.1/2.4).\n * 2. Resolves the per-kind verifier once (section 4) and applies it identically to every\n * `run_id`, independently (section 2.2-2.4, 3).\n * 3. Reduces the per-`run_id` verdicts to one record-level verdict, worst-wins (section\n * 2.5), and never discards the per-`run_id` detail (section 5).\n * 4. Invokes `build_augmented_payload`, if defined, when the record-level verdict is not\n * CONFIRMED (section 4).\n */\nexport async function adjudicate(\n\trequest: AdjudicationRequest,\n\tclient: AdaptOrchClient,\n\tregistry: { get(kind: string): VerifierRegistryEntry },\n): Promise<AdjudicationResult> {\n\tif (!request.kind || request.run_ids.length === 0) {\n\t\treturn {\n\t\t\tverdict: \"VERIFIER-ERROR\",\n\t\t\treason_code: \"MALFORMED_REQUEST\",\n\t\t\treason: !request.kind ? \"malformed-request-missing-kind\" : \"malformed-request-empty-run-ids\",\n\t\t\tper_run: [],\n\t\t};\n\t}\n\n\tconst entry = registry.get(request.kind);\n\tconst outcomes: RunOutcome[] = [];\n\tfor (const runId of request.run_ids) {\n\t\toutcomes.push(await adjudicateRun(runId, entry, client));\n\t}\n\n\tconst perRun: PerRunVerdict[] = outcomes.map((outcome) => ({\n\t\trun_id: outcome.run_id,\n\t\tverdict: outcome.verdict,\n\t\treason_code: outcome.reason_code,\n\t\treason: outcome.reason,\n\t\tevidence_refs: outcome.evidence,\n\t}));\n\n\tconst verdict = reduceVerdicts(perRun);\n\tconst result: AdjudicationResult = {\n\t\tverdict,\n\t\treason_code: buildRecordReasonCode(verdict, perRun),\n\t\treason: buildRecordReason(verdict, perRun),\n\t\tper_run: perRun,\n\t};\n\n\tif (entry.build_augmented_payload && verdict !== \"CONFIRMED\") {\n\t\tconst artifactsByRun = outcomes.map((outcome) => ({\n\t\t\trun_id: outcome.run_id,\n\t\t\tartifacts: outcome.evidence.artifacts,\n\t\t}));\n\t\tconst tracesByRun = outcomes.map((outcome) => ({ run_id: outcome.run_id, traces: outcome.evidence.traces }));\n\t\tresult.augmented_payload = entry.build_augmented_payload(verdict, artifactsByRun, tracesByRun);\n\t}\n\n\treturn result;\n}\n"]}
|
|
@@ -0,0 +1,356 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Outcome Adjudicator (Part 2 sections 2, 3, 5) - a wrapper-side verification layer that
|
|
3
|
+
* turns AdaptOrch's own terminal-status report into a corroborated 5-state verdict, using
|
|
4
|
+
* only `getRun`/`getArtifacts`/`getTraces` introspection.
|
|
5
|
+
*
|
|
6
|
+
* Out of scope for this file (not covered by the input contract available here): the
|
|
7
|
+
* scope check and freshness check from Part 2 section 2.3, and the retry-count threshold
|
|
8
|
+
* from section 2.3, since none of `AdjudicationRequest`, `VerifierRegistryEntry`, or the
|
|
9
|
+
* client return shapes carry lane-scope, artifact-timestamp, or run-start-time data to
|
|
10
|
+
* check them against. What is implemented here: the non-empty check (honoring
|
|
11
|
+
* `allow_zero_artifacts`), the section 3 ambiguous-data fork, `content_check`,
|
|
12
|
+
* `trace_check`, an error-span heuristic scan, and the `expected_min_actions` count.
|
|
13
|
+
*/
|
|
14
|
+
import { reduceReasonCodes } from "./adjudicator-registry.js";
|
|
15
|
+
const NON_TERMINAL_STATUSES = new Set([
|
|
16
|
+
"pending",
|
|
17
|
+
"queued",
|
|
18
|
+
"running",
|
|
19
|
+
"in_progress",
|
|
20
|
+
"in-progress",
|
|
21
|
+
"scheduled",
|
|
22
|
+
"dispatched",
|
|
23
|
+
"starting",
|
|
24
|
+
"initializing",
|
|
25
|
+
]);
|
|
26
|
+
const SUCCESS_STATUSES = new Set(["completed", "success", "succeeded", "done", "ok", "finished"]);
|
|
27
|
+
const FAILURE_STATUSES = new Set([
|
|
28
|
+
"failed",
|
|
29
|
+
"failure",
|
|
30
|
+
"error",
|
|
31
|
+
"errored",
|
|
32
|
+
"cancelled",
|
|
33
|
+
"canceled",
|
|
34
|
+
"timeout",
|
|
35
|
+
"timed_out",
|
|
36
|
+
"aborted",
|
|
37
|
+
]);
|
|
38
|
+
function isRecord(value) {
|
|
39
|
+
return typeof value === "object" && value !== null;
|
|
40
|
+
}
|
|
41
|
+
function readStringField(value, keys) {
|
|
42
|
+
if (!isRecord(value))
|
|
43
|
+
return undefined;
|
|
44
|
+
for (const key of keys) {
|
|
45
|
+
const field = value[key];
|
|
46
|
+
if (typeof field === "string")
|
|
47
|
+
return field;
|
|
48
|
+
}
|
|
49
|
+
return undefined;
|
|
50
|
+
}
|
|
51
|
+
/**
|
|
52
|
+
* Interprets a `getRun` payload into the branch the OA cares about (Part 2 section 2.2).
|
|
53
|
+
* A payload with no recognizable string status field is treated as unparseable - a
|
|
54
|
+
* checker failure, not a target-task failure (section 2.4, VERIFIER-ERROR).
|
|
55
|
+
*/
|
|
56
|
+
function interpretRunStatus(run) {
|
|
57
|
+
const status = readStringField(run, ["status", "state"]);
|
|
58
|
+
if (status === undefined)
|
|
59
|
+
return "unparseable";
|
|
60
|
+
const normalized = status.toLowerCase();
|
|
61
|
+
if (NON_TERMINAL_STATUSES.has(normalized))
|
|
62
|
+
return "non-terminal";
|
|
63
|
+
if (SUCCESS_STATUSES.has(normalized))
|
|
64
|
+
return "success";
|
|
65
|
+
if (FAILURE_STATUSES.has(normalized))
|
|
66
|
+
return "failure";
|
|
67
|
+
return "unparseable";
|
|
68
|
+
}
|
|
69
|
+
/**
|
|
70
|
+
* Coerces an unknown `getArtifacts`/`getTraces` payload into a flat list without assuming
|
|
71
|
+
* a specific shape (the concrete return type belongs to the concurrently-authored
|
|
72
|
+
* `adaptorch-client.ts`). Falls back to common pagination-style container keys, then to
|
|
73
|
+
* treating a single non-null value as a one-item list.
|
|
74
|
+
*/
|
|
75
|
+
function asList(value) {
|
|
76
|
+
if (value === null || value === undefined)
|
|
77
|
+
return [];
|
|
78
|
+
if (Array.isArray(value))
|
|
79
|
+
return value;
|
|
80
|
+
if (isRecord(value)) {
|
|
81
|
+
for (const key of ["items", "artifacts", "traces", "spans", "entries", "data", "results"]) {
|
|
82
|
+
const field = value[key];
|
|
83
|
+
if (Array.isArray(field))
|
|
84
|
+
return field;
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
return [value];
|
|
88
|
+
}
|
|
89
|
+
/** True unless the item is a recognizably empty/whitespace-only value (Part 2 section 2.3). */
|
|
90
|
+
function hasSubstance(item) {
|
|
91
|
+
if (typeof item === "string")
|
|
92
|
+
return item.trim().length > 0;
|
|
93
|
+
if (isRecord(item)) {
|
|
94
|
+
const size = item.size ?? item.byteLength ?? item.length;
|
|
95
|
+
if (typeof size === "number")
|
|
96
|
+
return size > 0;
|
|
97
|
+
const text = readStringField(item, ["content", "text", "body"]);
|
|
98
|
+
if (text !== undefined)
|
|
99
|
+
return text.trim().length > 0;
|
|
100
|
+
}
|
|
101
|
+
return true;
|
|
102
|
+
}
|
|
103
|
+
/** Heuristic ERROR-severity span scan (Part 2 section 2.3), tolerant of unknown span shapes. */
|
|
104
|
+
function countErrorSpans(traces) {
|
|
105
|
+
let count = 0;
|
|
106
|
+
for (const span of traces) {
|
|
107
|
+
if (!isRecord(span))
|
|
108
|
+
continue;
|
|
109
|
+
const level = readStringField(span, ["level", "severity", "status"]);
|
|
110
|
+
if (level !== undefined && level.toLowerCase() === "error") {
|
|
111
|
+
count += 1;
|
|
112
|
+
continue;
|
|
113
|
+
}
|
|
114
|
+
if (span.error === true || span.isError === true)
|
|
115
|
+
count += 1;
|
|
116
|
+
}
|
|
117
|
+
return count;
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Heuristic count of "action" spans for the `expected_min_actions` check (Part 2 sections
|
|
121
|
+
* 2.3/4). Recognizes a handful of common marker fields; if none of the spans carry any of
|
|
122
|
+
* them, falls back to the total span count so the check degrades to a coarse presence
|
|
123
|
+
* signal instead of always failing.
|
|
124
|
+
*/
|
|
125
|
+
function countActionSpans(traces) {
|
|
126
|
+
let recognized = 0;
|
|
127
|
+
let matched = 0;
|
|
128
|
+
for (const span of traces) {
|
|
129
|
+
if (!isRecord(span))
|
|
130
|
+
continue;
|
|
131
|
+
const marker = readStringField(span, ["action", "tool_call", "toolCall", "kind", "type"]);
|
|
132
|
+
if (marker !== undefined) {
|
|
133
|
+
recognized += 1;
|
|
134
|
+
if (/action|tool|write|edit|call/i.test(marker))
|
|
135
|
+
matched += 1;
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
return recognized > 0 ? matched : traces.length;
|
|
139
|
+
}
|
|
140
|
+
function describeError(error) {
|
|
141
|
+
return error instanceof Error ? error.message : String(error);
|
|
142
|
+
}
|
|
143
|
+
/**
|
|
144
|
+
* Adjudicates a single `run_id` against the resolved registry entry (Part 2 sections
|
|
145
|
+
* 2.2-2.4 and 3). Robust to a misbehaving fetch: any thrown error from the client during
|
|
146
|
+
* this run's own fetch/check sequence becomes that run_id's own VERIFIER-ERROR rather
|
|
147
|
+
* than propagating out of {@link adjudicate}.
|
|
148
|
+
*/
|
|
149
|
+
async function adjudicateRun(runId, entry, client) {
|
|
150
|
+
const evidence = {};
|
|
151
|
+
try {
|
|
152
|
+
const run = await client.getRun(runId);
|
|
153
|
+
evidence.run = run;
|
|
154
|
+
const branch = interpretRunStatus(run);
|
|
155
|
+
if (branch === "unparseable") {
|
|
156
|
+
return {
|
|
157
|
+
run_id: runId,
|
|
158
|
+
verdict: "VERIFIER-ERROR",
|
|
159
|
+
reason_code: "RUN_STATUS_UNPARSEABLE",
|
|
160
|
+
reason: "run-status-unparseable",
|
|
161
|
+
evidence,
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
if (branch === "non-terminal") {
|
|
165
|
+
return {
|
|
166
|
+
run_id: runId,
|
|
167
|
+
verdict: "INDETERMINATE",
|
|
168
|
+
reason_code: "RUN_NOT_TERMINAL",
|
|
169
|
+
reason: "run-not-terminal",
|
|
170
|
+
evidence,
|
|
171
|
+
};
|
|
172
|
+
}
|
|
173
|
+
const artifacts = await client.getArtifacts(runId);
|
|
174
|
+
evidence.artifacts = artifacts;
|
|
175
|
+
const traces = await client.getTraces(runId);
|
|
176
|
+
evidence.traces = traces;
|
|
177
|
+
const artifactsList = asList(artifacts);
|
|
178
|
+
const tracesList = asList(traces);
|
|
179
|
+
const artifactsEmpty = artifactsList.length === 0;
|
|
180
|
+
const tracesEmpty = tracesList.length === 0;
|
|
181
|
+
const isSuccess = branch === "success";
|
|
182
|
+
// Part 2 section 3: absence of contrary evidence is never read as confirming evidence.
|
|
183
|
+
if ((artifactsEmpty && !entry.allow_zero_artifacts) || tracesEmpty) {
|
|
184
|
+
if (isSuccess && artifactsEmpty && tracesEmpty && entry.no_evidence_on_success_is_contradiction) {
|
|
185
|
+
return {
|
|
186
|
+
run_id: runId,
|
|
187
|
+
verdict: "CONTRADICTED",
|
|
188
|
+
reason_code: "NO_EVIDENCE_ON_SUCCESS",
|
|
189
|
+
reason: "no-evidence-on-success-is-contradiction",
|
|
190
|
+
evidence,
|
|
191
|
+
};
|
|
192
|
+
}
|
|
193
|
+
const reasons = [];
|
|
194
|
+
if (artifactsEmpty && !entry.allow_zero_artifacts)
|
|
195
|
+
reasons.push("artifacts-empty-unexpected");
|
|
196
|
+
if (tracesEmpty)
|
|
197
|
+
reasons.push("traces-empty-unexpected");
|
|
198
|
+
return {
|
|
199
|
+
run_id: runId,
|
|
200
|
+
verdict: "INDETERMINATE",
|
|
201
|
+
reason_code: "EVIDENCE_EMPTY",
|
|
202
|
+
reason: reasons.join(","),
|
|
203
|
+
evidence,
|
|
204
|
+
};
|
|
205
|
+
}
|
|
206
|
+
if (!isSuccess) {
|
|
207
|
+
// CORROBORATED-FAILURE: reported failure, and the ambiguous-data fork above
|
|
208
|
+
// already ruled out the case where evidence was too thin to say anything.
|
|
209
|
+
return {
|
|
210
|
+
run_id: runId,
|
|
211
|
+
verdict: "CORROBORATED-FAILURE",
|
|
212
|
+
reason_code: "FAILURE_REPORTED",
|
|
213
|
+
reason: "failure-reported",
|
|
214
|
+
evidence,
|
|
215
|
+
};
|
|
216
|
+
}
|
|
217
|
+
const problems = [];
|
|
218
|
+
if (artifactsList.some((artifact) => !hasSubstance(artifact))) {
|
|
219
|
+
problems.push({ code: "EMPTY_ARTIFACT_CONTENT", message: "artifact-empty-or-whitespace" });
|
|
220
|
+
}
|
|
221
|
+
if (entry.content_check) {
|
|
222
|
+
for (const artifact of artifactsList) {
|
|
223
|
+
const result = entry.content_check(artifact);
|
|
224
|
+
if (!result.ok) {
|
|
225
|
+
problems.push({
|
|
226
|
+
code: result.code ?? "CONTENT_CHECK_FAILED",
|
|
227
|
+
message: `content-check-failed${result.reason ? `: ${result.reason}` : ""}`,
|
|
228
|
+
});
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
if (entry.trace_check) {
|
|
233
|
+
const result = entry.trace_check(traces);
|
|
234
|
+
if (!result.ok) {
|
|
235
|
+
problems.push({
|
|
236
|
+
code: result.code ?? "TRACE_CHECK_FAILED",
|
|
237
|
+
message: `trace-check-failed${result.reason ? `: ${result.reason}` : ""}`,
|
|
238
|
+
});
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
const errorSpanCount = countErrorSpans(tracesList);
|
|
242
|
+
if (errorSpanCount > 0) {
|
|
243
|
+
problems.push({ code: "TRACE_ERROR_SPAN", message: `error-span-scan: ${errorSpanCount} error span(s) found` });
|
|
244
|
+
}
|
|
245
|
+
if (typeof entry.expected_min_actions === "number" && entry.expected_min_actions > 0) {
|
|
246
|
+
const actionCount = countActionSpans(tracesList);
|
|
247
|
+
if (actionCount < entry.expected_min_actions) {
|
|
248
|
+
problems.push({
|
|
249
|
+
code: "MIN_ACTIONS_UNMET",
|
|
250
|
+
message: `expected-min-actions-not-met: found ${actionCount}, needed ${entry.expected_min_actions}`,
|
|
251
|
+
});
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
if (problems.length > 0) {
|
|
255
|
+
return {
|
|
256
|
+
run_id: runId,
|
|
257
|
+
verdict: "CONTRADICTED",
|
|
258
|
+
reason_code: reduceReasonCodes(problems.map((problem) => problem.code)),
|
|
259
|
+
reason: problems.map((problem) => problem.message).join("; "),
|
|
260
|
+
evidence,
|
|
261
|
+
};
|
|
262
|
+
}
|
|
263
|
+
return {
|
|
264
|
+
run_id: runId,
|
|
265
|
+
verdict: "CONFIRMED",
|
|
266
|
+
reason_code: "ALL_CHECKS_PASSED",
|
|
267
|
+
reason: "all-checks-passed",
|
|
268
|
+
evidence,
|
|
269
|
+
};
|
|
270
|
+
}
|
|
271
|
+
catch (error) {
|
|
272
|
+
return {
|
|
273
|
+
run_id: runId,
|
|
274
|
+
verdict: "VERIFIER-ERROR",
|
|
275
|
+
reason_code: "RUN_FETCH_FAILED",
|
|
276
|
+
reason: `run-adjudication-failed: ${describeError(error)}`,
|
|
277
|
+
evidence,
|
|
278
|
+
};
|
|
279
|
+
}
|
|
280
|
+
}
|
|
281
|
+
/** Worst-wins record-level reduction over per-`run_id` verdicts (Part 2 section 2.5). */
|
|
282
|
+
function reduceVerdicts(perRun) {
|
|
283
|
+
const priority = [
|
|
284
|
+
"VERIFIER-ERROR",
|
|
285
|
+
"CONTRADICTED",
|
|
286
|
+
"CORROBORATED-FAILURE",
|
|
287
|
+
"INDETERMINATE",
|
|
288
|
+
"CONFIRMED",
|
|
289
|
+
];
|
|
290
|
+
for (const state of priority) {
|
|
291
|
+
if (perRun.some((r) => r.verdict === state))
|
|
292
|
+
return state;
|
|
293
|
+
}
|
|
294
|
+
return "CONFIRMED";
|
|
295
|
+
}
|
|
296
|
+
function buildRecordReason(verdict, perRun) {
|
|
297
|
+
if (perRun.length === 1)
|
|
298
|
+
return perRun[0].reason;
|
|
299
|
+
const contributing = perRun.filter((r) => r.verdict === verdict).map((r) => `${r.run_id}: ${r.reason}`);
|
|
300
|
+
return `${verdict} via ${contributing.join("; ")}`;
|
|
301
|
+
}
|
|
302
|
+
/** Worst-wins record-level reason code, reduced over the runs sharing the record verdict. */
|
|
303
|
+
function buildRecordReasonCode(verdict, perRun) {
|
|
304
|
+
return reduceReasonCodes(perRun.filter((r) => r.verdict === verdict).map((r) => r.reason_code));
|
|
305
|
+
}
|
|
306
|
+
/**
|
|
307
|
+
* Adjudicates an `AdjudicationRequest` end to end (Part 2 sections 2.1-2.5, 3, 5).
|
|
308
|
+
*
|
|
309
|
+
* 1. Rejects malformed requests (missing `kind` or empty `run_ids`) as a record-level
|
|
310
|
+
* VERIFIER-ERROR with no per-`run_id` results (section 2.1/2.4).
|
|
311
|
+
* 2. Resolves the per-kind verifier once (section 4) and applies it identically to every
|
|
312
|
+
* `run_id`, independently (section 2.2-2.4, 3).
|
|
313
|
+
* 3. Reduces the per-`run_id` verdicts to one record-level verdict, worst-wins (section
|
|
314
|
+
* 2.5), and never discards the per-`run_id` detail (section 5).
|
|
315
|
+
* 4. Invokes `build_augmented_payload`, if defined, when the record-level verdict is not
|
|
316
|
+
* CONFIRMED (section 4).
|
|
317
|
+
*/
|
|
318
|
+
export async function adjudicate(request, client, registry) {
|
|
319
|
+
if (!request.kind || request.run_ids.length === 0) {
|
|
320
|
+
return {
|
|
321
|
+
verdict: "VERIFIER-ERROR",
|
|
322
|
+
reason_code: "MALFORMED_REQUEST",
|
|
323
|
+
reason: !request.kind ? "malformed-request-missing-kind" : "malformed-request-empty-run-ids",
|
|
324
|
+
per_run: [],
|
|
325
|
+
};
|
|
326
|
+
}
|
|
327
|
+
const entry = registry.get(request.kind);
|
|
328
|
+
const outcomes = [];
|
|
329
|
+
for (const runId of request.run_ids) {
|
|
330
|
+
outcomes.push(await adjudicateRun(runId, entry, client));
|
|
331
|
+
}
|
|
332
|
+
const perRun = outcomes.map((outcome) => ({
|
|
333
|
+
run_id: outcome.run_id,
|
|
334
|
+
verdict: outcome.verdict,
|
|
335
|
+
reason_code: outcome.reason_code,
|
|
336
|
+
reason: outcome.reason,
|
|
337
|
+
evidence_refs: outcome.evidence,
|
|
338
|
+
}));
|
|
339
|
+
const verdict = reduceVerdicts(perRun);
|
|
340
|
+
const result = {
|
|
341
|
+
verdict,
|
|
342
|
+
reason_code: buildRecordReasonCode(verdict, perRun),
|
|
343
|
+
reason: buildRecordReason(verdict, perRun),
|
|
344
|
+
per_run: perRun,
|
|
345
|
+
};
|
|
346
|
+
if (entry.build_augmented_payload && verdict !== "CONFIRMED") {
|
|
347
|
+
const artifactsByRun = outcomes.map((outcome) => ({
|
|
348
|
+
run_id: outcome.run_id,
|
|
349
|
+
artifacts: outcome.evidence.artifacts,
|
|
350
|
+
}));
|
|
351
|
+
const tracesByRun = outcomes.map((outcome) => ({ run_id: outcome.run_id, traces: outcome.evidence.traces }));
|
|
352
|
+
result.augmented_payload = entry.build_augmented_payload(verdict, artifactsByRun, tracesByRun);
|
|
353
|
+
}
|
|
354
|
+
return result;
|
|
355
|
+
}
|
|
356
|
+
//# sourceMappingURL=adjudicator.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"adjudicator.js","sourceRoot":"","sources":["../src/adjudicator.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AASH,OAAO,EAAE,iBAAiB,EAAE,MAAM,2BAA2B,CAAC;AAqE9D,MAAM,qBAAqB,GAAG,IAAI,GAAG,CAAC;IACrC,SAAS;IACT,QAAQ;IACR,SAAS;IACT,aAAa;IACb,aAAa;IACb,WAAW;IACX,YAAY;IACZ,UAAU;IACV,cAAc;CACd,CAAC,CAAC;AACH,MAAM,gBAAgB,GAAG,IAAI,GAAG,CAAC,CAAC,WAAW,EAAE,SAAS,EAAE,WAAW,EAAE,MAAM,EAAE,IAAI,EAAE,UAAU,CAAC,CAAC,CAAC;AAClG,MAAM,gBAAgB,GAAG,IAAI,GAAG,CAAC;IAChC,QAAQ;IACR,SAAS;IACT,OAAO;IACP,SAAS;IACT,WAAW;IACX,UAAU;IACV,SAAS;IACT,WAAW;IACX,SAAS;CACT,CAAC,CAAC;AAIH,SAAS,QAAQ,CAAC,KAAc,EAAoC;IACnE,OAAO,OAAO,KAAK,KAAK,QAAQ,IAAI,KAAK,KAAK,IAAI,CAAC;AAAA,CACnD;AAED,SAAS,eAAe,CAAC,KAAc,EAAE,IAAc,EAAsB;IAC5E,IAAI,CAAC,QAAQ,CAAC,KAAK,CAAC;QAAE,OAAO,SAAS,CAAC;IACvC,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;QACxB,MAAM,KAAK,GAAG,KAAK,CAAC,GAAG,CAAC,CAAC;QACzB,IAAI,OAAO,KAAK,KAAK,QAAQ;YAAE,OAAO,KAAK,CAAC;IAC7C,CAAC;IACD,OAAO,SAAS,CAAC;AAAA,CACjB;AAED;;;;GAIG;AACH,SAAS,kBAAkB,CAAC,GAAY,EAAmB;IAC1D,MAAM,MAAM,GAAG,eAAe,CAAC,GAAG,EAAE,CAAC,QAAQ,EAAE,OAAO,CAAC,CAAC,CAAC;IACzD,IAAI,MAAM,KAAK,SAAS;QAAE,OAAO,aAAa,CAAC;IAC/C,MAAM,UAAU,GAAG,MAAM,CAAC,WAAW,EAAE,CAAC;IACxC,IAAI,qBAAqB,CAAC,GAAG,CAAC,UAAU,CAAC;QAAE,OAAO,cAAc,CAAC;IACjE,IAAI,gBAAgB,CAAC,GAAG,CAAC,UAAU,CAAC;QAAE,OAAO,SAAS,CAAC;IACvD,IAAI,gBAAgB,CAAC,GAAG,CAAC,UAAU,CAAC;QAAE,OAAO,SAAS,CAAC;IACvD,OAAO,aAAa,CAAC;AAAA,CACrB;AAED;;;;;GAKG;AACH,SAAS,MAAM,CAAC,KAAc,EAAa;IAC1C,IAAI,KAAK,KAAK,IAAI,IAAI,KAAK,KAAK,SAAS;QAAE,OAAO,EAAE,CAAC;IACrD,IAAI,KAAK,CAAC,OAAO,CAAC,KAAK,CAAC;QAAE,OAAO,KAAK,CAAC;IACvC,IAAI,QAAQ,CAAC,KAAK,CAAC,EAAE,CAAC;QACrB,KAAK,MAAM,GAAG,IAAI,CAAC,OAAO,EAAE,WAAW,EAAE,QAAQ,EAAE,OAAO,EAAE,SAAS,EAAE,MAAM,EAAE,SAAS,CAAC,EAAE,CAAC;YAC3F,MAAM,KAAK,GAAG,KAAK,CAAC,GAAG,CAAC,CAAC;YACzB,IAAI,KAAK,CAAC,OAAO,CAAC,KAAK,CAAC;gBAAE,OAAO,KAAK,CAAC;QACxC,CAAC;IACF,CAAC;IACD,OAAO,CAAC,KAAK,CAAC,CAAC;AAAA,CACf;AAED,+FAA+F;AAC/F,SAAS,YAAY,CAAC,IAAa,EAAW;IAC7C,IAAI,OAAO,IAAI,KAAK,QAAQ;QAAE,OAAO,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,CAAC;IAC5D,IAAI,QAAQ,CAAC,IAAI,CAAC,EAAE,CAAC;QACpB,MAAM,IAAI,GAAG,IAAI,CAAC,IAAI,IAAI,IAAI,CAAC,UAAU,IAAI,IAAI,CAAC,MAAM,CAAC;QACzD,IAAI,OAAO,IAAI,KAAK,QAAQ;YAAE,OAAO,IAAI,GAAG,CAAC,CAAC;QAC9C,MAAM,IAAI,GAAG,eAAe,CAAC,IAAI,EAAE,CAAC,SAAS,EAAE,MAAM,EAAE,MAAM,CAAC,CAAC,CAAC;QAChE,IAAI,IAAI,KAAK,SAAS;YAAE,OAAO,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,CAAC;IACvD,CAAC;IACD,OAAO,IAAI,CAAC;AAAA,CACZ;AAED,gGAAgG;AAChG,SAAS,eAAe,CAAC,MAAiB,EAAU;IACnD,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,KAAK,MAAM,IAAI,IAAI,MAAM,EAAE,CAAC;QAC3B,IAAI,CAAC,QAAQ,CAAC,IAAI,CAAC;YAAE,SAAS;QAC9B,MAAM,KAAK,GAAG,eAAe,CAAC,IAAI,EAAE,CAAC,OAAO,EAAE,UAAU,EAAE,QAAQ,CAAC,CAAC,CAAC;QACrE,IAAI,KAAK,KAAK,SAAS,IAAI,KAAK,CAAC,WAAW,EAAE,KAAK,OAAO,EAAE,CAAC;YAC5D,KAAK,IAAI,CAAC,CAAC;YACX,SAAS;QACV,CAAC;QACD,IAAI,IAAI,CAAC,KAAK,KAAK,IAAI,IAAI,IAAI,CAAC,OAAO,KAAK,IAAI;YAAE,KAAK,IAAI,CAAC,CAAC;IAC9D,CAAC;IACD,OAAO,KAAK,CAAC;AAAA,CACb;AAED;;;;;GAKG;AACH,SAAS,gBAAgB,CAAC,MAAiB,EAAU;IACpD,IAAI,UAAU,GAAG,CAAC,CAAC;IACnB,IAAI,OAAO,GAAG,CAAC,CAAC;IAChB,KAAK,MAAM,IAAI,IAAI,MAAM,EAAE,CAAC;QAC3B,IAAI,CAAC,QAAQ,CAAC,IAAI,CAAC;YAAE,SAAS;QAC9B,MAAM,MAAM,GAAG,eAAe,CAAC,IAAI,EAAE,CAAC,QAAQ,EAAE,WAAW,EAAE,UAAU,EAAE,MAAM,EAAE,MAAM,CAAC,CAAC,CAAC;QAC1F,IAAI,MAAM,KAAK,SAAS,EAAE,CAAC;YAC1B,UAAU,IAAI,CAAC,CAAC;YAChB,IAAI,8BAA8B,CAAC,IAAI,CAAC,MAAM,CAAC;gBAAE,OAAO,IAAI,CAAC,CAAC;QAC/D,CAAC;IACF,CAAC;IACD,OAAO,UAAU,GAAG,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,MAAM,CAAC;AAAA,CAChD;AAED,SAAS,aAAa,CAAC,KAAc,EAAU;IAC9C,OAAO,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC;AAAA,CAC9D;AAED;;;;;GAKG;AACH,KAAK,UAAU,aAAa,CAC3B,KAAa,EACb,KAA4B,EAC5B,MAAuB,EACD;IACtB,MAAM,QAAQ,GAAgB,EAAE,CAAC;IACjC,IAAI,CAAC;QACJ,MAAM,GAAG,GAAG,MAAM,MAAM,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC;QACvC,QAAQ,CAAC,GAAG,GAAG,GAAG,CAAC;QAEnB,MAAM,MAAM,GAAG,kBAAkB,CAAC,GAAG,CAAC,CAAC;QACvC,IAAI,MAAM,KAAK,aAAa,EAAE,CAAC;YAC9B,OAAO;gBACN,MAAM,EAAE,KAAK;gBACb,OAAO,EAAE,gBAAgB;gBACzB,WAAW,EAAE,wBAAwB;gBACrC,MAAM,EAAE,wBAAwB;gBAChC,QAAQ;aACR,CAAC;QACH,CAAC;QACD,IAAI,MAAM,KAAK,cAAc,EAAE,CAAC;YAC/B,OAAO;gBACN,MAAM,EAAE,KAAK;gBACb,OAAO,EAAE,eAAe;gBACxB,WAAW,EAAE,kBAAkB;gBAC/B,MAAM,EAAE,kBAAkB;gBAC1B,QAAQ;aACR,CAAC;QACH,CAAC;QAED,MAAM,SAAS,GAAG,MAAM,MAAM,CAAC,YAAY,CAAC,KAAK,CAAC,CAAC;QACnD,QAAQ,CAAC,SAAS,GAAG,SAAS,CAAC;QAC/B,MAAM,MAAM,GAAG,MAAM,MAAM,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC;QAC7C,QAAQ,CAAC,MAAM,GAAG,MAAM,CAAC;QAEzB,MAAM,aAAa,GAAG,MAAM,CAAC,SAAS,CAAC,CAAC;QACxC,MAAM,UAAU,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC;QAClC,MAAM,cAAc,GAAG,aAAa,CAAC,MAAM,KAAK,CAAC,CAAC;QAClD,MAAM,WAAW,GAAG,UAAU,CAAC,MAAM,KAAK,CAAC,CAAC;QAC5C,MAAM,SAAS,GAAG,MAAM,KAAK,SAAS,CAAC;QAEvC,uFAAuF;QACvF,IAAI,CAAC,cAAc,IAAI,CAAC,KAAK,CAAC,oBAAoB,CAAC,IAAI,WAAW,EAAE,CAAC;YACpE,IAAI,SAAS,IAAI,cAAc,IAAI,WAAW,IAAI,KAAK,CAAC,uCAAuC,EAAE,CAAC;gBACjG,OAAO;oBACN,MAAM,EAAE,KAAK;oBACb,OAAO,EAAE,cAAc;oBACvB,WAAW,EAAE,wBAAwB;oBACrC,MAAM,EAAE,yCAAyC;oBACjD,QAAQ;iBACR,CAAC;YACH,CAAC;YACD,MAAM,OAAO,GAAa,EAAE,CAAC;YAC7B,IAAI,cAAc,IAAI,CAAC,KAAK,CAAC,oBAAoB;gBAAE,OAAO,CAAC,IAAI,CAAC,4BAA4B,CAAC,CAAC;YAC9F,IAAI,WAAW;gBAAE,OAAO,CAAC,IAAI,CAAC,yBAAyB,CAAC,CAAC;YACzD,OAAO;gBACN,MAAM,EAAE,KAAK;gBACb,OAAO,EAAE,eAAe;gBACxB,WAAW,EAAE,gBAAgB;gBAC7B,MAAM,EAAE,OAAO,CAAC,IAAI,CAAC,GAAG,CAAC;gBACzB,QAAQ;aACR,CAAC;QACH,CAAC;QAED,IAAI,CAAC,SAAS,EAAE,CAAC;YAChB,4EAA4E;YAC5E,0EAA0E;YAC1E,OAAO;gBACN,MAAM,EAAE,KAAK;gBACb,OAAO,EAAE,sBAAsB;gBAC/B,WAAW,EAAE,kBAAkB;gBAC/B,MAAM,EAAE,kBAAkB;gBAC1B,QAAQ;aACR,CAAC;QACH,CAAC;QAED,MAAM,QAAQ,GAAwD,EAAE,CAAC;QACzE,IAAI,aAAa,CAAC,IAAI,CAAC,CAAC,QAAQ,EAAE,EAAE,CAAC,CAAC,YAAY,CAAC,QAAQ,CAAC,CAAC,EAAE,CAAC;YAC/D,QAAQ,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,wBAAwB,EAAE,OAAO,EAAE,8BAA8B,EAAE,CAAC,CAAC;QAC5F,CAAC;QACD,IAAI,KAAK,CAAC,aAAa,EAAE,CAAC;YACzB,KAAK,MAAM,QAAQ,IAAI,aAAa,EAAE,CAAC;gBACtC,MAAM,MAAM,GAAgB,KAAK,CAAC,aAAa,CAAC,QAAQ,CAAC,CAAC;gBAC1D,IAAI,CAAC,MAAM,CAAC,EAAE,EAAE,CAAC;oBAChB,QAAQ,CAAC,IAAI,CAAC;wBACb,IAAI,EAAE,MAAM,CAAC,IAAI,IAAI,sBAAsB;wBAC3C,OAAO,EAAE,uBAAuB,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC,KAAK,MAAM,CAAC,MAAM,EAAE,CAAC,CAAC,CAAC,EAAE,EAAE;qBAC3E,CAAC,CAAC;gBACJ,CAAC;YACF,CAAC;QACF,CAAC;QACD,IAAI,KAAK,CAAC,WAAW,EAAE,CAAC;YACvB,MAAM,MAAM,GAAgB,KAAK,CAAC,WAAW,CAAC,MAAM,CAAC,CAAC;YACtD,IAAI,CAAC,MAAM,CAAC,EAAE,EAAE,CAAC;gBAChB,QAAQ,CAAC,IAAI,CAAC;oBACb,IAAI,EAAE,MAAM,CAAC,IAAI,IAAI,oBAAoB;oBACzC,OAAO,EAAE,qBAAqB,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC,KAAK,MAAM,CAAC,MAAM,EAAE,CAAC,CAAC,CAAC,EAAE,EAAE;iBACzE,CAAC,CAAC;YACJ,CAAC;QACF,CAAC;QACD,MAAM,cAAc,GAAG,eAAe,CAAC,UAAU,CAAC,CAAC;QACnD,IAAI,cAAc,GAAG,CAAC,EAAE,CAAC;YACxB,QAAQ,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,kBAAkB,EAAE,OAAO,EAAE,oBAAoB,cAAc,sBAAsB,EAAE,CAAC,CAAC;QAChH,CAAC;QACD,IAAI,OAAO,KAAK,CAAC,oBAAoB,KAAK,QAAQ,IAAI,KAAK,CAAC,oBAAoB,GAAG,CAAC,EAAE,CAAC;YACtF,MAAM,WAAW,GAAG,gBAAgB,CAAC,UAAU,CAAC,CAAC;YACjD,IAAI,WAAW,GAAG,KAAK,CAAC,oBAAoB,EAAE,CAAC;gBAC9C,QAAQ,CAAC,IAAI,CAAC;oBACb,IAAI,EAAE,mBAAmB;oBACzB,OAAO,EAAE,uCAAuC,WAAW,YAAY,KAAK,CAAC,oBAAoB,EAAE;iBACnG,CAAC,CAAC;YACJ,CAAC;QACF,CAAC;QAED,IAAI,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YACzB,OAAO;gBACN,MAAM,EAAE,KAAK;gBACb,OAAO,EAAE,cAAc;gBACvB,WAAW,EAAE,iBAAiB,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC,OAAO,EAAE,EAAE,CAAC,OAAO,CAAC,IAAI,CAAC,CAAC;gBACvE,MAAM,EAAE,QAAQ,CAAC,GAAG,CAAC,CAAC,OAAO,EAAE,EAAE,CAAC,OAAO,CAAC,OAAO,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC;gBAC7D,QAAQ;aACR,CAAC;QACH,CAAC;QACD,OAAO;YACN,MAAM,EAAE,KAAK;YACb,OAAO,EAAE,WAAW;YACpB,WAAW,EAAE,mBAAmB;YAChC,MAAM,EAAE,mBAAmB;YAC3B,QAAQ;SACR,CAAC;IACH,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QAChB,OAAO;YACN,MAAM,EAAE,KAAK;YACb,OAAO,EAAE,gBAAgB;YACzB,WAAW,EAAE,kBAAkB;YAC/B,MAAM,EAAE,4BAA4B,aAAa,CAAC,KAAK,CAAC,EAAE;YAC1D,QAAQ;SACR,CAAC;IACH,CAAC;AAAA,CACD;AAED,yFAAyF;AACzF,SAAS,cAAc,CAAC,MAAuB,EAAgB;IAC9D,MAAM,QAAQ,GAAmB;QAChC,gBAAgB;QAChB,cAAc;QACd,sBAAsB;QACtB,eAAe;QACf,WAAW;KACX,CAAC;IACF,KAAK,MAAM,KAAK,IAAI,QAAQ,EAAE,CAAC;QAC9B,IAAI,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,OAAO,KAAK,KAAK,CAAC;YAAE,OAAO,KAAK,CAAC;IAC3D,CAAC;IACD,OAAO,WAAW,CAAC;AAAA,CACnB;AAED,SAAS,iBAAiB,CAAC,OAAqB,EAAE,MAAuB,EAAU;IAClF,IAAI,MAAM,CAAC,MAAM,KAAK,CAAC;QAAE,OAAO,MAAM,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC;IACjD,MAAM,YAAY,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,OAAO,KAAK,OAAO,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,GAAG,CAAC,CAAC,MAAM,KAAK,CAAC,CAAC,MAAM,EAAE,CAAC,CAAC;IACxG,OAAO,GAAG,OAAO,QAAQ,YAAY,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;AAAA,CACnD;AAED,6FAA6F;AAC7F,SAAS,qBAAqB,CAAC,OAAqB,EAAE,MAAuB,EAA0B;IACtG,OAAO,iBAAiB,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,OAAO,KAAK,OAAO,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,WAAW,CAAC,CAAC,CAAC;AAAA,CAChG;AAED;;;;;;;;;;;GAWG;AACH,MAAM,CAAC,KAAK,UAAU,UAAU,CAC/B,OAA4B,EAC5B,MAAuB,EACvB,QAAsD,EACxB;IAC9B,IAAI,CAAC,OAAO,CAAC,IAAI,IAAI,OAAO,CAAC,OAAO,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACnD,OAAO;YACN,OAAO,EAAE,gBAAgB;YACzB,WAAW,EAAE,mBAAmB;YAChC,MAAM,EAAE,CAAC,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,gCAAgC,CAAC,CAAC,CAAC,iCAAiC;YAC5F,OAAO,EAAE,EAAE;SACX,CAAC;IACH,CAAC;IAED,MAAM,KAAK,GAAG,QAAQ,CAAC,GAAG,CAAC,OAAO,CAAC,IAAI,CAAC,CAAC;IACzC,MAAM,QAAQ,GAAiB,EAAE,CAAC;IAClC,KAAK,MAAM,KAAK,IAAI,OAAO,CAAC,OAAO,EAAE,CAAC;QACrC,QAAQ,CAAC,IAAI,CAAC,MAAM,aAAa,CAAC,KAAK,EAAE,KAAK,EAAE,MAAM,CAAC,CAAC,CAAC;IAC1D,CAAC;IAED,MAAM,MAAM,GAAoB,QAAQ,CAAC,GAAG,CAAC,CAAC,OAAO,EAAE,EAAE,CAAC,CAAC;QAC1D,MAAM,EAAE,OAAO,CAAC,MAAM;QACtB,OAAO,EAAE,OAAO,CAAC,OAAO;QACxB,WAAW,EAAE,OAAO,CAAC,WAAW;QAChC,MAAM,EAAE,OAAO,CAAC,MAAM;QACtB,aAAa,EAAE,OAAO,CAAC,QAAQ;KAC/B,CAAC,CAAC,CAAC;IAEJ,MAAM,OAAO,GAAG,cAAc,CAAC,MAAM,CAAC,CAAC;IACvC,MAAM,MAAM,GAAuB;QAClC,OAAO;QACP,WAAW,EAAE,qBAAqB,CAAC,OAAO,EAAE,MAAM,CAAC;QACnD,MAAM,EAAE,iBAAiB,CAAC,OAAO,EAAE,MAAM,CAAC;QAC1C,OAAO,EAAE,MAAM;KACf,CAAC;IAEF,IAAI,KAAK,CAAC,uBAAuB,IAAI,OAAO,KAAK,WAAW,EAAE,CAAC;QAC9D,MAAM,cAAc,GAAG,QAAQ,CAAC,GAAG,CAAC,CAAC,OAAO,EAAE,EAAE,CAAC,CAAC;YACjD,MAAM,EAAE,OAAO,CAAC,MAAM;YACtB,SAAS,EAAE,OAAO,CAAC,QAAQ,CAAC,SAAS;SACrC,CAAC,CAAC,CAAC;QACJ,MAAM,WAAW,GAAG,QAAQ,CAAC,GAAG,CAAC,CAAC,OAAO,EAAE,EAAE,CAAC,CAAC,EAAE,MAAM,EAAE,OAAO,CAAC,MAAM,EAAE,MAAM,EAAE,OAAO,CAAC,QAAQ,CAAC,MAAM,EAAE,CAAC,CAAC,CAAC;QAC7G,MAAM,CAAC,iBAAiB,GAAG,KAAK,CAAC,uBAAuB,CAAC,OAAO,EAAE,cAAc,EAAE,WAAW,CAAC,CAAC;IAChG,CAAC;IAED,OAAO,MAAM,CAAC;AAAA,CACd","sourcesContent":["/**\n * Outcome Adjudicator (Part 2 sections 2, 3, 5) - a wrapper-side verification layer that\n * turns AdaptOrch's own terminal-status report into a corroborated 5-state verdict, using\n * only `getRun`/`getArtifacts`/`getTraces` introspection.\n *\n * Out of scope for this file (not covered by the input contract available here): the\n * scope check and freshness check from Part 2 section 2.3, and the retry-count threshold\n * from section 2.3, since none of `AdjudicationRequest`, `VerifierRegistryEntry`, or the\n * client return shapes carry lane-scope, artifact-timestamp, or run-start-time data to\n * check them against. What is implemented here: the non-empty check (honoring\n * `allow_zero_artifacts`), the section 3 ambiguous-data fork, `content_check`,\n * `trace_check`, an error-span heuristic scan, and the `expected_min_actions` count.\n */\n\nimport type { AdaptOrchClient } from \"./adaptorch-client.ts\";\nimport type {\n\tAdjudicationReasonCode,\n\tCheckResult,\n\tVerdictState,\n\tVerifierRegistryEntry,\n} from \"./adjudicator-registry.ts\";\nimport { reduceReasonCodes } from \"./adjudicator-registry.ts\";\n\n/** Raw payload shape returned by `AdaptOrchClient.getRun`, inferred rather than duplicated. */\ntype RunPayload = Awaited<ReturnType<AdaptOrchClient[\"getRun\"]>>;\n/** Raw payload shape returned by `AdaptOrchClient.getArtifacts`, inferred rather than duplicated. */\ntype ArtifactsPayload = Awaited<ReturnType<AdaptOrchClient[\"getArtifacts\"]>>;\n/** Raw payload shape returned by `AdaptOrchClient.getTraces`, inferred rather than duplicated. */\ntype TracesPayload = Awaited<ReturnType<AdaptOrchClient[\"getTraces\"]>>;\n\n/**\n * Adjudication input (Part 2 section 2.1), sourced from the Dispatch Record and Work\n * Packet on the core-algorithm side. Both fields are mandatory; `run_ids` must be\n * non-empty (an empty list is treated as a malformed request, section 2.1/2.4).\n */\nexport interface AdjudicationRequest {\n\tdispatch_record_id: string;\n\tkind: string;\n\trun_ids: string[];\n}\n\n/**\n * A single `run_id`'s own verdict, reason, and evidence references (Part 2 section 5).\n * Preserved verbatim inside the record-level {@link AdjudicationResult}; never summarized\n * away, even when the request has only one `run_id`.\n */\nexport interface PerRunVerdict {\n\trun_id: string;\n\tverdict: VerdictState;\n\t/** Machine-readable classification; disposition logic branches on this, never on `reason`. */\n\treason_code: AdjudicationReasonCode;\n\t/** Human-readable explanation, for logs and review UIs only. */\n\treason: string;\n\tevidence_refs: unknown;\n}\n\n/**\n * Record-level adjudication output (Part 2 section 2.5 aggregation; section 5 record\n * shape). `augmented_payload` is present only when the looked-up registry entry defines\n * `build_augmented_payload` and the record-level verdict is not CONFIRMED (section 4).\n */\nexport interface AdjudicationResult {\n\tverdict: VerdictState;\n\t/**\n\t * Machine-readable classification, reduced worst-wins from the `per_run` entries that\n\t * share the record-level verdict. `projectVerdictToDisposition` branches on this code\n\t * exclusively; `reason` is never string-matched.\n\t */\n\treason_code: AdjudicationReasonCode;\n\t/** Human-readable explanation, for logs and review UIs only. */\n\treason: string;\n\tper_run: PerRunVerdict[];\n\taugmented_payload?: unknown;\n}\n\n/** Snapshot of the raw payloads a single `run_id`'s verdict was computed from (section 5). */\ninterface RunEvidence {\n\trun?: RunPayload;\n\tartifacts?: ArtifactsPayload;\n\ttraces?: TracesPayload;\n}\n\ninterface RunOutcome {\n\trun_id: string;\n\tverdict: VerdictState;\n\treason_code: AdjudicationReasonCode;\n\treason: string;\n\tevidence: RunEvidence;\n}\n\nconst NON_TERMINAL_STATUSES = new Set([\n\t\"pending\",\n\t\"queued\",\n\t\"running\",\n\t\"in_progress\",\n\t\"in-progress\",\n\t\"scheduled\",\n\t\"dispatched\",\n\t\"starting\",\n\t\"initializing\",\n]);\nconst SUCCESS_STATUSES = new Set([\"completed\", \"success\", \"succeeded\", \"done\", \"ok\", \"finished\"]);\nconst FAILURE_STATUSES = new Set([\n\t\"failed\",\n\t\"failure\",\n\t\"error\",\n\t\"errored\",\n\t\"cancelled\",\n\t\"canceled\",\n\t\"timeout\",\n\t\"timed_out\",\n\t\"aborted\",\n]);\n\ntype RunStatusBranch = \"non-terminal\" | \"success\" | \"failure\" | \"unparseable\";\n\nfunction isRecord(value: unknown): value is Record<string, unknown> {\n\treturn typeof value === \"object\" && value !== null;\n}\n\nfunction readStringField(value: unknown, keys: string[]): string | undefined {\n\tif (!isRecord(value)) return undefined;\n\tfor (const key of keys) {\n\t\tconst field = value[key];\n\t\tif (typeof field === \"string\") return field;\n\t}\n\treturn undefined;\n}\n\n/**\n * Interprets a `getRun` payload into the branch the OA cares about (Part 2 section 2.2).\n * A payload with no recognizable string status field is treated as unparseable - a\n * checker failure, not a target-task failure (section 2.4, VERIFIER-ERROR).\n */\nfunction interpretRunStatus(run: unknown): RunStatusBranch {\n\tconst status = readStringField(run, [\"status\", \"state\"]);\n\tif (status === undefined) return \"unparseable\";\n\tconst normalized = status.toLowerCase();\n\tif (NON_TERMINAL_STATUSES.has(normalized)) return \"non-terminal\";\n\tif (SUCCESS_STATUSES.has(normalized)) return \"success\";\n\tif (FAILURE_STATUSES.has(normalized)) return \"failure\";\n\treturn \"unparseable\";\n}\n\n/**\n * Coerces an unknown `getArtifacts`/`getTraces` payload into a flat list without assuming\n * a specific shape (the concrete return type belongs to the concurrently-authored\n * `adaptorch-client.ts`). Falls back to common pagination-style container keys, then to\n * treating a single non-null value as a one-item list.\n */\nfunction asList(value: unknown): unknown[] {\n\tif (value === null || value === undefined) return [];\n\tif (Array.isArray(value)) return value;\n\tif (isRecord(value)) {\n\t\tfor (const key of [\"items\", \"artifacts\", \"traces\", \"spans\", \"entries\", \"data\", \"results\"]) {\n\t\t\tconst field = value[key];\n\t\t\tif (Array.isArray(field)) return field;\n\t\t}\n\t}\n\treturn [value];\n}\n\n/** True unless the item is a recognizably empty/whitespace-only value (Part 2 section 2.3). */\nfunction hasSubstance(item: unknown): boolean {\n\tif (typeof item === \"string\") return item.trim().length > 0;\n\tif (isRecord(item)) {\n\t\tconst size = item.size ?? item.byteLength ?? item.length;\n\t\tif (typeof size === \"number\") return size > 0;\n\t\tconst text = readStringField(item, [\"content\", \"text\", \"body\"]);\n\t\tif (text !== undefined) return text.trim().length > 0;\n\t}\n\treturn true;\n}\n\n/** Heuristic ERROR-severity span scan (Part 2 section 2.3), tolerant of unknown span shapes. */\nfunction countErrorSpans(traces: unknown[]): number {\n\tlet count = 0;\n\tfor (const span of traces) {\n\t\tif (!isRecord(span)) continue;\n\t\tconst level = readStringField(span, [\"level\", \"severity\", \"status\"]);\n\t\tif (level !== undefined && level.toLowerCase() === \"error\") {\n\t\t\tcount += 1;\n\t\t\tcontinue;\n\t\t}\n\t\tif (span.error === true || span.isError === true) count += 1;\n\t}\n\treturn count;\n}\n\n/**\n * Heuristic count of \"action\" spans for the `expected_min_actions` check (Part 2 sections\n * 2.3/4). Recognizes a handful of common marker fields; if none of the spans carry any of\n * them, falls back to the total span count so the check degrades to a coarse presence\n * signal instead of always failing.\n */\nfunction countActionSpans(traces: unknown[]): number {\n\tlet recognized = 0;\n\tlet matched = 0;\n\tfor (const span of traces) {\n\t\tif (!isRecord(span)) continue;\n\t\tconst marker = readStringField(span, [\"action\", \"tool_call\", \"toolCall\", \"kind\", \"type\"]);\n\t\tif (marker !== undefined) {\n\t\t\trecognized += 1;\n\t\t\tif (/action|tool|write|edit|call/i.test(marker)) matched += 1;\n\t\t}\n\t}\n\treturn recognized > 0 ? matched : traces.length;\n}\n\nfunction describeError(error: unknown): string {\n\treturn error instanceof Error ? error.message : String(error);\n}\n\n/**\n * Adjudicates a single `run_id` against the resolved registry entry (Part 2 sections\n * 2.2-2.4 and 3). Robust to a misbehaving fetch: any thrown error from the client during\n * this run's own fetch/check sequence becomes that run_id's own VERIFIER-ERROR rather\n * than propagating out of {@link adjudicate}.\n */\nasync function adjudicateRun(\n\trunId: string,\n\tentry: VerifierRegistryEntry,\n\tclient: AdaptOrchClient,\n): Promise<RunOutcome> {\n\tconst evidence: RunEvidence = {};\n\ttry {\n\t\tconst run = await client.getRun(runId);\n\t\tevidence.run = run;\n\n\t\tconst branch = interpretRunStatus(run);\n\t\tif (branch === \"unparseable\") {\n\t\t\treturn {\n\t\t\t\trun_id: runId,\n\t\t\t\tverdict: \"VERIFIER-ERROR\",\n\t\t\t\treason_code: \"RUN_STATUS_UNPARSEABLE\",\n\t\t\t\treason: \"run-status-unparseable\",\n\t\t\t\tevidence,\n\t\t\t};\n\t\t}\n\t\tif (branch === \"non-terminal\") {\n\t\t\treturn {\n\t\t\t\trun_id: runId,\n\t\t\t\tverdict: \"INDETERMINATE\",\n\t\t\t\treason_code: \"RUN_NOT_TERMINAL\",\n\t\t\t\treason: \"run-not-terminal\",\n\t\t\t\tevidence,\n\t\t\t};\n\t\t}\n\n\t\tconst artifacts = await client.getArtifacts(runId);\n\t\tevidence.artifacts = artifacts;\n\t\tconst traces = await client.getTraces(runId);\n\t\tevidence.traces = traces;\n\n\t\tconst artifactsList = asList(artifacts);\n\t\tconst tracesList = asList(traces);\n\t\tconst artifactsEmpty = artifactsList.length === 0;\n\t\tconst tracesEmpty = tracesList.length === 0;\n\t\tconst isSuccess = branch === \"success\";\n\n\t\t// Part 2 section 3: absence of contrary evidence is never read as confirming evidence.\n\t\tif ((artifactsEmpty && !entry.allow_zero_artifacts) || tracesEmpty) {\n\t\t\tif (isSuccess && artifactsEmpty && tracesEmpty && entry.no_evidence_on_success_is_contradiction) {\n\t\t\t\treturn {\n\t\t\t\t\trun_id: runId,\n\t\t\t\t\tverdict: \"CONTRADICTED\",\n\t\t\t\t\treason_code: \"NO_EVIDENCE_ON_SUCCESS\",\n\t\t\t\t\treason: \"no-evidence-on-success-is-contradiction\",\n\t\t\t\t\tevidence,\n\t\t\t\t};\n\t\t\t}\n\t\t\tconst reasons: string[] = [];\n\t\t\tif (artifactsEmpty && !entry.allow_zero_artifacts) reasons.push(\"artifacts-empty-unexpected\");\n\t\t\tif (tracesEmpty) reasons.push(\"traces-empty-unexpected\");\n\t\t\treturn {\n\t\t\t\trun_id: runId,\n\t\t\t\tverdict: \"INDETERMINATE\",\n\t\t\t\treason_code: \"EVIDENCE_EMPTY\",\n\t\t\t\treason: reasons.join(\",\"),\n\t\t\t\tevidence,\n\t\t\t};\n\t\t}\n\n\t\tif (!isSuccess) {\n\t\t\t// CORROBORATED-FAILURE: reported failure, and the ambiguous-data fork above\n\t\t\t// already ruled out the case where evidence was too thin to say anything.\n\t\t\treturn {\n\t\t\t\trun_id: runId,\n\t\t\t\tverdict: \"CORROBORATED-FAILURE\",\n\t\t\t\treason_code: \"FAILURE_REPORTED\",\n\t\t\t\treason: \"failure-reported\",\n\t\t\t\tevidence,\n\t\t\t};\n\t\t}\n\n\t\tconst problems: { code: AdjudicationReasonCode; message: string }[] = [];\n\t\tif (artifactsList.some((artifact) => !hasSubstance(artifact))) {\n\t\t\tproblems.push({ code: \"EMPTY_ARTIFACT_CONTENT\", message: \"artifact-empty-or-whitespace\" });\n\t\t}\n\t\tif (entry.content_check) {\n\t\t\tfor (const artifact of artifactsList) {\n\t\t\t\tconst result: CheckResult = entry.content_check(artifact);\n\t\t\t\tif (!result.ok) {\n\t\t\t\t\tproblems.push({\n\t\t\t\t\t\tcode: result.code ?? \"CONTENT_CHECK_FAILED\",\n\t\t\t\t\t\tmessage: `content-check-failed${result.reason ? `: ${result.reason}` : \"\"}`,\n\t\t\t\t\t});\n\t\t\t\t}\n\t\t\t}\n\t\t}\n\t\tif (entry.trace_check) {\n\t\t\tconst result: CheckResult = entry.trace_check(traces);\n\t\t\tif (!result.ok) {\n\t\t\t\tproblems.push({\n\t\t\t\t\tcode: result.code ?? \"TRACE_CHECK_FAILED\",\n\t\t\t\t\tmessage: `trace-check-failed${result.reason ? `: ${result.reason}` : \"\"}`,\n\t\t\t\t});\n\t\t\t}\n\t\t}\n\t\tconst errorSpanCount = countErrorSpans(tracesList);\n\t\tif (errorSpanCount > 0) {\n\t\t\tproblems.push({ code: \"TRACE_ERROR_SPAN\", message: `error-span-scan: ${errorSpanCount} error span(s) found` });\n\t\t}\n\t\tif (typeof entry.expected_min_actions === \"number\" && entry.expected_min_actions > 0) {\n\t\t\tconst actionCount = countActionSpans(tracesList);\n\t\t\tif (actionCount < entry.expected_min_actions) {\n\t\t\t\tproblems.push({\n\t\t\t\t\tcode: \"MIN_ACTIONS_UNMET\",\n\t\t\t\t\tmessage: `expected-min-actions-not-met: found ${actionCount}, needed ${entry.expected_min_actions}`,\n\t\t\t\t});\n\t\t\t}\n\t\t}\n\n\t\tif (problems.length > 0) {\n\t\t\treturn {\n\t\t\t\trun_id: runId,\n\t\t\t\tverdict: \"CONTRADICTED\",\n\t\t\t\treason_code: reduceReasonCodes(problems.map((problem) => problem.code)),\n\t\t\t\treason: problems.map((problem) => problem.message).join(\"; \"),\n\t\t\t\tevidence,\n\t\t\t};\n\t\t}\n\t\treturn {\n\t\t\trun_id: runId,\n\t\t\tverdict: \"CONFIRMED\",\n\t\t\treason_code: \"ALL_CHECKS_PASSED\",\n\t\t\treason: \"all-checks-passed\",\n\t\t\tevidence,\n\t\t};\n\t} catch (error) {\n\t\treturn {\n\t\t\trun_id: runId,\n\t\t\tverdict: \"VERIFIER-ERROR\",\n\t\t\treason_code: \"RUN_FETCH_FAILED\",\n\t\t\treason: `run-adjudication-failed: ${describeError(error)}`,\n\t\t\tevidence,\n\t\t};\n\t}\n}\n\n/** Worst-wins record-level reduction over per-`run_id` verdicts (Part 2 section 2.5). */\nfunction reduceVerdicts(perRun: PerRunVerdict[]): VerdictState {\n\tconst priority: VerdictState[] = [\n\t\t\"VERIFIER-ERROR\",\n\t\t\"CONTRADICTED\",\n\t\t\"CORROBORATED-FAILURE\",\n\t\t\"INDETERMINATE\",\n\t\t\"CONFIRMED\",\n\t];\n\tfor (const state of priority) {\n\t\tif (perRun.some((r) => r.verdict === state)) return state;\n\t}\n\treturn \"CONFIRMED\";\n}\n\nfunction buildRecordReason(verdict: VerdictState, perRun: PerRunVerdict[]): string {\n\tif (perRun.length === 1) return perRun[0].reason;\n\tconst contributing = perRun.filter((r) => r.verdict === verdict).map((r) => `${r.run_id}: ${r.reason}`);\n\treturn `${verdict} via ${contributing.join(\"; \")}`;\n}\n\n/** Worst-wins record-level reason code, reduced over the runs sharing the record verdict. */\nfunction buildRecordReasonCode(verdict: VerdictState, perRun: PerRunVerdict[]): AdjudicationReasonCode {\n\treturn reduceReasonCodes(perRun.filter((r) => r.verdict === verdict).map((r) => r.reason_code));\n}\n\n/**\n * Adjudicates an `AdjudicationRequest` end to end (Part 2 sections 2.1-2.5, 3, 5).\n *\n * 1. Rejects malformed requests (missing `kind` or empty `run_ids`) as a record-level\n * VERIFIER-ERROR with no per-`run_id` results (section 2.1/2.4).\n * 2. Resolves the per-kind verifier once (section 4) and applies it identically to every\n * `run_id`, independently (section 2.2-2.4, 3).\n * 3. Reduces the per-`run_id` verdicts to one record-level verdict, worst-wins (section\n * 2.5), and never discards the per-`run_id` detail (section 5).\n * 4. Invokes `build_augmented_payload`, if defined, when the record-level verdict is not\n * CONFIRMED (section 4).\n */\nexport async function adjudicate(\n\trequest: AdjudicationRequest,\n\tclient: AdaptOrchClient,\n\tregistry: { get(kind: string): VerifierRegistryEntry },\n): Promise<AdjudicationResult> {\n\tif (!request.kind || request.run_ids.length === 0) {\n\t\treturn {\n\t\t\tverdict: \"VERIFIER-ERROR\",\n\t\t\treason_code: \"MALFORMED_REQUEST\",\n\t\t\treason: !request.kind ? \"malformed-request-missing-kind\" : \"malformed-request-empty-run-ids\",\n\t\t\tper_run: [],\n\t\t};\n\t}\n\n\tconst entry = registry.get(request.kind);\n\tconst outcomes: RunOutcome[] = [];\n\tfor (const runId of request.run_ids) {\n\t\toutcomes.push(await adjudicateRun(runId, entry, client));\n\t}\n\n\tconst perRun: PerRunVerdict[] = outcomes.map((outcome) => ({\n\t\trun_id: outcome.run_id,\n\t\tverdict: outcome.verdict,\n\t\treason_code: outcome.reason_code,\n\t\treason: outcome.reason,\n\t\tevidence_refs: outcome.evidence,\n\t}));\n\n\tconst verdict = reduceVerdicts(perRun);\n\tconst result: AdjudicationResult = {\n\t\tverdict,\n\t\treason_code: buildRecordReasonCode(verdict, perRun),\n\t\treason: buildRecordReason(verdict, perRun),\n\t\tper_run: perRun,\n\t};\n\n\tif (entry.build_augmented_payload && verdict !== \"CONFIRMED\") {\n\t\tconst artifactsByRun = outcomes.map((outcome) => ({\n\t\t\trun_id: outcome.run_id,\n\t\t\tartifacts: outcome.evidence.artifacts,\n\t\t}));\n\t\tconst tracesByRun = outcomes.map((outcome) => ({ run_id: outcome.run_id, traces: outcome.evidence.traces }));\n\t\tresult.augmented_payload = entry.build_augmented_payload(verdict, artifactsByRun, tracesByRun);\n\t}\n\n\treturn result;\n}\n"]}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Maps OA adjudication + policy flags to B2C {@link VerdictCard} and apply/submit gates.
|
|
3
|
+
*/
|
|
4
|
+
import type { AdjudicationResult } from "./adjudicator.ts";
|
|
5
|
+
import { type VerdictCard, type VerificationReceipt } from "./b2c-verdict.ts";
|
|
6
|
+
import { type PolicyFlag } from "./policy-wall.ts";
|
|
7
|
+
export interface MapToB2CInput {
|
|
8
|
+
kind: string;
|
|
9
|
+
packetId?: string;
|
|
10
|
+
dispatchRecordId?: string;
|
|
11
|
+
runIds: string[];
|
|
12
|
+
previewOnly: boolean;
|
|
13
|
+
policyFlags: PolicyFlag[];
|
|
14
|
+
diffPaths: string[];
|
|
15
|
+
adjudication?: AdjudicationResult;
|
|
16
|
+
evaluatedAt?: string;
|
|
17
|
+
repairHints?: string[];
|
|
18
|
+
}
|
|
19
|
+
export interface MapToB2COutput {
|
|
20
|
+
verdictCard: VerdictCard;
|
|
21
|
+
receipt: VerificationReceipt;
|
|
22
|
+
}
|
|
23
|
+
export declare const BATCH1_NO_DOCKER_RUNNER: "BATCH1_NO_DOCKER_RUNNER";
|
|
24
|
+
export declare function mapToB2C(input: MapToB2CInput): MapToB2COutput;
|
|
25
|
+
//# sourceMappingURL=b2c-mapper.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"b2c-mapper.d.ts","sourceRoot":"","sources":["../src/b2c-mapper.ts"],"names":[],"mappings":"AAAA;;GAEG;AAEH,OAAO,KAAK,EAAE,kBAAkB,EAAE,MAAM,kBAAkB,CAAC;AAE3D,OAAO,EAIN,KAAK,WAAW,EAGhB,KAAK,mBAAmB,EACxB,MAAM,kBAAkB,CAAC;AAC1B,OAAO,EAAe,KAAK,UAAU,EAAE,MAAM,kBAAkB,CAAC;AAkFhE,MAAM,WAAW,aAAa;IAC7B,IAAI,EAAE,MAAM,CAAC;IACb,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,MAAM,EAAE,MAAM,EAAE,CAAC;IACjB,WAAW,EAAE,OAAO,CAAC;IACrB,WAAW,EAAE,UAAU,EAAE,CAAC;IAC1B,SAAS,EAAE,MAAM,EAAE,CAAC;IACpB,YAAY,CAAC,EAAE,kBAAkB,CAAC;IAClC,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,WAAW,CAAC,EAAE,MAAM,EAAE,CAAC;CACvB;AAED,MAAM,WAAW,cAAc;IAC9B,WAAW,EAAE,WAAW,CAAC;IACzB,OAAO,EAAE,mBAAmB,CAAC;CAC7B;AAED,eAAO,MAAM,uBAAuB,2BAAqC,CAAC;AA8B1E,wBAAgB,QAAQ,CAAC,KAAK,EAAE,aAAa,GAAG,cAAc,CA0F7D","sourcesContent":["/**\n * Maps OA adjudication + policy flags to B2C {@link VerdictCard} and apply/submit gates.\n */\n\nimport type { AdjudicationResult } from \"./adjudicator.ts\";\nimport type { AdjudicationReasonCode, VerdictState } from \"./adjudicator-registry.ts\";\nimport {\n\tB2C_VERDICT_SCHEMA_VERSION,\n\ttype UserRiskLevel,\n\ttype UserVerdict,\n\ttype VerdictCard,\n\ttype VerdictLimits,\n\ttype VerdictNextAction,\n\ttype VerificationReceipt,\n} from \"./b2c-verdict.ts\";\nimport { POLICY_FLAG, type PolicyFlag } from \"./policy-wall.ts\";\n\nconst USER_VERDICT_SEVERITY: UserVerdict[] = [\"BLOCKED\", \"INCONCLUSIVE\", \"ADVISORY\", \"PASS\"];\n\nfunction worstUserVerdict(current: UserVerdict, next: UserVerdict): UserVerdict {\n\tconst cur = USER_VERDICT_SEVERITY.indexOf(current);\n\tconst nxt = USER_VERDICT_SEVERITY.indexOf(next);\n\treturn cur <= nxt ? current : next;\n}\n\nfunction verdictFromOa(verdict: VerdictState): UserVerdict {\n\tswitch (verdict) {\n\t\tcase \"CONFIRMED\":\n\t\t\treturn \"PASS\";\n\t\tcase \"CORROBORATED-FAILURE\":\n\t\t\treturn \"ADVISORY\";\n\t\tcase \"CONTRADICTED\":\n\t\t\treturn \"BLOCKED\";\n\t\tcase \"INDETERMINATE\":\n\t\t\treturn \"INCONCLUSIVE\";\n\t\tcase \"VERIFIER-ERROR\":\n\t\t\treturn \"INCONCLUSIVE\";\n\t}\n}\n\nfunction riskFromUserVerdict(verdict: UserVerdict, flags: PolicyFlag[]): UserRiskLevel {\n\tif (verdict === \"BLOCKED\" || flags.includes(POLICY_FLAG.NON_NEGOTIABLE_BLOCKING)) {\n\t\treturn \"critical\";\n\t}\n\tif (verdict === \"INCONCLUSIVE\") return \"high\";\n\tif (verdict === \"ADVISORY\") return \"medium\";\n\treturn \"low\";\n}\n\nconst REASON_CODE_USER: Record<AdjudicationReasonCode, { passed?: string; blocked?: string }> = {\n\tSCOPE_VIOLATION: { blocked: \"Write scope violation detected by outcome adjudication.\" },\n\tSCHEMA_DRIFT: { blocked: \"Output schema does not match the expected contract.\" },\n\tCONTENT_CHECK_FAILED: { blocked: \"Content verification failed for one or more artifacts.\" },\n\tNO_EVIDENCE_ON_SUCCESS: { blocked: \"Run reported success but left no usable evidence.\" },\n\tEMPTY_ARTIFACT_CONTENT: { blocked: \"An artifact was present but contained no substantive content.\" },\n\tTRACE_CHECK_FAILED: { blocked: \"Trace verification failed for this run.\" },\n\tTRACE_ERROR_SPAN: { blocked: \"Error-level spans were found under a reported success.\" },\n\tMIN_ACTIONS_UNMET: { blocked: \"Fewer actions were recorded than required for this kind.\" },\n\tRUN_STATUS_UNPARSEABLE: { blocked: \"Run status could not be interpreted.\" },\n\tRUN_FETCH_FAILED: { blocked: \"Evidence could not be fetched for this run.\" },\n\tMALFORMED_REQUEST: { blocked: \"Adjudication request was malformed.\" },\n\tRUN_NOT_TERMINAL: { blocked: \"Run has not reached a terminal status yet.\" },\n\tEVIDENCE_EMPTY: { blocked: \"Evidence was too thin to reach a confident verdict.\" },\n\tFAILURE_REPORTED: { blocked: \"The run reported failure and evidence agrees.\" },\n\tALL_CHECKS_PASSED: { passed: \"All outcome adjudication checks passed.\" },\n};\n\nfunction userStringsForReasonCode(code: AdjudicationReasonCode): { passed: string[]; blocked: string[] } {\n\tconst entry = REASON_CODE_USER[code];\n\tconst passed: string[] = [];\n\tconst blocked: string[] = [];\n\tif (entry.passed) passed.push(entry.passed);\n\tif (entry.blocked) blocked.push(entry.blocked);\n\treturn { passed, blocked };\n}\n\nfunction userVerdictFromPolicyFlags(flags: PolicyFlag[], previewOnly: boolean): UserVerdict | null {\n\tlet v: UserVerdict = \"PASS\";\n\tif (\n\t\tflags.includes(POLICY_FLAG.NON_NEGOTIABLE_BLOCKING) ||\n\t\tflags.includes(POLICY_FLAG.CANDIDATE_LEAK_SUSPECT) ||\n\t\tflags.includes(POLICY_FLAG.SECRET_SUSPECT)\n\t) {\n\t\tv = worstUserVerdict(v, \"BLOCKED\");\n\t}\n\tif (flags.includes(POLICY_FLAG.EVIDENCE_DAG_INCOMPLETE)) {\n\t\tv = worstUserVerdict(v, \"INCONCLUSIVE\");\n\t}\n\tif (\n\t\tpreviewOnly &&\n\t\t(flags.includes(POLICY_FLAG.REPRO_OVERFIT_SUSPECT) || flags.includes(POLICY_FLAG.LOW_DISCRIMINATION))\n\t) {\n\t\tv = worstUserVerdict(v, \"ADVISORY\");\n\t}\n\treturn v === \"PASS\" && flags.length === 0 ? null : v;\n}\n\nexport interface MapToB2CInput {\n\tkind: string;\n\tpacketId?: string;\n\tdispatchRecordId?: string;\n\trunIds: string[];\n\tpreviewOnly: boolean;\n\tpolicyFlags: PolicyFlag[];\n\tdiffPaths: string[];\n\tadjudication?: AdjudicationResult;\n\tevaluatedAt?: string;\n\trepairHints?: string[];\n}\n\nexport interface MapToB2COutput {\n\tverdictCard: VerdictCard;\n\treceipt: VerificationReceipt;\n}\n\nexport const BATCH1_NO_DOCKER_RUNNER = \"BATCH1_NO_DOCKER_RUNNER\" as const;\n\nfunction buildNextActions(verdict: UserVerdict, previewOnly: boolean, canApply: boolean): VerdictNextAction[] {\n\tif (verdict === \"BLOCKED\") {\n\t\treturn [\"Regenerate\"];\n\t}\n\tif (verdict === \"INCONCLUSIVE\") {\n\t\treturn previewOnly ? [\"Deep Check\", \"Regenerate\"] : [\"Deep Check\"];\n\t}\n\tif (verdict === \"ADVISORY\") {\n\t\treturn previewOnly ? [\"Deep Check\", \"Apply\"] : [\"Apply\"];\n\t}\n\tif (canApply) {\n\t\treturn [\"Apply\"];\n\t}\n\treturn [\"Deep Check\"];\n}\n\nfunction buildLimitsCode(previewOnly: boolean): string | undefined {\n\treturn previewOnly ? BATCH1_NO_DOCKER_RUNNER : undefined;\n}\n\nfunction buildDisclaimer(previewOnly: boolean, limitsCode: string | undefined): string | undefined {\n\tif (!previewOnly) {\n\t\treturn undefined;\n\t}\n\tconst code = limitsCode ?? BATCH1_NO_DOCKER_RUNNER;\n\treturn `Preview-only evaluation (${code}): no docker runner; apply and submit gates are conservative.`;\n}\n\nexport function mapToB2C(input: MapToB2CInput): MapToB2COutput {\n\tconst evaluatedAt = input.evaluatedAt ?? new Date().toISOString();\n\tlet userVerdict: UserVerdict = \"INCONCLUSIVE\";\n\tconst passed_checks: string[] = [];\n\tconst blocked_reasons: string[] = [];\n\n\tif (input.diffPaths.length === 0 && input.adjudication === undefined) {\n\t\tuserVerdict = \"INCONCLUSIVE\";\n\t\tblocked_reasons.push(\"No changed paths were found in the diff; cannot assess write scope.\");\n\t} else {\n\t\tuserVerdict = \"PASS\";\n\t}\n\n\tconst policyVerdict = userVerdictFromPolicyFlags(input.policyFlags, input.previewOnly);\n\tif (policyVerdict !== null) {\n\t\tuserVerdict = worstUserVerdict(userVerdict, policyVerdict);\n\t}\n\n\tif (input.policyFlags.includes(POLICY_FLAG.SECRET_SUSPECT)) {\n\t\tblocked_reasons.push(\"Diff lines may contain secrets or credentials; remove before proceeding.\");\n\t}\n\tif (input.policyFlags.includes(POLICY_FLAG.CANDIDATE_LEAK_SUSPECT)) {\n\t\tblocked_reasons.push(\"One or more changed paths fall outside the approved write scope.\");\n\t}\n\tif (input.policyFlags.includes(POLICY_FLAG.EVIDENCE_DAG_INCOMPLETE)) {\n\t\tblocked_reasons.push(\"Run evidence is required but was not evaluated in this receipt.\");\n\t}\n\tif (input.previewOnly && input.policyFlags.includes(POLICY_FLAG.LOW_DISCRIMINATION)) {\n\t\tblocked_reasons.push(\"Preview-only mode: discrimination against production evidence is limited.\");\n\t}\n\n\tif (input.adjudication !== undefined) {\n\t\tconst fromOa = verdictFromOa(input.adjudication.verdict);\n\t\tuserVerdict = worstUserVerdict(userVerdict, fromOa);\n\t\tconst strings = userStringsForReasonCode(input.adjudication.reason_code);\n\t\tpassed_checks.push(...strings.passed);\n\t\tblocked_reasons.push(...strings.blocked);\n\t} else if (userVerdict === \"PASS\" && input.diffPaths.length > 0) {\n\t\tpassed_checks.push(\"Changed paths are within approved write scope (fast wall).\");\n\t}\n\n\tconst limitsCode = buildLimitsCode(input.previewOnly);\n\tconst limits: VerdictLimits = {\n\t\trequiresHumanReview: userVerdict === \"BLOCKED\" || userVerdict === \"INCONCLUSIVE\",\n\t\tpreviewOnly: input.previewOnly,\n\t\tcode: limitsCode,\n\t};\n\n\tconst canApply =\n\t\tuserVerdict === \"PASS\" &&\n\t\t!input.policyFlags.includes(POLICY_FLAG.NON_NEGOTIABLE_BLOCKING) &&\n\t\t!input.policyFlags.includes(POLICY_FLAG.CANDIDATE_LEAK_SUSPECT) &&\n\t\t!input.policyFlags.includes(POLICY_FLAG.SECRET_SUSPECT) &&\n\t\t(input.adjudication === undefined || input.adjudication.verdict === \"CONFIRMED\");\n\n\tconst shouldSubmit = userVerdict !== \"BLOCKED\" && !input.policyFlags.includes(POLICY_FLAG.NON_NEGOTIABLE_BLOCKING);\n\n\tconst disclaimer = buildDisclaimer(input.previewOnly, limitsCode);\n\n\tconst verdictCard: VerdictCard = {\n\t\tschemaVersion: B2C_VERDICT_SCHEMA_VERSION,\n\t\tverdict: userVerdict,\n\t\trisk: riskFromUserVerdict(userVerdict, input.policyFlags),\n\t\tlimits,\n\t\tpassed_checks,\n\t\tblocked_reasons: [...new Set(blocked_reasons)],\n\t\tnext_actions: buildNextActions(userVerdict, input.previewOnly, canApply),\n\t\trepairHints: input.repairHints !== undefined && input.repairHints.length > 0 ? input.repairHints : undefined,\n\t\tpacketId: input.packetId,\n\t\tkind: input.kind,\n\t};\n\n\tconst receipt: VerificationReceipt = {\n\t\tschemaVersion: B2C_VERDICT_SCHEMA_VERSION,\n\t\tevaluatedAt,\n\t\tkind: input.kind,\n\t\tpacketId: input.packetId,\n\t\tdispatchRecordId: input.dispatchRecordId,\n\t\trunIds: input.runIds,\n\t\tpreviewOnly: input.previewOnly,\n\t\tcanApply,\n\t\tshouldSubmit,\n\t\tpolicyFlags: [...input.policyFlags],\n\t\tadjudicationVerdict: input.adjudication?.verdict,\n\t\tadjudicationReasonCode: input.adjudication?.reason_code,\n\t\tdiffPaths: input.diffPaths,\n\t\tdisclaimer,\n\t};\n\n\treturn { verdictCard, receipt };\n}\n"]}
|