@pome-sh/cli 0.17.0 → 0.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/README.md +13 -5
  2. package/dist/build-info.json +3 -3
  3. package/dist/src/capture-server/index.js +1 -1
  4. package/dist/src/cli/eval.js +12 -10
  5. package/dist/src/cli/main.js +1 -1
  6. package/dist/src/hosted/evalResultView.d.ts +1 -1
  7. package/dist/src/hosted/evalResultView.js +22 -5
  8. package/dist/src/runner/groupRender.d.ts +12 -2
  9. package/dist/src/runner/groupRender.js +32 -8
  10. package/dist/src/runner/mergeAdapterSignals.d.ts +23 -0
  11. package/dist/src/runner/mergeAdapterSignals.js +45 -3
  12. package/dist/src/runner/runTaskHosted.d.ts +8 -1
  13. package/dist/src/runner/runTaskHosted.js +11 -1
  14. package/dist/src/runner/runTrialGroup.js +15 -2
  15. package/dist/src/scaffolds/mcp-loop/loop.js +3 -3
  16. package/node_modules/@pome-sh/sdk/dist/check-state-path.d.ts +71 -0
  17. package/node_modules/@pome-sh/sdk/dist/check-state-path.js +200 -0
  18. package/node_modules/@pome-sh/sdk/dist/check-state-path.js.map +1 -0
  19. package/node_modules/@pome-sh/sdk/dist/checks.d.ts +2 -0
  20. package/node_modules/@pome-sh/sdk/dist/checks.js +5 -0
  21. package/node_modules/@pome-sh/sdk/dist/checks.js.map +1 -1
  22. package/node_modules/@pome-sh/sdk/dist/recorder.d.ts +10 -1
  23. package/node_modules/@pome-sh/sdk/dist/recorder.js +11 -2
  24. package/node_modules/@pome-sh/sdk/dist/recorder.js.map +1 -1
  25. package/node_modules/@pome-sh/sdk/package.json +2 -2
  26. package/node_modules/@pome-sh/shared-types/dist/otel/event-schema.d.ts +111 -24
  27. package/node_modules/@pome-sh/shared-types/dist/otel/fixtures/data.js +3 -3
  28. package/node_modules/@pome-sh/shared-types/dist/otel/fixtures/data.js.map +1 -1
  29. package/node_modules/@pome-sh/shared-types/dist/otel/legacy-shim.d.ts +6 -3
  30. package/node_modules/@pome-sh/shared-types/dist/otel/legacy-shim.js +7 -2
  31. package/node_modules/@pome-sh/shared-types/dist/otel/legacy-shim.js.map +1 -1
  32. package/node_modules/@pome-sh/shared-types/dist/otel/map-span.js +1 -1
  33. package/node_modules/@pome-sh/shared-types/dist/otel/map-span.js.map +1 -1
  34. package/node_modules/@pome-sh/shared-types/dist/otel/span-event.d.ts +68 -4
  35. package/node_modules/@pome-sh/shared-types/dist/otel/span-event.js +19 -5
  36. package/node_modules/@pome-sh/shared-types/dist/otel/span-event.js.map +1 -1
  37. package/node_modules/@pome-sh/shared-types/dist/recorder-events.d.ts +60 -28
  38. package/node_modules/@pome-sh/shared-types/dist/recorder-events.js +81 -10
  39. package/node_modules/@pome-sh/shared-types/dist/recorder-events.js.map +1 -1
  40. package/node_modules/@pome-sh/shared-types/package.json +1 -1
  41. package/node_modules/@pome-sh/shared-types/trace-contract.json +49 -1
  42. package/node_modules/@pome-sh/twin-github/dist/src/check-issues.js +35 -9
  43. package/node_modules/@pome-sh/twin-github/dist/src/check-issues.js.map +1 -1
  44. package/node_modules/@pome-sh/twin-github/dist/src/check-pulls.js +21 -5
  45. package/node_modules/@pome-sh/twin-github/dist/src/check-pulls.js.map +1 -1
  46. package/node_modules/@pome-sh/twin-github/dist/src/check-repos.js +26 -6
  47. package/node_modules/@pome-sh/twin-github/dist/src/check-repos.js.map +1 -1
  48. package/node_modules/@pome-sh/twin-github/dist/src/check-state.d.ts +20 -1
  49. package/node_modules/@pome-sh/twin-github/dist/src/check-state.js +62 -10
  50. package/node_modules/@pome-sh/twin-github/dist/src/check-state.js.map +1 -1
  51. package/node_modules/@pome-sh/twin-github/package.json +3 -3
  52. package/node_modules/@pome-sh/twin-gmail/CHANGELOG.md +29 -0
  53. package/node_modules/@pome-sh/twin-gmail/dist/src/check-drafts.js +19 -4
  54. package/node_modules/@pome-sh/twin-gmail/dist/src/check-drafts.js.map +1 -1
  55. package/node_modules/@pome-sh/twin-gmail/dist/src/check-labels.js +15 -4
  56. package/node_modules/@pome-sh/twin-gmail/dist/src/check-labels.js.map +1 -1
  57. package/node_modules/@pome-sh/twin-gmail/dist/src/check-messages.js +19 -3
  58. package/node_modules/@pome-sh/twin-gmail/dist/src/check-messages.js.map +1 -1
  59. package/node_modules/@pome-sh/twin-gmail/dist/src/check-state.d.ts +26 -1
  60. package/node_modules/@pome-sh/twin-gmail/dist/src/check-state.js +52 -16
  61. package/node_modules/@pome-sh/twin-gmail/dist/src/check-state.js.map +1 -1
  62. package/node_modules/@pome-sh/twin-gmail/package.json +2 -2
  63. package/node_modules/@pome-sh/twin-linear/CHANGELOG.md +29 -0
  64. package/node_modules/@pome-sh/twin-linear/dist/src/check-comments.js +15 -3
  65. package/node_modules/@pome-sh/twin-linear/dist/src/check-comments.js.map +1 -1
  66. package/node_modules/@pome-sh/twin-linear/dist/src/check-issues.js +31 -4
  67. package/node_modules/@pome-sh/twin-linear/dist/src/check-issues.js.map +1 -1
  68. package/node_modules/@pome-sh/twin-linear/dist/src/check-state.d.ts +27 -2
  69. package/node_modules/@pome-sh/twin-linear/dist/src/check-state.js +73 -20
  70. package/node_modules/@pome-sh/twin-linear/dist/src/check-state.js.map +1 -1
  71. package/node_modules/@pome-sh/twin-linear/package.json +2 -2
  72. package/node_modules/@pome-sh/twin-slack/dist/src/check-messages.js +28 -7
  73. package/node_modules/@pome-sh/twin-slack/dist/src/check-messages.js.map +1 -1
  74. package/node_modules/@pome-sh/twin-slack/dist/src/check-secrets.js +16 -3
  75. package/node_modules/@pome-sh/twin-slack/dist/src/check-secrets.js.map +1 -1
  76. package/node_modules/@pome-sh/twin-slack/dist/src/check-state.d.ts +26 -1
  77. package/node_modules/@pome-sh/twin-slack/dist/src/check-state.js +31 -4
  78. package/node_modules/@pome-sh/twin-slack/dist/src/check-state.js.map +1 -1
  79. package/node_modules/@pome-sh/twin-slack/package.json +3 -3
  80. package/node_modules/@pome-sh/twin-stripe/dist/src/check-payments.d.ts +1 -1
  81. package/node_modules/@pome-sh/twin-stripe/dist/src/check-payments.js +16 -0
  82. package/node_modules/@pome-sh/twin-stripe/dist/src/check-payments.js.map +1 -1
  83. package/node_modules/@pome-sh/twin-stripe/dist/src/check-refunds.js +11 -3
  84. package/node_modules/@pome-sh/twin-stripe/dist/src/check-refunds.js.map +1 -1
  85. package/node_modules/@pome-sh/twin-stripe/dist/src/check-state.d.ts +25 -1
  86. package/node_modules/@pome-sh/twin-stripe/dist/src/check-state.js +31 -3
  87. package/node_modules/@pome-sh/twin-stripe/dist/src/check-state.js.map +1 -1
  88. package/node_modules/@pome-sh/twin-stripe/package.json +3 -3
  89. package/package.json +8 -8
package/README.md CHANGED
@@ -64,22 +64,30 @@ code is the verdict**. Gate CI on it directly.
64
64
  | Exit code | Meaning |
65
65
  | --- | --- |
66
66
  | `0` | pass (hosted/scored run), or trace captured (`--local`, not scored) |
67
- | `1` | ran and scored **below** the pass threshold |
67
+ | `1` | ran and scored **below** the pass threshold, **or ran `INCOMPLETE`** |
68
68
  | `2` | twin / orchestration error (network, 5xx, twin spawn failed) |
69
69
  | `3` | auth error (401/403) — `pome login` again, or set `POME_API_KEY` in CI |
70
70
  | `4` | quota exceeded (402/429) |
71
71
  | `5` | usage error (bad flags, missing task file) |
72
72
 
73
- Two rules CI must honor:
73
+ Three rules CI must honor:
74
74
 
75
75
  - **`--local` is not a verdict.** A `--local` run captures a raw trace and never
76
76
  scores, so its exit `0` means "trace captured," not "passed." Never gate CI on
77
77
  a `--local` exit code — score it later with `pome eval <run-dir>`.
78
+ - **`INCOMPLETE` shares exit `1`, and it is not the agent's failure.** A run
79
+ whose criteria could not all be graded exits `1` rather than mapping its
80
+ partial score to a code — a run whose checks never ran is not a green CI
81
+ signal. The cost is stated rather than hidden: **`1` cannot tell "the agent
82
+ regressed" from "we could not grade it."** Read the verdict word printed
83
+ beside the score (`INCOMPLETE` vs a sub-threshold number) to separate them.
78
84
  - **Trial groups map as a whole.** `pome run -n k` (k>1) collapses the whole
79
85
  group to one code: `0` = at least one trial completed and every completed
80
- trial passed; `1` = at least one completed trial failed its threshold; `2` =
81
- no trial completed. Errored trials are excluded from the verdict fraction and
82
- never drag a passing group below `0` on their own.
86
+ trial passed; `1` = at least one completed trial failed its threshold **or was
87
+ incomplete**; `2` = no trial completed. Errored and incomplete trials are
88
+ excluded from the verdict fraction (`3 of 4 passed · 1 incomplete`) so neither
89
+ is counted as a pass nor charged to the agent as a loss — but a group holding
90
+ one cannot exit `0`.
83
91
 
84
92
  ## Development
85
93
 
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "package": "pome-sh",
3
- "version": "0.17.0",
4
- "git_sha": "46758262874f65435e8bfcff58f3881be4f731ca",
5
- "build_time": "2026-08-03T08:56:27.021Z"
3
+ "version": "0.19.0",
4
+ "git_sha": "f6fc19d8c27d039a35859dd942d4f0783671b695",
5
+ "build_time": "2026-08-04T00:45:27.607Z"
6
6
  }
@@ -127,7 +127,7 @@ function handleConnect({ req, clientSocket, head, writer, egressWriter, allowHos
127
127
  const row = {
128
128
  ts: new Date().toISOString(),
129
129
  event_id: randomUUID(),
130
- parent_id: null,
130
+ parent_event_id: null,
131
131
  kind: "LlmCallEvent",
132
132
  host,
133
133
  port,
@@ -296,7 +296,7 @@ export async function runEval(options) {
296
296
  return client.finalize(sid, {
297
297
  stopReason: "eval_upload",
298
298
  // /finalize's schema requires an integer exit_code, so `null` (agent
299
- // timed out, or meta.json lacked the field) cannot pass through.
299
+ // timed out, or meta.json lacked the field) is not a legal value.
300
300
  // Send -1 as the explicit "unknown" sentinel — never a fabricated 0,
301
301
  // which would report a clean agent exit the trace can't vouch for.
302
302
  exitCode: artifacts.meta.exitCode ?? -1,
@@ -340,14 +340,16 @@ export async function runEval(options) {
340
340
  // never persisted next to the trace. Local artifacts stay trace-only (no
341
341
  // score.json), and the verdict lives in the cloud (see the dashboard URL).
342
342
  const score = scoreFromFinalizeResponse(finalized);
343
- // Exit-code policy — DELIBERATE DIVERGENCE from hosted `pome run`
344
- // (FDRS-618): `pome run` maps the raw cloud score (score >= threshold →
345
- // 0), because pre-FDRS-618 cloud builds don't emit criteria_results and
346
- // the cloud exit decision is documented as score-only. `pome eval` is a
347
- // NEW command with no such compatibility surface, so it adopts the full
348
- // FDRS-591/611 A5 guard up front: exit 0 ONLY when the run was evaluated,
349
- // every criterion was judged (can_pass), AND the score clears the
350
- // threshold. An UNEVAL verdict (e.g. all criteria skipped) exits 1.
343
+ // Exit-code policy — the full FDRS-591/611 A5 guard: exit 0 ONLY when the run
344
+ // was evaluated, every criterion was judged (can_pass), AND the score clears
345
+ // the threshold. An INCOMPLETE verdict (any criterion not evaluated) exits 1.
346
+ //
347
+ // F-925 retired the divergence that used to be documented here. `pome run`
348
+ // mapped the raw cloud score because "pre-FDRS-618 cloud builds don't emit
349
+ // criteria_results" but `scoreFromFinalizeResponse` already handles that
350
+ // case (`hasCriteriaResults ? : true`), so the guard degrades to score-only
351
+ // for exactly those builds on its own. The divergence was protecting a case
352
+ // its own helper already protected, and the two commands now agree.
351
353
  const exitCode = scoreStatus(score, EVAL_PASS_THRESHOLD) === "pass" ? 0 : 1;
352
354
  return {
353
355
  taskName,
@@ -403,7 +405,7 @@ export async function runEvalCommand(runDirArg, opts) {
403
405
  });
404
406
  // Same verdict shape as hosted `pome run`: LABEL, score line, cloud URL.
405
407
  const status = scoreStatus(result.score, EVAL_PASS_THRESHOLD);
406
- const label = status === "pass" ? "PASS" : status === "fail" ? "FAIL" : "UNEVAL";
408
+ const label = status === "pass" ? "PASS" : status === "fail" ? "FAIL" : "INCOMPLETE";
407
409
  console.error(`${label} ${result.taskName}`);
408
410
  console.error(` ${runScoreLine(result.score, EVAL_PASS_THRESHOLD, "cloud score")}`);
409
411
  if (result.score.results.length > 0) {
@@ -660,7 +660,7 @@ export function createProgram() {
660
660
  agentVersion: options.agentVersion,
661
661
  });
662
662
  const status = scoreStatus(result.score, result.scenario.config.passThreshold);
663
- const label = status === "pass" ? "PASS" : status === "fail" ? "FAIL" : "UNEVAL";
663
+ const label = status === "pass" ? "PASS" : status === "fail" ? "FAIL" : "INCOMPLETE";
664
664
  console.error(`${label} ${result.scenario.title}`);
665
665
  console.error(` ${runScoreLine(result.score, result.scenario.config.passThreshold, "cloud score")}`);
666
666
  console.error(` local: ${result.artifacts.runDir}`);
@@ -29,7 +29,7 @@ export type Score = {
29
29
  judge_tokens_out: number | null;
30
30
  };
31
31
  export declare function outcomeOf(result: CriterionResult): CriterionOutcome;
32
- export type ScoreStatus = "pass" | "fail" | "unevaluated";
32
+ export type ScoreStatus = "pass" | "fail" | "incomplete";
33
33
  export declare function scoreStatus(score: Score, passThreshold: number): ScoreStatus;
34
34
  export declare function taskPassed(score: Score, passThreshold: number): boolean;
35
35
  export declare function markerFor(outcome: CriterionOutcome): string;
@@ -31,9 +31,20 @@ export function outcomeOf(result) {
31
31
  // Encodes the A5 guard: a run is only a PASS when it was evaluated, every
32
32
  // required criterion was evaluated (can_pass), AND satisfaction cleared the
33
33
  // threshold. PURE — no computation of the score itself.
34
+ //
35
+ // F-932 renamed the third state from `unevaluated` to `incomplete` and CHANGED
36
+ // NOTHING ELSE HERE. The guard is the one place the CLI refuses to inflate a
37
+ // partial run into a pass — the same refusal pome-cloud added server-side in
38
+ // F-925 — so the rename must not become a loosening.
39
+ //
40
+ // One rule, two repos: `can_pass` is false for ANY abstention
41
+ // (`uploadAndFinalize.ts`), and pome-cloud's `isRunIncomplete` says
42
+ // `notEvaluated > 0` over the same `criteria_results`. Deliberately NOT read
43
+ // from the wire's `all_skipped`, which is the narrower every-abstained
44
+ // predicate and would loosen this guard.
34
45
  export function scoreStatus(score, passThreshold) {
35
46
  if (!score.evaluated || !score.can_pass)
36
- return "unevaluated";
47
+ return "incomplete";
37
48
  return score.satisfaction >= passThreshold ? "pass" : "fail";
38
49
  }
39
50
  export function taskPassed(score, passThreshold) {
@@ -55,14 +66,14 @@ export function markerFor(outcome) {
55
66
  // Multi-twin (M3): the per-criterion bracket for terminal display —
56
67
  // `[code]` / `[model]`, plus the `:<twin>` suffix when the criterion attributes
57
68
  // to a specific twin (so a `[code:slack]`/`[model:github]` marker survives into the
58
- // UNEVAL / criteria list). A bare (primary-twin) criterion renders `[code]`
69
+ // INCOMPLETE / criteria list). A bare (primary-twin) criterion renders `[code]`
59
70
  // unchanged.
60
71
  export function criterionMarkerLabel(criterion) {
61
72
  return criterion.twin ? `[${criterion.type}:${criterion.twin}]` : `[${criterion.type}]`;
62
73
  }
63
74
  // Multi-twin (M3): when the cloud could not evaluate a criterion for a
64
75
  // twin-related reason (a twin-tagged criterion, or a `no_matching_predicate`
65
- // skip), name the twin inline so the UNEVAL line explains WHICH twin's timeline
76
+ // skip), name the twin inline so the INCOMPLETE line explains WHICH twin's timeline
66
77
  // came up empty. Returns "" when there's nothing twin-specific to add.
67
78
  export function twinSkipSuffix(result) {
68
79
  const twin = result.criterion.twin;
@@ -79,8 +90,14 @@ export function scoreCountsSummary(score) {
79
90
  }
80
91
  export function runScoreLine(score, passThreshold, unevaluatedNumericLabel) {
81
92
  const status = scoreStatus(score, passThreshold);
82
- if (status === "unevaluated") {
83
- return `score: un-evaluated (cannot pass) ${scoreCountsSummary(score)}; ${unevaluatedNumericLabel}: ${score.satisfaction}/100`;
93
+ if (status === "incomplete") {
94
+ // Leads with the COUNT, which is the fact the reader needs and the same
95
+ // fact the cloud's own header now states. The old copy said "cannot pass",
96
+ // which is a verdict about the AGENT for a gap in the GRADER — the exact
97
+ // inversion F-925 exists to stop, one surface over.
98
+ const notEvaluated = score.skipped + score.errored;
99
+ const total = score.total_required + notEvaluated;
100
+ return `score: incomplete — ${notEvaluated} of ${total} criteria not evaluated; ${scoreCountsSummary(score)}; ${unevaluatedNumericLabel}: ${score.satisfaction}/100`;
84
101
  }
85
102
  return `score: ${score.satisfaction}/100`;
86
103
  }
@@ -1,9 +1,19 @@
1
+ import type { ScoreStatus } from "../hosted/evalResultView.js";
1
2
  export type TrialRow = {
2
3
  kind: "completed";
3
4
  /** Cloud-authoritative satisfaction score, 0-100. */
4
5
  score: number;
5
- /** Cleared the scenario's pass threshold. */
6
- passed: boolean;
6
+ /**
7
+ * F-925 — three states, not a boolean. `incomplete` means the trial ran
8
+ * and finalized but at least one criterion never produced a verdict, so
9
+ * it is neither a pass nor the agent's failure. It was `passed: boolean`
10
+ * fed from `exitCode === 0`, which counted a 100/100 run with 3 of 4
11
+ * criteria skipped as a clean passing trial.
12
+ *
13
+ * Typed as the CLI's own `ScoreStatus` rather than a second enum, so the
14
+ * trial line and the single-run headline cannot drift apart.
15
+ */
16
+ verdict: ScoreStatus;
7
17
  seconds: number;
8
18
  /** Failing-criteria summary ("a · b"), absent when none were reported. */
9
19
  note?: string;
@@ -45,20 +45,40 @@ export function trialRowLine(n, row) {
45
45
  if (row.kind === "errored") {
46
46
  return `trial ${n} ⚠ ${"errored".padEnd(16)}${row.reason} — excluded`;
47
47
  }
48
- const mark = row.passed ? "✓" : "✗";
48
+ // A dash for the ungradable trial: it ran, and it asserts nothing. Reusing
49
+ // ✗ would make a grader gap look like the agent's failure at a glance, which
50
+ // is the whole reading F-925 removes.
51
+ const mark = row.verdict === "pass" ? "✓" : row.verdict === "incomplete" ? "–" : "✗";
49
52
  const base = `trial ${n} ${mark} ${String(row.score).padEnd(9)}${row.seconds.toFixed(1)}s`;
50
53
  return row.note ? `${base} ${row.note}` : base;
51
54
  }
52
55
  export function groupSummaryLines(input) {
53
56
  const completed = input.rows.filter((r) => r.kind === "completed");
54
- const passed = completed.filter((r) => r.passed).length;
57
+ const passed = completed.filter((r) => r.verdict === "pass").length;
58
+ const incomplete = completed.filter((r) => r.verdict === "incomplete").length;
59
+ // F-925 — the fraction's denominator is the GRADED trials. A trial that
60
+ // finalized but could not be fully graded leaves both the numerator and the
61
+ // denominator, so a 5-trial set with one of them reads "3 of 4", never the
62
+ // "4 of 5" that counted it as a pass.
63
+ const graded = completed.length - incomplete;
55
64
  const errored = input.rows.length - completed.length;
56
65
  const lines = ["─────"];
57
- // The fraction counts COMPLETED trials only; errored trials are named and
58
- // excluded, never silently folded into the denominator.
59
- let fraction = completed.length === 0
60
- ? "no trials completed"
61
- : `${passed} of ${completed.length} passed`;
66
+ // The fraction counts GRADED trials only; incomplete and errored trials are
67
+ // named and excluded, never silently folded into the denominator. They stay
68
+ // two clauses rather than one: "the trial died" is ours to retry, "the trial
69
+ // ran and could not be graded" is a grader gap, and a reader told the wrong
70
+ // one goes looking in the wrong place.
71
+ let fraction;
72
+ if (graded === 0) {
73
+ fraction =
74
+ completed.length === 0 ? "no trials completed" : "no trials could be graded";
75
+ }
76
+ else {
77
+ fraction = `${passed} of ${graded} passed`;
78
+ }
79
+ if (incomplete > 0) {
80
+ fraction += ` · ${incomplete} incomplete, excluded from the fraction`;
81
+ }
62
82
  if (errored > 0) {
63
83
  fraction += ` · ${errored} errored, excluded from the fraction`;
64
84
  }
@@ -85,7 +105,11 @@ export function groupExitCode(rows) {
85
105
  const completed = rows.filter((r) => r.kind === "completed");
86
106
  if (completed.length === 0)
87
107
  return 2;
88
- return completed.every((r) => r.passed) ? 0 : 1;
108
+ // F-925 an ungradable trial is not a pass, so a group holding one cannot
109
+ // exit 0: green here would tell CI the set was verified when part of it was
110
+ // never checked. It stays 1 rather than 2 even when EVERY trial was
111
+ // incomplete — `2` means nothing completed, and these completed.
112
+ return completed.every((r) => r.verdict === "pass") ? 0 : 1;
89
113
  }
90
114
  /** Modal failed-criterion text across the group's completed trials, as the
91
115
  * short phrase the "start there" line renders. */
@@ -1,3 +1,26 @@
1
+ import { type Event } from "../types/shared.js";
2
+ /**
3
+ * F-1200 — give every twin HTTP row the tool call that caused it.
4
+ *
5
+ * The twin writes `parent_event_id: null` because it runs in its own process
6
+ * and cannot know the agent-side `event_id`; the adapter knows the `event_id`
7
+ * but never sees the twin's tape. This merge is the one place both halves are
8
+ * in hand, so the join belongs here — and it is a pure data operation over two
9
+ * finished files, with no dependency on the order the two writers ran in.
10
+ *
11
+ * The join key is the SDK's `tool_use_id`, which since F-1200 is what the
12
+ * adapter stamps on `x-pome-correlation-id`. The twin persists that header as
13
+ * `correlation_id` ALWAYS and as `tool_call_id` only when it pins
14
+ * `stampToolCallId` (github's frozen tape shape) — so `correlation_id` is the
15
+ * key that works for every twin, with `tool_call_id` preferred when present
16
+ * because it is unambiguously the header rather than the request-id fallback.
17
+ *
18
+ * A row keeps its null parent when the id names no tool call: a pre-F-1200
19
+ * `tlc_…`, a `req_…` from the no-header fallback, or a direct REST call made
20
+ * outside any tool handler. An unresolvable parent is the honest answer there —
21
+ * inventing one would put twin calls under tools that did not make them.
22
+ */
23
+ export declare function resolveTwinHttpParents(rows: Event[]): Event[];
1
24
  export declare function mergeAdapterSignalsIntoEvents(signalsPath: string, eventsJsonlPath: string): Promise<{
2
25
  appended: number;
3
26
  dropped: number;
@@ -24,6 +24,48 @@ import { redactEvent } from "../recorder/redaction.js";
24
24
  // lines are dropped and counted; the caller can log the drop count. Existing
25
25
  // events.jsonl rows that fail to parse are passed through unsorted at the
26
26
  // head of the file so a corrupted in-flight write is never silently dropped.
27
+ /**
28
+ * F-1200 — give every twin HTTP row the tool call that caused it.
29
+ *
30
+ * The twin writes `parent_event_id: null` because it runs in its own process
31
+ * and cannot know the agent-side `event_id`; the adapter knows the `event_id`
32
+ * but never sees the twin's tape. This merge is the one place both halves are
33
+ * in hand, so the join belongs here — and it is a pure data operation over two
34
+ * finished files, with no dependency on the order the two writers ran in.
35
+ *
36
+ * The join key is the SDK's `tool_use_id`, which since F-1200 is what the
37
+ * adapter stamps on `x-pome-correlation-id`. The twin persists that header as
38
+ * `correlation_id` ALWAYS and as `tool_call_id` only when it pins
39
+ * `stampToolCallId` (github's frozen tape shape) — so `correlation_id` is the
40
+ * key that works for every twin, with `tool_call_id` preferred when present
41
+ * because it is unambiguously the header rather than the request-id fallback.
42
+ *
43
+ * A row keeps its null parent when the id names no tool call: a pre-F-1200
44
+ * `tlc_…`, a `req_…` from the no-header fallback, or a direct REST call made
45
+ * outside any tool handler. An unresolvable parent is the honest answer there —
46
+ * inventing one would put twin calls under tools that did not make them.
47
+ */
48
+ export function resolveTwinHttpParents(rows) {
49
+ const eventIdByToolUseId = new Map();
50
+ for (const row of rows) {
51
+ if (row.kind === "ToolUseEvent")
52
+ eventIdByToolUseId.set(row.tool_use_id, row.event_id);
53
+ }
54
+ if (eventIdByToolUseId.size === 0)
55
+ return rows;
56
+ return rows.map((row) => {
57
+ if (row.kind !== "TwinHttpEvent")
58
+ return row;
59
+ // Never overwrite a parent the writer already established.
60
+ if (row.parent_event_id != null)
61
+ return row;
62
+ const causingToolUseId = row.tool_call_id ?? row.correlation_id ?? null;
63
+ if (causingToolUseId === null)
64
+ return row;
65
+ const parentEventId = eventIdByToolUseId.get(causingToolUseId);
66
+ return parentEventId === undefined ? row : { ...row, parent_event_id: parentEventId };
67
+ });
68
+ }
27
69
  export async function mergeAdapterSignalsIntoEvents(signalsPath, eventsJsonlPath) {
28
70
  let rawSignals;
29
71
  try {
@@ -84,9 +126,9 @@ export async function mergeAdapterSignalsIntoEvents(signalsPath, eventsJsonlPath
84
126
  unparseablePassthrough.push(line);
85
127
  }
86
128
  }
87
- // Concat + stable sort by ts. ISO-8601 with `Z` sorts chronologically under
88
- // lexicographic compare.
89
- const merged = eventRows.concat(signalRows);
129
+ // Concat, resolve twin-HTTP parentage, then stable sort by ts. ISO-8601 with
130
+ // `Z` sorts chronologically under lexicographic compare.
131
+ const merged = resolveTwinHttpParents(eventRows.concat(signalRows));
90
132
  merged.sort((a, b) => (a.ts < b.ts ? -1 : a.ts > b.ts ? 1 : 0));
91
133
  const sortedJsonl = merged.map((r) => JSON.stringify(r)).join("\n");
92
134
  const head = unparseablePassthrough.length > 0 ? unparseablePassthrough.join("\n") + "\n" : "";
@@ -1,7 +1,7 @@
1
1
  import { type HostedClient } from "../hosted/client.js";
2
2
  import type { CreateSessionResponse } from "../types/shared.js";
3
3
  import type { Task } from "../task/taskSchema.js";
4
- import type { Score } from "../hosted/evalResultView.js";
4
+ import { type Score, type ScoreStatus } from "../hosted/evalResultView.js";
5
5
  import type { RunArtifacts } from "../recorder/artifacts.js";
6
6
  export interface RunTaskHostedOptions {
7
7
  taskPath: string;
@@ -44,6 +44,13 @@ export interface RunTaskHostedResult {
44
44
  cloudDashboardUrl: string;
45
45
  artifacts: RunArtifacts;
46
46
  score: Score;
47
+ /**
48
+ * F-925 — the run's own three-state verdict, computed ONCE here and carried
49
+ * out so a caller never re-derives it. `exitCode` cannot express it: 1 means
50
+ * both "failed" and "could not be graded", and a trial group needs to tell
51
+ * them apart to keep the ungradable one out of its fraction.
52
+ */
53
+ verdict: ScoreStatus;
47
54
  exitCode: number;
48
55
  /** Wall time from run start to post-agent state capture — the same value
49
56
  * reported to /finalize as duration_ms. FDRS-636 renders it as the trial
@@ -13,6 +13,7 @@ import { ensureMcpSuffix } from "../cli/session.js";
13
13
  import { redactJsonl, scoreFromFinalizeResponse, uploadRunBlobs, } from "../hosted/uploadAndFinalize.js";
14
14
  import { resolveRunAgentIdentity } from "../cli/agent-identity.js";
15
15
  import { HostedOrchError, HostedTrialError } from "../hosted/errors.js";
16
+ import { scoreStatus, } from "../hosted/evalResultView.js";
16
17
  /** Build the agent subprocess env for a hosted run.
17
18
  *
18
19
  * Two layers, in this order:
@@ -491,7 +492,15 @@ export async function runTaskHosted(options) {
491
492
  // Pre-finalize agent failures (auth, quota, twin spawn, exec
492
493
  // errors) take other code paths via thrown HostedAuthError /
493
494
  // HostedQuotaError / HostedOrchError and never reach this line.
494
- const exitCode = finalized.score >= scenario.config.passThreshold ? 0 : 1;
495
+ // F-925 the score alone is not the verdict. A 100/100 run with a
496
+ // criterion that never ran is `incomplete`, and exiting 0 on it is how
497
+ // "the fix passed" came to mean "the check never ran". `scoreStatus`
498
+ // applies the A5 guard `pome eval` has always applied; the FDRS-618
499
+ // compat this used to preserve is already preserved inside
500
+ // `scoreFromFinalizeResponse`, which sets can_pass true when the
501
+ // response carries no criteria_results at all.
502
+ const verdict = scoreStatus(score, scenario.config.passThreshold);
503
+ const exitCode = verdict === "pass" ? 0 : 1;
495
504
  // 11. FDRS-644 — cache the CLOUD verdict payload next to the raw trace
496
505
  // (verdict.json, provenance-labeled `source: "cloud-finalize"`).
497
506
  // Not a local score: evaluation stayed in the cloud; this records
@@ -528,6 +537,7 @@ export async function runTaskHosted(options) {
528
537
  cloudDashboardUrl: finalized.dashboard_url,
529
538
  artifacts,
530
539
  score,
540
+ verdict,
531
541
  exitCode,
532
542
  durationMs,
533
543
  };
@@ -179,7 +179,14 @@ export async function runTrialGroup(options) {
179
179
  row = {
180
180
  kind: "completed",
181
181
  score: result.score.satisfaction,
182
- passed: result.exitCode === 0,
182
+ // F-925 the trial's own verdict, carried out of the run rather than
183
+ // re-derived here. It was `result.exitCode === 0`, which cannot express
184
+ // the third state (1 means both "failed" and "could not be graded"), so
185
+ // a 100/100 trial with 3 of 4 criteria skipped counted as a clean pass.
186
+ // Reusing the run's own value rather than calling `scoreStatus` again
187
+ // on the same inputs is what keeps the trial line and the single-run
188
+ // headline from ever disagreeing.
189
+ verdict: result.verdict,
183
190
  seconds: result.durationMs / 1000,
184
191
  note: failing.length > 0
185
192
  ? failing.map((r) => criterionPhrase(r.criterion.text)).join(" · ")
@@ -235,7 +242,13 @@ export async function runTrialGroup(options) {
235
242
  // 4. FDRS-644 — the fix & green handoff, only when a COMPLETED trial
236
243
  // failed. Errored trials are sandbox noise: the answer there is re-run,
237
244
  // not a code fix, so an errored-only group gets no handoff.
238
- const failedCompleted = rows.filter((r) => r.kind === "completed" && !r.passed).length;
245
+ //
246
+ // F-925 — `fail`, NOT "anything that isn't a pass". This was `!r.passed`,
247
+ // which now includes the incomplete trial, and pointing someone at
248
+ // `pome fix-prompt` for a criterion that never ran tells them to fix an agent
249
+ // that may be blameless. An abstention is a grader gap; the handoff is for
250
+ // agent defects.
251
+ const failedCompleted = rows.filter((r) => r.kind === "completed" && r.verdict === "fail").length;
239
252
  if (failedCompleted > 0) {
240
253
  const artifactsDir = options.artifactsDir ?? "runs";
241
254
  const fixPromptCommand = artifactsDir === "runs"
@@ -116,7 +116,7 @@ function emitToolUse(signalsPath, row) {
116
116
  appendFileSync(signalsPath, JSON.stringify({
117
117
  ts: new Date().toISOString(),
118
118
  event_id: randomUUID(),
119
- parent_id: null,
119
+ parent_event_id: null,
120
120
  kind: "ToolUseEvent",
121
121
  tool_use_id: row.tool_use_id,
122
122
  tool_name: row.tool_name,
@@ -134,7 +134,7 @@ function emitToolResult(signalsPath, row) {
134
134
  appendFileSync(signalsPath, JSON.stringify({
135
135
  ts: new Date().toISOString(),
136
136
  event_id: randomUUID(),
137
- parent_id: null,
137
+ parent_event_id: null,
138
138
  kind: "ToolResultEvent",
139
139
  tool_use_id: row.tool_use_id,
140
140
  output: row.output,
@@ -163,7 +163,7 @@ function emitLlmCall(signalsPath, row) {
163
163
  appendFileSync(signalsPath, JSON.stringify({
164
164
  ts: new Date().toISOString(),
165
165
  event_id: randomUUID(),
166
- parent_id: null,
166
+ parent_event_id: null,
167
167
  kind: "LlmCallEvent",
168
168
  host: row.host,
169
169
  port: 443,
@@ -0,0 +1,71 @@
1
+ import type { CheckDefinition } from "./checks.js";
2
+ /**
3
+ * A pointer from the ROOT of the exported tree.
4
+ *
5
+ * Numbers are array indices and strings are object keys, but this function does
6
+ * not enforce that — the tree decides, at resolution time. Passing no segments
7
+ * returns `""`, which RFC 6901 defines as the whole document; a check with
8
+ * nothing narrower to name must OMIT `evidenceStatePaths` rather than cite the
9
+ * root, for the same reason it must omit rather than send `[]`.
10
+ */
11
+ export declare function statePath(...segments: readonly (string | number)[]): string;
12
+ /**
13
+ * A pointer relative to one a resolver already built.
14
+ *
15
+ * The twins' `resolve*` helpers walk the tree to find an entity and now hand
16
+ * back the pointer they walked; a check appends the field it went on to read.
17
+ * Written as its own function rather than string concatenation at 37 call sites
18
+ * because the escaping has to happen on the appended segments and NOT on the
19
+ * base, which already contains real `/` separators — concatenating by hand is
20
+ * how a `~1` ends up double-encoded.
21
+ */
22
+ export declare function childStatePath(base: string, ...segments: readonly (string | number)[]): string;
23
+ /**
24
+ * What a pointer addresses in a tree, or `null` when it addresses nothing.
25
+ *
26
+ * `null` is a NORMAL answer and every caller must handle it: the state blob a
27
+ * report renders can be a different export from the one the check read (a
28
+ * re-run, a truncated upload, a snapshot predating a field). A consumer turns
29
+ * `null` into "no affordance", exactly as `findEventIndexForEvidence` turns an
30
+ * unresolvable event id into one.
31
+ *
32
+ * Wrapped in `{ value }` rather than returned bare because `undefined` and
33
+ * `null` are both legal JSON-ish values a tree can hold at a pointer, and a bare
34
+ * return could not tell "the path is absent" from "the path holds null" — the
35
+ * difference between no evidence and evidence that the field is empty.
36
+ */
37
+ export declare function resolveStatePath(tree: unknown, pointer: string): {
38
+ value: unknown;
39
+ } | null;
40
+ export type StateCitationArm = "passing" | "failing";
41
+ export type StateCitationVerdict = {
42
+ kind: "cites";
43
+ } | {
44
+ kind: "declined";
45
+ } | {
46
+ kind: "uncited";
47
+ arm: StateCitationArm;
48
+ } | {
49
+ kind: "unresolvable";
50
+ arm: StateCitationArm;
51
+ pointer: string;
52
+ } | {
53
+ kind: "malformed";
54
+ arm: StateCitationArm;
55
+ detail: string;
56
+ };
57
+ /**
58
+ * Does this state-reading check say WHERE it looked, in a form that resolves?
59
+ *
60
+ * Probes BOTH arms, and that is deliberate. A citation present on the passing
61
+ * world and absent on the failing one is worse than no citation at all: its
62
+ * absence starts reading as a verdict class — "no evidence" would come to mean
63
+ * "this one failed" — and the reader has no way to know that is an accident of
64
+ * how the predicate was written. So a check must be able to say where it looked
65
+ * whether or not it liked what it found there.
66
+ *
67
+ * Only meaningful for `final` / `seed+final` checks. A `tape` check cites
68
+ * `evidenceEventIds` instead and is not this gate's business; callers filter on
69
+ * `substrate` before probing.
70
+ */
71
+ export declare function probeStateCitation<TState, TArgs extends Record<string, string>>(def: CheckDefinition<TState, TArgs>, args: TArgs): StateCitationVerdict;