@dzhechkov/harness-core 0.8.46 → 0.8.48

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (153) hide show
  1. package/.dz-manifest.json +200 -168
  2. package/README.md +130 -2
  3. package/dist/agentdb-index.d.ts.map +1 -1
  4. package/dist/agentdb-index.js +32 -2
  5. package/dist/agentdb-index.js.map +1 -1
  6. package/dist/agentdb-snapshot.d.ts +14 -1
  7. package/dist/agentdb-snapshot.d.ts.map +1 -1
  8. package/dist/agentdb-snapshot.js +100 -4
  9. package/dist/agentdb-snapshot.js.map +1 -1
  10. package/dist/apply-leg.d.ts +9 -1
  11. package/dist/apply-leg.d.ts.map +1 -1
  12. package/dist/apply-leg.js +26 -4
  13. package/dist/apply-leg.js.map +1 -1
  14. package/dist/architecture.d.ts +0 -5
  15. package/dist/architecture.d.ts.map +1 -1
  16. package/dist/architecture.js +25 -2
  17. package/dist/architecture.js.map +1 -1
  18. package/dist/codex-rollouts.d.ts +35 -7
  19. package/dist/codex-rollouts.d.ts.map +1 -1
  20. package/dist/codex-rollouts.js +201 -113
  21. package/dist/codex-rollouts.js.map +1 -1
  22. package/dist/cost-ledger.d.ts +42 -3
  23. package/dist/cost-ledger.d.ts.map +1 -1
  24. package/dist/cost-ledger.js +478 -15
  25. package/dist/cost-ledger.js.map +1 -1
  26. package/dist/feature-adr-routing.d.ts +16 -0
  27. package/dist/feature-adr-routing.d.ts.map +1 -1
  28. package/dist/feature-adr-routing.js +21 -0
  29. package/dist/feature-adr-routing.js.map +1 -1
  30. package/dist/guard.d.ts +7 -0
  31. package/dist/guard.d.ts.map +1 -1
  32. package/dist/guard.js +48 -0
  33. package/dist/guard.js.map +1 -1
  34. package/dist/index.d.ts +11 -6
  35. package/dist/index.d.ts.map +1 -1
  36. package/dist/index.js +10 -3
  37. package/dist/index.js.map +1 -1
  38. package/dist/loop-plan.d.ts +4 -0
  39. package/dist/loop-plan.d.ts.map +1 -1
  40. package/dist/loop-plan.js +4 -0
  41. package/dist/loop-plan.js.map +1 -1
  42. package/dist/mutation-gate.d.ts +36 -1
  43. package/dist/mutation-gate.d.ts.map +1 -1
  44. package/dist/mutation-gate.js +109 -19
  45. package/dist/mutation-gate.js.map +1 -1
  46. package/dist/npm-homepage.d.ts +64 -0
  47. package/dist/npm-homepage.d.ts.map +1 -0
  48. package/dist/npm-homepage.js +109 -0
  49. package/dist/npm-homepage.js.map +1 -0
  50. package/dist/operations.d.ts +6 -0
  51. package/dist/operations.d.ts.map +1 -1
  52. package/dist/operations.js +150 -24
  53. package/dist/operations.js.map +1 -1
  54. package/dist/pack-inventory.d.ts.map +1 -1
  55. package/dist/pack-inventory.js +6 -1
  56. package/dist/pack-inventory.js.map +1 -1
  57. package/dist/parity.d.ts.map +1 -1
  58. package/dist/parity.js +4 -1
  59. package/dist/parity.js.map +1 -1
  60. package/dist/publish-sibling-drift.d.ts +43 -1
  61. package/dist/publish-sibling-drift.d.ts.map +1 -1
  62. package/dist/publish-sibling-drift.js +269 -13
  63. package/dist/publish-sibling-drift.js.map +1 -1
  64. package/dist/publish.d.ts +5 -1
  65. package/dist/publish.d.ts.map +1 -1
  66. package/dist/publish.js +18 -1
  67. package/dist/publish.js.map +1 -1
  68. package/dist/qe-bridge.d.ts +63 -1
  69. package/dist/qe-bridge.d.ts.map +1 -1
  70. package/dist/qe-bridge.js +82 -0
  71. package/dist/qe-bridge.js.map +1 -1
  72. package/dist/release-package-audit.d.ts +86 -0
  73. package/dist/release-package-audit.d.ts.map +1 -0
  74. package/dist/release-package-audit.js +272 -0
  75. package/dist/release-package-audit.js.map +1 -0
  76. package/dist/release.d.ts +8 -1
  77. package/dist/release.d.ts.map +1 -1
  78. package/dist/release.js +58 -32
  79. package/dist/release.js.map +1 -1
  80. package/dist/round.d.ts +10 -1
  81. package/dist/round.d.ts.map +1 -1
  82. package/dist/round.js +5 -1
  83. package/dist/round.js.map +1 -1
  84. package/dist/run-records.d.ts +11 -0
  85. package/dist/run-records.d.ts.map +1 -1
  86. package/dist/run-records.js +124 -20
  87. package/dist/run-records.js.map +1 -1
  88. package/dist/setup-memory-deps.d.ts +14 -0
  89. package/dist/setup-memory-deps.d.ts.map +1 -0
  90. package/dist/setup-memory-deps.js +210 -0
  91. package/dist/setup-memory-deps.js.map +1 -0
  92. package/dist/setup.d.ts +2 -0
  93. package/dist/setup.d.ts.map +1 -1
  94. package/dist/setup.js +379 -378
  95. package/dist/setup.js.map +1 -1
  96. package/dist/stage-usage.d.ts +248 -0
  97. package/dist/stage-usage.d.ts.map +1 -0
  98. package/dist/stage-usage.js +511 -0
  99. package/dist/stage-usage.js.map +1 -0
  100. package/dist/statusline.d.ts +23 -0
  101. package/dist/statusline.d.ts.map +1 -1
  102. package/dist/statusline.js +167 -1
  103. package/dist/statusline.js.map +1 -1
  104. package/dist/workflow-run-dispatch.d.ts +36 -19
  105. package/dist/workflow-run-dispatch.d.ts.map +1 -1
  106. package/dist/workflow-run-dispatch.js +255 -102
  107. package/dist/workflow-run-dispatch.js.map +1 -1
  108. package/dist/workflow-run.d.ts +19 -3
  109. package/dist/workflow-run.d.ts.map +1 -1
  110. package/dist/workflow-run.js +93 -2
  111. package/dist/workflow-run.js.map +1 -1
  112. package/package.json +3 -3
  113. package/sbom.json +303 -223
  114. package/src/agentdb-index.ts +31 -2
  115. package/src/agentdb-snapshot.ts +105 -4
  116. package/src/apply-leg.ts +26 -4
  117. package/src/architecture.ts +15 -2
  118. package/src/codex-rollouts.ts +155 -147
  119. package/src/cost-ledger.ts +308 -19
  120. package/src/feature-adr-routing.ts +22 -0
  121. package/src/guard.ts +52 -0
  122. package/src/index.ts +15 -3
  123. package/src/loop-plan.ts +8 -0
  124. package/src/mutation-gate.ts +153 -19
  125. package/src/npm-homepage.ts +142 -0
  126. package/src/operations.ts +177 -22
  127. package/src/pack-inventory.ts +6 -1
  128. package/src/parity.ts +4 -1
  129. package/src/publish-sibling-drift.ts +214 -11
  130. package/src/publish.ts +20 -1
  131. package/src/qe-bridge.ts +86 -1
  132. package/src/release-package-audit.ts +264 -0
  133. package/src/release.ts +55 -22
  134. package/src/round.ts +8 -0
  135. package/src/run-records.ts +101 -22
  136. package/src/setup-memory-deps.ts +166 -0
  137. package/src/setup.ts +95 -98
  138. package/src/stage-usage.ts +385 -0
  139. package/src/statusline.ts +149 -1
  140. package/src/workflow-run-dispatch.ts +197 -93
  141. package/src/workflow-run.ts +101 -5
  142. package/dist/ledger-cost-fill.d.ts +0 -58
  143. package/dist/ledger-cost-fill.d.ts.map +0 -1
  144. package/dist/ledger-cost-fill.js +0 -78
  145. package/dist/ledger-cost-fill.js.map +0 -1
  146. package/dist/retro.d.ts +0 -131
  147. package/dist/retro.d.ts.map +0 -1
  148. package/dist/retro.js +0 -207
  149. package/dist/retro.js.map +0 -1
  150. package/dist/sbom.d.ts +0 -42
  151. package/dist/sbom.d.ts.map +0 -1
  152. package/dist/sbom.js +0 -120
  153. package/dist/sbom.js.map +0 -1
@@ -60,7 +60,18 @@ export interface DispatchResult {
60
60
  /** null when the runtime did not report a count — NEVER 0, never estimated (ADR-004 C4). */
61
61
  tokensIn: number | null;
62
62
  tokensOut: number | null;
63
- tokensSource: 'claude-envelope' | 'codex-stderr' | null;
63
+ tokensSource: 'claude-envelope' | 'codex-stderr' | 'codex-json' | null;
64
+ tokensTotal?: number | null;
65
+ tokensCacheRead?: number | null;
66
+ tokensCacheWrite?: number | null;
67
+ tokensReasoning?: number | null;
68
+ reportedTotalBasis?: 'raw-inclusive' | 'uncached-display' | 'reported-unknown' | 'output-only' | 'unknown';
69
+ inputCacheSemantics?: 'includes-cache-read-write' | 'excludes-cache-read-write' | 'uncached-display' | 'unknown';
70
+ reportedCostUsd?: number | null;
71
+ usageDiagnostics?: string[];
72
+ totalDerivation?: string;
73
+ modelProvenance?: string;
74
+ usageSource?: { schema: string; scope: string; threadId: string | null; turnId: string | null; receiptId: string | null };
64
75
  failure?: DispatchFailure;
65
76
  }
66
77
 
@@ -69,6 +80,13 @@ export interface ProbeOutcome {
69
80
  id: string | null;
70
81
  wallMs: number;
71
82
  detail: string;
83
+ provenance?: {
84
+ schema: 'wf-probe-attempts-1'; complete: boolean; totalConsidered: number;
85
+ attempts: Array<{ ordinal: number; model: string | null; family: BridgeFamily; wrapperInvoked: boolean;
86
+ outcome: 'answered' | 'failed' | 'rejected'; selected: boolean;
87
+ reason: 'answered' | 'timeout' | 'spawn-error' | 'no-exit-code' | 'exit-nonzero' | 'unexpected-response' | 'invalid-candidate' }>;
88
+ } | null;
89
+ provenanceReason?: 'candidate-model-invalid' | 'wrapper-result-invalid' | null;
72
90
  }
73
91
 
74
92
  export interface Dispatcher {
@@ -181,7 +199,7 @@ export { CODEX_EXEC_PROMPT_CEILING_CHARS };
181
199
  export function codexExecArgv(modelId: string, prompt: string, deliverable: Deliverable): string[] {
182
200
  const returnValue = (deliverable ?? 'return-value') !== 'file';
183
201
  const text = returnValue ? CODEX_SCOPING_PREFIX + '\n\n' + prompt : prompt;
184
- return ['exec', '-m', modelId, ...(returnValue ? ['--sandbox', 'read-only'] : []), text];
202
+ return ['exec', '-m', modelId, ...(returnValue ? ['--sandbox', 'read-only'] : []), '--json', text];
185
203
  }
186
204
 
187
205
  /** The liveness probe: an allowlist says an id is SPELLABLE, only a probe says it ANSWERS. */
@@ -205,55 +223,160 @@ export function interpretCodexProbe(out: { stdout: string; exitCode: number | nu
205
223
  * builds/configs do print one; it is tested, not assumed. (Named consequence: codex runs report no
206
224
  * token counts today. That is the honest state, not a bug to paper over — see the manifest.)
207
225
  */
208
- export function extractCodexTokens(stderr: string): { tokensIn: number | null; tokensOut: number | null } {
209
- const text = String(stderr ?? '');
210
- const num = (raw: string | undefined): number | null => {
211
- if (raw === undefined) return null;
212
- const n = Number(raw.replace(/[,_\s]/g, ''));
213
- return Number.isFinite(n) && n >= 0 ? n : null;
226
+ type ObservedUsage = Pick<DispatchResult, 'tokensIn' | 'tokensOut' | 'tokensSource' | 'tokensTotal' | 'tokensCacheRead' | 'tokensCacheWrite' | 'tokensReasoning' | 'reportedTotalBasis' | 'inputCacheSemantics' | 'reportedCostUsd' | 'usageDiagnostics' | 'usageSource' | 'totalDerivation'>;
227
+ const count = (value: unknown): number | null => typeof value === 'number' && Number.isSafeInteger(value) && value >= 0 ? value : null;
228
+ const emptyUsage = (): ObservedUsage => ({ tokensIn: null, tokensOut: null, tokensTotal: null, tokensCacheRead: null,
229
+ tokensCacheWrite: null, tokensReasoning: null, tokensSource: null, reportedTotalBasis: 'unknown', inputCacheSemantics: 'unknown',
230
+ reportedCostUsd: null, usageDiagnostics: [] });
231
+
232
+ export function extractCodexTokens(stderr: string): ObservedUsage {
233
+ const text = String(stderr ?? ''); const usage = emptyUsage();
234
+ const numeric = (raw: string | undefined): number | null => raw === undefined ? null : count(Number(raw.replace(/[,_\s]/g, '')));
235
+ const input = /\binput\b[^\n\d+.-]{0,20}([+-]?\d[\d,_]*(?:\.\d+)?)|\btokens?\s+in\b[^\n\d+.-]{0,10}([+-]?\d[\d,_]*(?:\.\d+)?)/i.exec(text);
236
+ const output = /\boutput\b[^\n\d+.-]{0,20}([+-]?\d[\d,_]*(?:\.\d+)?)|\btokens?\s+out\b[^\n\d+.-]{0,10}([+-]?\d[\d,_]*(?:\.\d+)?)/i.exec(text);
237
+ const total = /tokens used\s*\n\s*([+-]?\d[\d,_]*(?:\.\d+)?)|Token usage:\s*total=([+-]?\d[\d,_]*(?:\.\d+)?)/i.exec(text);
238
+ usage.tokensIn = numeric(input?.[1] ?? input?.[2]); usage.tokensOut = numeric(output?.[1] ?? output?.[2]);
239
+ usage.tokensTotal = numeric(total?.[1] ?? total?.[2]);
240
+ if (input && usage.tokensIn === null) usage.usageDiagnostics!.push('invalid-counter:input');
241
+ if (output && usage.tokensOut === null) usage.usageDiagnostics!.push('invalid-counter:output');
242
+ if (total && usage.tokensTotal === null) usage.usageDiagnostics!.push('invalid-counter:total');
243
+ const cached = /\(\+([\d,_]+) cached\)/i.exec(text);
244
+ if (total?.[2] && cached) usage.tokensCacheRead = numeric(cached[1]);
245
+ usage.reportedTotalBasis = total?.[2] ? 'uncached-display' : 'reported-unknown';
246
+ usage.inputCacheSemantics = total?.[2] ? 'uncached-display' : 'unknown';
247
+ usage.tokensSource = usage.tokensTotal != null || usage.tokensIn != null || usage.tokensOut != null ? 'codex-stderr' : null;
248
+ usage.totalDerivation = usage.tokensTotal === null ? 'not-recorded' : 'reported';
249
+ return usage;
250
+ }
251
+
252
+ export function extractClaudeUsage(stdout: string): ObservedUsage {
253
+ let chosen: Record<string, unknown> | null = null;
254
+ for (const line of [String(stdout).trim(), ...String(stdout).split(/\r?\n/)]) {
255
+ try { const obj = JSON.parse(line) as Record<string, unknown>; if (obj && obj['type'] === 'result') chosen = obj; } catch { /* non-envelope line */ }
256
+ }
257
+ const result = emptyUsage(); if (!chosen) return result;
258
+ const usage = chosen['usage']; if (!usage || typeof usage !== 'object' || Array.isArray(usage)) return result;
259
+ const raw = usage as Record<string, unknown>;
260
+ const invalid = (obj: Record<string, unknown>, key: string) => obj[key] != null && count(obj[key]) === null;
261
+ const field = (obj: Record<string, unknown>, key: string): number | null => {
262
+ if (invalid(obj, key)) result.usageDiagnostics!.push('invalid-count:' + key);
263
+ return count(obj[key]);
214
264
  };
215
- // Two accepted spellings per half: `input tokens: N` / `input: N tokens` and `tokens in: N`.
216
- const inMatch = /\binput\b[^\n\d]{0,20}([\d,_]+)|\btokens?\s+in\b[^\n\d]{0,10}([\d,_]+)/i.exec(text);
217
- const outMatch = /\boutput\b[^\n\d]{0,20}([\d,_]+)|\btokens?\s+out\b[^\n\d]{0,10}([\d,_]+)/i.exec(text);
218
- const tokensIn = inMatch === null ? null : num(inMatch[1] ?? inMatch[2]);
219
- const tokensOut = outMatch === null ? null : num(outMatch[1] ?? outMatch[2]);
220
- return { tokensIn, tokensOut };
265
+ result.tokensIn = field(raw, 'input_tokens'); result.tokensOut = field(raw, 'output_tokens');
266
+ result.tokensCacheRead = field(raw, 'cache_read_input_tokens'); result.tokensCacheWrite = field(raw, 'cache_creation_input_tokens');
267
+ const creation = raw['cache_creation'];
268
+ if (creation != null) {
269
+ if (typeof creation !== 'object' || Array.isArray(creation)) {
270
+ result.usageDiagnostics!.push('invalid-cache-creation-shape'); result.tokensCacheWrite = null;
271
+ } else {
272
+ const nested = creation as Record<string, unknown>;
273
+ const a = field(nested, 'ephemeral_5m_input_tokens'); const b = field(nested, 'ephemeral_1h_input_tokens');
274
+ const nestedInvalid = invalid(nested, 'ephemeral_5m_input_tokens') || invalid(nested, 'ephemeral_1h_input_tokens');
275
+ const total = a !== null && b !== null ? count(a + b) : null;
276
+ if (a !== null && b !== null && total === null) result.usageDiagnostics!.push('cache-creation-overflow');
277
+ if (nestedInvalid || (a !== null && b !== null && total === null)) result.tokensCacheWrite = null;
278
+ else if (raw['cache_creation_input_tokens'] == null) result.tokensCacheWrite = total;
279
+ else if (result.tokensCacheWrite !== null && total !== null && total !== result.tokensCacheWrite) {
280
+ result.usageDiagnostics!.push('cache-creation-mismatch'); result.tokensCacheWrite = null;
281
+ }
282
+ }
283
+ }
284
+ result.tokensReasoning = field(raw, 'reasoning_output_tokens');
285
+ if (result.tokensReasoning !== null && result.tokensOut !== null && result.tokensReasoning > result.tokensOut) {
286
+ result.usageDiagnostics!.push('reasoning-exceeds-output'); result.tokensReasoning = null;
287
+ }
288
+ result.tokensTotal = field(raw, 'total_tokens');
289
+ const components = [result.tokensIn, result.tokensCacheRead, result.tokensCacheWrite, result.tokensOut];
290
+ const partitionKnown = components.every((n) => n !== null);
291
+ const derived = partitionKnown ? count(components.reduce<number>((n, v) => n + (v ?? 0), 0)) : null;
292
+ if (partitionKnown && derived === null) result.usageDiagnostics!.push('total-overflow');
293
+ if (invalid(raw, 'total_tokens')) result.totalDerivation = 'invalid-reported-total';
294
+ else if (raw['total_tokens'] == null) {
295
+ result.tokensTotal = derived; result.totalDerivation = derived === null ? 'not-recorded' : 'disjoint-dimensions';
296
+ } else result.totalDerivation = 'reported';
297
+ if (derived !== null && result.tokensTotal !== null && result.tokensTotal !== derived) result.usageDiagnostics!.push('total-split-mismatch');
298
+ result.reportedCostUsd = typeof chosen['total_cost_usd'] === 'number' && Number.isFinite(chosen['total_cost_usd']) && chosen['total_cost_usd'] >= 0 ? chosen['total_cost_usd'] : null;
299
+ result.reportedTotalBasis = 'raw-inclusive'; result.inputCacheSemantics = 'excludes-cache-read-write'; result.tokensSource = 'claude-envelope';
300
+ return result;
221
301
  }
222
302
 
223
- /**
224
- * The claude `--output-format json` USAGE fields, pinned to the LIVE envelope shape.
225
- *
226
- * MEASURED this session (`claude -p --output-format json`, sonnet):
227
- * `usage.input_tokens = 2`, `usage.output_tokens = 4`, alongside `cache_read_input_tokens` and
228
- * `cache_creation_input_tokens`. Only the two plain counters are reported — cache tokens are a
229
- * SEPARATE dimension `wf-budget-1` has no field for, and silently folding them into `tokensIn`
230
- * would inflate every cached run's cost picture. Added to the architecture's export list; see the
231
- * manifest.
232
- */
233
- export function extractClaudeUsage(stdout: string): { tokensIn: number | null; tokensOut: number | null } {
234
- const env = extractClaudeResult(String(stdout ?? ''));
235
- if (!env.ok) return { tokensIn: null, tokensOut: null };
236
- // re-scan for the envelope object itself: extractClaudeResult hands back only the text
237
- const lines = String(stdout ?? '').split(/\r?\n/);
238
- const candidates = [String(stdout ?? '').trim(), ...lines.map((l) => l.trim())].filter((c) => c.startsWith('{') && c.endsWith('}'));
239
- let usage: Record<string, unknown> | null = null;
240
- for (const c of candidates) {
241
- let obj: unknown;
242
- try {
243
- obj = JSON.parse(c);
244
- } catch {
245
- continue;
303
+ function codexStructuredOutput(stdout: string): { structured: boolean; valid: boolean; text: string; usage: ObservedUsage; reportedModel?: string | null } {
304
+ const lines = stdout.split(/\r?\n/).filter((line) => line.trim());
305
+ const events: Record<string, unknown>[] = []; let malformed = false; let recognized = false;
306
+ const types = new Set(['thread.started', 'turn.started', 'turn.completed', 'turn.failed', 'item.started', 'item.updated', 'item.completed', 'error']);
307
+ for (const line of lines) {
308
+ if (/"type"\s*:\s*"(?:thread\.|turn\.|item\.|error")/.test(line)) recognized = true;
309
+ try { const event = JSON.parse(line) as Record<string, unknown>;
310
+ if (event && typeof event === 'object' && typeof event['type'] === 'string' && (types.has(event['type']) || /^(thread|turn|item)\./.test(event['type']))) recognized = true;
311
+ events.push(event);
312
+ } catch { malformed = true; }
313
+ }
314
+ if (!recognized) return { structured: false, valid: true, text: stdout.trim(), usage: emptyUsage() };
315
+ const result = emptyUsage(); let failedTerminal = false; let text = ''; let threadId: string | null = null; const terminals: Record<string, unknown>[] = []; const reportedModels = new Set<string>();
316
+ for (const event of events) {
317
+ if (!event || typeof event !== 'object' || !types.has(String(event['type']))) { malformed = true; continue; }
318
+ if (typeof event['model'] === 'string' && event['model'].trim()) reportedModels.add(event['model']);
319
+ if (event['type'] === 'thread.started' && typeof event['thread_id'] === 'string') { if (threadId !== null && threadId !== event['thread_id']) malformed = true; threadId = event['thread_id']; }
320
+ if (event['type'] === 'item.completed' && event['item'] && typeof event['item'] === 'object') {
321
+ const item = event['item'] as Record<string, unknown>; if (item['type'] === 'agent_message' && typeof item['text'] === 'string') text = item['text'];
246
322
  }
247
- if (obj === null || typeof obj !== 'object') continue;
248
- const u = (obj as Record<string, unknown>)['usage'];
249
- if (u !== null && typeof u === 'object' && !Array.isArray(u)) usage = u as Record<string, unknown>;
323
+ if (event['type'] === 'turn.completed' || (event['type'] === 'turn.failed' && event['usage'] != null)) terminals.push(event);
324
+ if (event['type'] === 'turn.failed' || event['type'] === 'error') failedTerminal = true;
325
+ }
326
+ const distinct = new Set(terminals.map((v) => { const u = v['usage'] && typeof v['usage'] === 'object' ? v['usage'] as Record<string,unknown> : {}; return JSON.stringify([v['turn_id'] ?? null, ...['input_tokens','cached_input_tokens','cache_write_input_tokens','output_tokens','reasoning_output_tokens','total_tokens'].map((key) => u[key] ?? null)]); }));
327
+ if (distinct.size !== 1) { malformed = true; result.usageDiagnostics!.push(distinct.size === 0 ? 'terminal-usage-unavailable' : 'conflicting-terminal-usage'); }
328
+ if (distinct.size === 1) {
329
+ const terminal = terminals[0]!; const raw = terminal['usage'];
330
+ if (raw && typeof raw === 'object' && !Array.isArray(raw)) {
331
+ const usage = raw as Record<string, unknown>;
332
+ result.tokensIn = count(usage['input_tokens']); result.tokensOut = count(usage['output_tokens']);
333
+ result.tokensCacheRead = count(usage['cached_input_tokens']); result.tokensCacheWrite = count(usage['cache_write_input_tokens']);
334
+ result.tokensReasoning = count(usage['reasoning_output_tokens']);
335
+ const derived = result.tokensIn !== null && result.tokensOut !== null ? count(result.tokensIn + result.tokensOut) : null;
336
+ result.tokensTotal = count(usage['total_tokens']) ?? derived;
337
+ result.totalDerivation = count(usage['total_tokens']) !== null ? 'reported' : derived === null ? 'not-recorded' : 'input-plus-output';
338
+ if (derived !== null && result.tokensTotal !== derived) result.usageDiagnostics!.push('total-split-mismatch');
339
+ if (result.tokensCacheRead != null && result.tokensIn != null && result.tokensCacheRead > result.tokensIn) result.usageDiagnostics!.push('cache-exceeds-input');
340
+ if (result.tokensReasoning != null && result.tokensOut != null && result.tokensReasoning > result.tokensOut) result.usageDiagnostics!.push('reasoning-exceeds-output');
341
+ for (const [key, value] of Object.entries(usage)) if (key.endsWith('_tokens') && count(value) === null) result.usageDiagnostics!.push('invalid-counter:' + key);
342
+ result.tokensSource = 'codex-json'; result.reportedTotalBasis = 'raw-inclusive'; result.inputCacheSemantics = 'includes-cache-read-write';
343
+ result.usageSource = { schema: 'codex-exec-json', scope: 'terminal-turn', threadId, turnId: typeof terminal['turn_id'] === 'string' ? terminal['turn_id'] : null, receiptId: null };
344
+ } else malformed = true;
250
345
  }
251
- if (usage === null) return { tokensIn: null, tokensOut: null };
252
- const pick = (k: string): number | null => {
253
- const v = usage?.[k];
254
- return typeof v === 'number' && Number.isFinite(v) && v >= 0 ? v : null;
346
+ if (malformed) result.usageDiagnostics!.push('malformed-event-stream');
347
+ if (failedTerminal) result.usageDiagnostics!.push('runtime-turn-failed');
348
+ if (reportedModels.size > 1) result.usageDiagnostics!.push('provider-model-mismatch');
349
+ return { structured: true, valid: !malformed && !failedTerminal, text, usage: result, reportedModel: reportedModels.size === 1 ? [...reportedModels][0]! : null };
350
+ }
351
+
352
+ /** Observe the existing wrapper seam; this does not attest OS child start. */
353
+ function collectProbeMetadata(family: BridgeFamily) {
354
+ const attempts: NonNullable<ProbeOutcome['provenance']>['attempts'] = [];
355
+ let totalConsidered = 0;
356
+ let invalid: Exclude<ProbeOutcome['provenanceReason'], undefined> = null;
357
+ return {
358
+ record(id: string, wrapperInvoked: boolean, r: ChildRun | null, selected: boolean) {
359
+ const ordinal = ++totalConsidered;
360
+ if (wrapperInvoked && (typeof id !== 'string' || !/^[A-Za-z0-9][A-Za-z0-9._:/-]{0,127}$/.test(id))) invalid = 'candidate-model-invalid';
361
+ let reason: NonNullable<ProbeOutcome['provenance']>['attempts'][number]['reason'] = 'invalid-candidate';
362
+ if (r !== null) {
363
+ const typed = typeof r.stdout === 'string' && typeof r.stderr === 'string' && typeof r.timedOut === 'boolean'
364
+ && (r.exitCode === null || (Number.isSafeInteger(r.exitCode) && r.exitCode >= 0))
365
+ && (r.spawnError === null || (typeof r.spawnError === 'string' && r.spawnError.length > 0));
366
+ const consistent = typed && !(r.timedOut && (r.spawnError !== null || r.exitCode === 0))
367
+ && !(r.spawnError !== null && r.exitCode !== null) && !(selected && (r.timedOut || r.spawnError !== null || r.exitCode !== 0));
368
+ if (!consistent && invalid !== 'candidate-model-invalid') invalid = 'wrapper-result-invalid';
369
+ reason = r.timedOut ? 'timeout' : r.spawnError !== null ? 'spawn-error' : r.exitCode === null ? 'no-exit-code'
370
+ : r.exitCode !== 0 ? 'exit-nonzero' : selected ? 'answered' : 'unexpected-response';
371
+ }
372
+ if (ordinal <= 32) attempts.push({ ordinal, model: wrapperInvoked ? id : null, family, wrapperInvoked,
373
+ outcome: !wrapperInvoked ? 'rejected' : selected ? 'answered' : 'failed', reason, selected });
374
+ },
375
+ finish(): Pick<ProbeOutcome, 'provenance' | 'provenanceReason'> {
376
+ return { provenance: invalid === null ? { schema: 'wf-probe-attempts-1', complete: totalConsidered <= 32, totalConsidered, attempts } : null,
377
+ provenanceReason: invalid };
378
+ },
255
379
  };
256
- return { tokensIn: pick('input_tokens'), tokensOut: pick('output_tokens') };
257
380
  }
258
381
 
259
382
  // ── the adapter factories (pure over the injected ChildRunner) ───────────────────────────────────
@@ -264,6 +387,7 @@ function failed(family: BridgeFamily, model: string | null, wallMs: number, reas
264
387
  text: null,
265
388
  family,
266
389
  modelUsed: model,
390
+ modelProvenance: 'probed-request',
267
391
  wallMs,
268
392
  tokensIn: null,
269
393
  tokensOut: null,
@@ -294,12 +418,16 @@ export function makeCodexExecDispatcher(run: ChildRunner, opts?: { bin?: string;
294
418
  const t0 = clock();
295
419
  const list = candidates.length > 0 ? candidates : ['gpt-5.5'];
296
420
  const started: string[] = [];
421
+ const observation = collectProbeMetadata('openai');
297
422
  for (const id of list) {
423
+ const wrapperInvoked = true;
298
424
  const r = await run(bin, codexProbeArgv(id), { stdinText: null, timeoutMs: 120_000, cwd: isolatedCwd(), detached: true });
299
425
  started.push(id);
300
- if (interpretCodexProbe(r)) return { id, wallMs: clock() - t0, detail: `codex answered on ${id}` };
426
+ const selected = interpretCodexProbe(r);
427
+ observation.record(id, wrapperInvoked, r, selected);
428
+ if (selected) return { id, wallMs: clock() - t0, detail: `codex answered on ${id}`, ...observation.finish() };
301
429
  }
302
- return { id: null, wallMs: clock() - t0, detail: `no codex candidate answered a probe (tried: ${started.join(', ')}) — an allowlist says an id is spellable, only a probe says it answers` };
430
+ return { id: null, wallMs: clock() - t0, detail: `no codex candidate answered a probe (tried: ${started.join(', ')}) — an allowlist says an id is spellable, only a probe says it answers`, ...observation.finish() };
303
431
  },
304
432
  dispatch: async (req) => {
305
433
  const t0 = clock();
@@ -314,25 +442,15 @@ export function makeCodexExecDispatcher(run: ChildRunner, opts?: { bin?: string;
314
442
  cwd: req.cwd,
315
443
  detached: true,
316
444
  });
445
+ const parsed = codexStructuredOutput(String(r.stdout ?? ''));
446
+ const usage = parsed.structured ? parsed.usage : extractCodexTokens(r.stderr);
317
447
  const cls = classifyChildRun(r);
318
- if (cls.kind === 'timeout') return failed('openai', req.resolvedModelId, req.timeoutMs, 'dispatch-timeout', `the ${req.timeoutMs}ms deadline fired on step ${req.stepId}`);
319
- if (cls.kind === 'dead') return failed('openai', req.resolvedModelId, clock() - t0, 'dispatch-dead', cls.detail);
320
- const text = String(r.stdout ?? '').trim();
321
- if (cls.exitCode !== 0 || text === '') {
322
- return failed('openai', req.resolvedModelId, clock() - t0, 'dispatch-dead',
323
- `codex exited ${cls.exitCode} with ${text === '' ? 'NO stdout' : String(text.length) + ' chars of stdout'} — a clean exit with nothing to read is the spawned-but-mute case, not an empty success`);
324
- }
325
- const tokens = extractCodexTokens(r.stderr);
326
- return {
327
- outcome: 'ok',
328
- text,
329
- family: 'openai',
330
- modelUsed: req.resolvedModelId,
331
- wallMs: clock() - t0,
332
- tokensIn: tokens.tokensIn,
333
- tokensOut: tokens.tokensOut,
334
- tokensSource: tokens.tokensIn === null && tokens.tokensOut === null ? null : 'codex-stderr',
335
- };
448
+ if (cls.kind === 'timeout') return { ...failed('openai', req.resolvedModelId, req.timeoutMs, 'dispatch-timeout', `the ${req.timeoutMs}ms deadline fired on step ${req.stepId}`), ...usage };
449
+ if (cls.kind === 'dead') return { ...failed('openai', req.resolvedModelId, clock() - t0, 'dispatch-dead', cls.detail), ...usage };
450
+ if (cls.exitCode !== 0 || !parsed.valid || parsed.text === '') return { ...failed('openai', req.resolvedModelId, clock() - t0, 'dispatch-dead', 'Codex exited nonzero, had an invalid structured event stream, or was spawned-but-mute with no final text'), ...usage };
451
+ return { outcome: 'ok', text: parsed.text, family: 'openai', modelUsed: parsed.reportedModel ?? req.resolvedModelId,
452
+ modelProvenance: parsed.reportedModel ? 'provider-reported' : 'probed-request', wallMs: clock() - t0, ...usage };
453
+
336
454
  },
337
455
  };
338
456
  }
@@ -358,14 +476,18 @@ export function makeClaudePDispatcher(run: ChildRunner, opts?: { bin?: string; i
358
476
  const t0 = clock();
359
477
  const list = candidates.length > 0 ? candidates : ['sonnet'];
360
478
  const tried: string[] = [];
479
+ const observation = collectProbeMetadata('claude');
361
480
  for (const id of list) {
362
481
  const argv = claudeProbeArgs(id);
363
482
  tried.push(id);
364
- if (argv === null) continue; // an unsafe id is not spellable, let alone answerable
483
+ if (argv === null) { observation.record(id, false, null, false); continue; } // existing unsafe rejection
484
+ const wrapperInvoked = true;
365
485
  const r = await run(bin, argv, { stdinText: null, timeoutMs: 120_000, cwd: isolatedCwd(), detached: true });
366
- if (interpretClaudeProbe({ stdout: r.stdout, exitCode: r.exitCode ?? 1 })) return { id, wallMs: clock() - t0, detail: `claude answered on ${id}` };
486
+ const selected = interpretClaudeProbe({ stdout: r.stdout, exitCode: r.exitCode ?? 1 });
487
+ observation.record(id, wrapperInvoked, r, selected);
488
+ if (selected) return { id, wallMs: clock() - t0, detail: `claude answered on ${id}`, ...observation.finish() };
367
489
  }
368
- return { id: null, wallMs: clock() - t0, detail: `no claude candidate answered a probe (tried: ${tried.join(', ')})` };
490
+ return { id: null, wallMs: clock() - t0, detail: `no claude candidate answered a probe (tried: ${tried.join(', ')})`, ...observation.finish() };
369
491
  },
370
492
  dispatch: async (req) => {
371
493
  const t0 = clock();
@@ -380,33 +502,15 @@ export function makeClaudePDispatcher(run: ChildRunner, opts?: { bin?: string; i
380
502
  cwd: fileMode ? req.cwd : isolatedCwd(),
381
503
  detached: true,
382
504
  });
505
+ const usage = extractClaudeUsage(r.stdout);
383
506
  const cls = classifyChildRun(r);
384
- if (cls.kind === 'timeout') return failed('claude', req.resolvedModelId, req.timeoutMs, 'dispatch-timeout', `the ${req.timeoutMs}ms deadline fired on step ${req.stepId}`);
385
- if (cls.kind === 'dead') return failed('claude', req.resolvedModelId, clock() - t0, 'dispatch-dead', cls.detail);
386
- // The EXIT CODE and the ENVELOPE must agree (Step-8 HIGH-10). A parseable success envelope
387
- // from a process that exited nonzero is a CONTRADICTION, not a success: the runtime told us
388
- // twice and the two answers differ, so believing the friendlier one is how a failed dispatch
389
- // becomes a green step. The codex adapter already required exit 0; this one did not.
390
- if (cls.exitCode !== 0) {
391
- return failed('claude', req.resolvedModelId, clock() - t0, 'dispatch-dead',
392
- `claude exited ${cls.exitCode} — a nonzero exit is a failed dispatch even when stdout carries a parseable success envelope; the two disagree and the exit code is the runtime's own verdict`);
393
- }
507
+ if (cls.kind === 'timeout') return { ...failed('claude', req.resolvedModelId, req.timeoutMs, 'dispatch-timeout', `the ${req.timeoutMs}ms deadline fired on step ${req.stepId}`), ...usage };
508
+ if (cls.kind === 'dead') return { ...failed('claude', req.resolvedModelId, clock() - t0, 'dispatch-dead', cls.detail), ...usage };
509
+ if (cls.exitCode !== 0) return { ...failed('claude', req.resolvedModelId, clock() - t0, 'dispatch-dead', `claude exited ${cls.exitCode} — a nonzero exit is a failed dispatch even with a success envelope`), ...usage };
394
510
  const env = extractClaudeResult(r.stdout);
395
- if (!env.ok) {
396
- return failed('claude', req.resolvedModelId, clock() - t0, 'dispatch-dead',
397
- `claude exited ${cls.exitCode} but the reply is not readable as a result envelope: ${env.detail}`);
398
- }
399
- const usage = extractClaudeUsage(r.stdout);
400
- return {
401
- outcome: 'ok',
402
- text: env.text,
403
- family: 'claude',
404
- modelUsed: req.resolvedModelId,
405
- wallMs: clock() - t0,
406
- tokensIn: usage.tokensIn,
407
- tokensOut: usage.tokensOut,
408
- tokensSource: usage.tokensIn === null && usage.tokensOut === null ? null : 'claude-envelope',
409
- };
511
+ if (!env.ok) return { ...failed('claude', req.resolvedModelId, clock() - t0, 'dispatch-dead', `reply is not readable as a result envelope: ${env.detail}`), ...usage };
512
+ return { outcome: 'ok', text: env.text, family: 'claude', modelUsed: req.resolvedModelId, modelProvenance: 'probed-request', wallMs: clock() - t0, ...usage };
513
+
410
514
  },
411
515
  };
412
516
  }
@@ -43,7 +43,7 @@ import { classifyFailure, errSnap, gateVerdict, joinRegion, stepContractLines, t
43
43
  import { checkpointInputHash, decideCheckpointResume, parseCheckpointRead, serializeCheckpoint } from './feature-adr-checkpoints.js';
44
44
  import { modelFamily, type BridgeFamily } from './qe-bridge.js';
45
45
  import { buildReqeDebt } from './reqe.js';
46
- import { CODEX_EXEC_XHIGH_TIMEOUT_MS, defangGateEchoes, type DispatchResult, type Dispatcher } from './workflow-run-dispatch.js';
46
+ import { CODEX_EXEC_XHIGH_TIMEOUT_MS, defangGateEchoes, type DispatchResult, type Dispatcher, type ProbeOutcome } from './workflow-run-dispatch.js';
47
47
 
48
48
  // ─────────────────────────────────────────────────────────────────────────────
49
49
  // Schemas / constants
@@ -164,7 +164,8 @@ export interface WfRunState {
164
164
  dispatcherOverride?: boolean;
165
165
  }
166
166
 
167
- export interface WfBudgetRow {
167
+ export interface WfBudgetRow extends Pick<DispatchResult, 'tokensTotal' | 'tokensCacheRead' | 'tokensCacheWrite' | 'tokensReasoning' | 'reportedTotalBasis' | 'inputCacheSemantics' | 'reportedCostUsd' | 'usageDiagnostics' | 'usageSource' | 'totalDerivation'> {
168
+ projectRoot?: string;
168
169
  schema: typeof WF_BUDGET_ROW_SCHEMA;
169
170
  kind: 'stage' | 'probe';
170
171
  runId: string;
@@ -175,10 +176,23 @@ export interface WfBudgetRow {
175
176
  attempt: number | null;
176
177
  family: BridgeFamily;
177
178
  model: string | null;
179
+ modelProvenance?: string;
180
+ requestedModel?: string | null;
181
+ plannedModel?: string | null;
182
+ plannedModelSource?: 'plan-declared' | 'plan-omitted' | 'unavailable';
183
+ probeId?: string | null;
184
+ probeProvenance?: Exclude<ProbeOutcome['provenance'], undefined>;
185
+ probeSource?: 'dispatcher-child-seam' | 'scripted-dispatcher';
186
+ probeObservationReason?: 'producer-not-recorded' | 'id-factory-missing' | 'id-factory-invalid' | 'provenance-invalid' | 'candidate-model-invalid' | 'wrapper-result-invalid' | 'selected-model-invalid' | null;
178
187
  wallMs: number;
179
188
  tokensIn: number | null;
180
189
  tokensOut: number | null;
181
- tokensSource: 'claude-envelope' | 'codex-stderr' | null;
190
+ tokensSource: DispatchResult['tokensSource'];
191
+ phase?: string | null;
192
+ role?: string | null;
193
+ tier?: string | null;
194
+ mode?: string | null;
195
+ estimate?: unknown;
182
196
  outcome: 'ok' | 'null' | 'error' | null;
183
197
  timeoutMs: number | null;
184
198
  }
@@ -239,6 +253,7 @@ export interface RunnerInputs {
239
253
  wallClockExtraMs: number | null;
240
254
  runnerVersion: string;
241
255
  cwdRoot: string;
256
+ projectRoot?: string;
242
257
  }
243
258
 
244
259
  /** Small, dependency-free 64-bit FNV — the same shape the checkpoint plane uses, kept local so this
@@ -927,6 +942,35 @@ export interface RunStore {
927
942
  writeReqeDebt(record: object): void;
928
943
  }
929
944
 
945
+ const safeRoutingModel = (value: unknown): value is string => typeof value === 'string' && /^[A-Za-z0-9][A-Za-z0-9._:/-]{0,127}$/.test(value);
946
+ function probeRecordReason(probe: ProbeOutcome, family: BridgeFamily): Exclude<WfBudgetRow['probeObservationReason'], undefined> {
947
+ const exact = (value: unknown, keys: string[]): value is Record<string, unknown> => value !== null && typeof value === 'object'
948
+ && (Object.getPrototypeOf(value) === Object.prototype || Object.getPrototypeOf(value) === null)
949
+ && Reflect.ownKeys(value).length === keys.length && keys.every(key => Object.hasOwn(value, key));
950
+ if (probe.id !== null && !safeRoutingModel(probe.id)) return 'selected-model-invalid';
951
+ const present = [Object.hasOwn(probe, 'provenance'), Object.hasOwn(probe, 'provenanceReason')];
952
+ if (present.every(value => !value)) return 'producer-not-recorded';
953
+ if (!present.every(Boolean)) return 'provenance-invalid';
954
+ if (probe.provenance === null && (probe.provenanceReason === 'candidate-model-invalid' || probe.provenanceReason === 'wrapper-result-invalid')) return probe.provenanceReason;
955
+ const p = probe.provenance;
956
+ if (probe.provenanceReason !== null || !exact(p, ['schema', 'complete', 'totalConsidered', 'attempts']) || p.schema !== 'wf-probe-attempts-1'
957
+ || typeof p.complete !== 'boolean' || !Number.isSafeInteger(p.totalConsidered) || p.totalConsidered <= 0 || !Array.isArray(p.attempts)
958
+ || p.attempts.length !== Math.min(p.totalConsidered, 32) || p.complete !== (p.totalConsidered <= 32)) return 'provenance-invalid';
959
+ for (const [index, a] of p.attempts.entries()) {
960
+ if (!exact(a, ['ordinal', 'model', 'family', 'wrapperInvoked', 'outcome', 'reason', 'selected']) || a.ordinal !== index + 1
961
+ || a.family !== family || typeof a.wrapperInvoked !== 'boolean' || typeof a.selected !== 'boolean') return 'provenance-invalid';
962
+ const rejected = family === 'claude' && !a.wrapperInvoked && !a.selected && a.model === null && a.outcome === 'rejected' && a.reason === 'invalid-candidate';
963
+ const answered = a.wrapperInvoked && a.selected && safeRoutingModel(a.model) && a.outcome === 'answered' && a.reason === 'answered';
964
+ const failed = a.wrapperInvoked && !a.selected && safeRoutingModel(a.model) && a.outcome === 'failed'
965
+ && ['timeout', 'spawn-error', 'no-exit-code', 'exit-nonzero', 'unexpected-response'].includes(a.reason);
966
+ if (!(rejected || answered || failed)) return 'provenance-invalid';
967
+ }
968
+ const selected = p.attempts.filter(a => a.selected);
969
+ if (!p.complete ? selected.length !== 0 : probe.id === null ? selected.length !== 0
970
+ : selected.length !== 1 || selected[0]?.ordinal !== p.totalConsidered || selected[0]?.model !== probe.id) return 'provenance-invalid';
971
+ return null;
972
+ }
973
+
930
974
  export interface SchedulerDeps {
931
975
  store: RunStore;
932
976
  dispatchers: Record<BridgeFamily, Dispatcher>;
@@ -935,6 +979,7 @@ export interface SchedulerDeps {
935
979
  /** Injected ISO clock (determinism, NFR-2). */
936
980
  now(): string;
937
981
  monotonicMs(): number;
982
+ newProbeId?: () => string;
938
983
  /**
939
984
  * TEST SEAM (named, never a casual flag): disables the landed barrier so the F5 mutant can show
940
985
  * the lying file-step passing. Default false; a run that sets it records `dispatcherOverride`.
@@ -978,6 +1023,7 @@ function failureClassOf(res: DispatchResult): FailureClass | null {
978
1023
  }
979
1024
 
980
1025
  interface RunCtx {
1026
+ probeIds: Partial<Record<BridgeFamily, string | null>>;
981
1027
  inputs: RunnerInputs;
982
1028
  pre: PreflightOk;
983
1029
  deps: SchedulerDeps;
@@ -1053,16 +1099,36 @@ export async function runWorkflow(inputs: RunnerInputs, pre: PreflightOk, deps:
1053
1099
 
1054
1100
  const usedFamilies = [...new Set(Object.values(pre.families))];
1055
1101
  const probedIds: Partial<Record<BridgeFamily, string | null>> = {};
1102
+ const probeIds: Partial<Record<BridgeFamily, string | null>> = {};
1056
1103
  let probeAgentCalls = 0;
1057
1104
  const probeDetail: Partial<Record<BridgeFamily, string>> = {};
1058
1105
  for (const family of usedFamilies) {
1059
1106
  const candidates = [...new Set(allSpecs(pre.projection).filter((s) => pre.families[s.stepId] === family).map((s) => s.model).filter((m): m is string => typeof m === 'string'))];
1060
1107
  const probe = await deps.dispatchers[family].probe(candidates);
1061
1108
  probeAgentCalls++;
1109
+ let probeId: string | null = null;
1110
+ let idReason: Exclude<WfBudgetRow['probeObservationReason'], undefined> = 'id-factory-missing';
1111
+ if (deps.newProbeId) {
1112
+ try {
1113
+ const id = deps.newProbeId();
1114
+ if (typeof id === 'string' && /^[0-9a-f]{32}$/.test(id)) {
1115
+ probeId = id;
1116
+ idReason = null;
1117
+ } else idReason = 'id-factory-invalid';
1118
+ } catch { idReason = 'id-factory-invalid'; }
1119
+ }
1120
+ const producerReason = probeRecordReason(probe, family);
1121
+ const probeObservationReason = producerReason && producerReason !== 'producer-not-recorded' ? producerReason : idReason ?? producerReason;
1122
+ probeIds[family] = probeId;
1062
1123
  store.appendBudgetRow({
1063
1124
  schema: WF_BUDGET_ROW_SCHEMA,
1064
1125
  kind: 'probe',
1126
+ probeId,
1127
+ probeProvenance: probeObservationReason === null ? probe.provenance! : null,
1128
+ probeSource: deps.dispatcherOverride ? 'scripted-dispatcher' : 'dispatcher-child-seam',
1129
+ probeObservationReason,
1065
1130
  runId: inputs.runId,
1131
+ projectRoot: inputs.projectRoot ?? inputs.cwdRoot,
1066
1132
  dispatchSeq: null,
1067
1133
  stepId: null,
1068
1134
  itemKey: null,
@@ -1073,6 +1139,14 @@ export async function runWorkflow(inputs: RunnerInputs, pre: PreflightOk, deps:
1073
1139
  tokensIn: null,
1074
1140
  tokensOut: null,
1075
1141
  tokensSource: null,
1142
+ tokensTotal: null,
1143
+ tokensCacheRead: null,
1144
+ tokensCacheWrite: null,
1145
+ tokensReasoning: null,
1146
+ reportedTotalBasis: 'unknown',
1147
+ inputCacheSemantics: 'unknown',
1148
+ reportedCostUsd: null,
1149
+ usageDiagnostics: ['probe-usage-not-recorded'],
1076
1150
  outcome: null,
1077
1151
  timeoutMs: null,
1078
1152
  });
@@ -1189,6 +1263,7 @@ export async function runWorkflow(inputs: RunnerInputs, pre: PreflightOk, deps:
1189
1263
 
1190
1264
  const opened = openTrace(inputs, pre, store, inputs.resume !== null);
1191
1265
  const ctx: RunCtx = {
1266
+ probeIds,
1192
1267
  inputs,
1193
1268
  pre,
1194
1269
  deps,
@@ -1500,16 +1575,37 @@ async function dispatchOnce(
1500
1575
  schema: WF_BUDGET_ROW_SCHEMA,
1501
1576
  kind: 'stage',
1502
1577
  runId: ctx.inputs.runId,
1578
+ projectRoot: ctx.inputs.projectRoot ?? ctx.inputs.cwdRoot,
1503
1579
  dispatchSeq,
1504
1580
  stepId: spec.stepId,
1505
1581
  itemKey,
1506
1582
  attempt,
1507
1583
  family,
1508
- model: res.modelUsed ?? model,
1584
+ model: res.modelUsed,
1585
+ requestedModel: model,
1586
+ plannedModel: safeRoutingModel(spec.model) ? spec.model : null,
1587
+ plannedModelSource: safeRoutingModel(spec.model) ? 'plan-declared' : spec.model == null ? 'plan-omitted' : 'unavailable',
1588
+ probeId: ctx.probeIds[family] ?? null,
1589
+ modelProvenance: res.modelProvenance ?? (res.modelUsed === null ? 'not-recorded' : 'dispatcher-reported'),
1509
1590
  wallMs,
1510
1591
  tokensIn: res.tokensIn,
1511
1592
  tokensOut: res.tokensOut,
1512
1593
  tokensSource: res.tokensSource,
1594
+ tokensTotal: res.tokensTotal ?? null,
1595
+ totalDerivation: res.totalDerivation ?? 'not-recorded',
1596
+ tokensCacheRead: res.tokensCacheRead ?? null,
1597
+ tokensCacheWrite: res.tokensCacheWrite ?? null,
1598
+ tokensReasoning: res.tokensReasoning ?? null,
1599
+ reportedTotalBasis: res.reportedTotalBasis ?? 'unknown',
1600
+ inputCacheSemantics: res.inputCacheSemantics ?? 'unknown',
1601
+ reportedCostUsd: res.reportedCostUsd ?? null,
1602
+ usageDiagnostics: res.usageDiagnostics ?? [],
1603
+ ...(res.usageSource !== undefined ? { usageSource: res.usageSource } : {}),
1604
+ phase: spec.phase,
1605
+ role: spec.role ?? null,
1606
+ tier: spec.tier ?? null,
1607
+ mode: spec.mode ?? null,
1608
+ estimate: spec.estimate ?? null,
1513
1609
  outcome,
1514
1610
  timeoutMs,
1515
1611
  });
@@ -1699,7 +1795,7 @@ function appendLedger(ctx: RunCtx, outcome: string): void {
1699
1795
  outcome,
1700
1796
  date: null,
1701
1797
  });
1702
- if (line !== null) ctx.deps.store.appendLedgerLine(line);
1798
+ if (line !== null) ctx.deps.store.appendLedgerLine(JSON.stringify({ ...JSON.parse(line), summary: true, sourceProjection: 'workflow-budget', workflowRunId: ctx.inputs.runId }));
1703
1799
  }
1704
1800
 
1705
1801
  /** PAUSE — flush WITHOUT `run.closed` (parity with the render's top-level terminal return, which