@relayflows/sdk 2.0.12 → 2.0.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/dist/authored-budget.d.ts.map +1 -1
  2. package/dist/authored-budget.js +16 -7
  3. package/dist/authored-budget.js.map +1 -1
  4. package/dist/authored-flow-executor.d.ts.map +1 -1
  5. package/dist/authored-flow-executor.js +5 -0
  6. package/dist/authored-flow-executor.js.map +1 -1
  7. package/dist/budget-preflight.d.ts +17 -6
  8. package/dist/budget-preflight.d.ts.map +1 -1
  9. package/dist/budget-preflight.js +9 -6
  10. package/dist/budget-preflight.js.map +1 -1
  11. package/dist/cli/build.d.ts.map +1 -1
  12. package/dist/cli/build.js +8 -0
  13. package/dist/cli/build.js.map +1 -1
  14. package/dist/cli/deploy.d.ts.map +1 -1
  15. package/dist/cli/deploy.js +5 -0
  16. package/dist/cli/deploy.js.map +1 -1
  17. package/dist/cli/direct-run.d.ts.map +1 -1
  18. package/dist/cli/direct-run.js +8 -1
  19. package/dist/cli/direct-run.js.map +1 -1
  20. package/dist/cli/run.d.ts +23 -0
  21. package/dist/cli/run.d.ts.map +1 -1
  22. package/dist/cli/run.js +78 -11
  23. package/dist/cli/run.js.map +1 -1
  24. package/dist/cli/step-failure.d.ts +40 -0
  25. package/dist/cli/step-failure.d.ts.map +1 -0
  26. package/dist/cli/step-failure.js +150 -0
  27. package/dist/cli/step-failure.js.map +1 -0
  28. package/dist/cli.d.ts.map +1 -1
  29. package/dist/cli.js +4 -4
  30. package/dist/cli.js.map +1 -1
  31. package/dist/daemon-connection.d.ts +7 -0
  32. package/dist/daemon-connection.d.ts.map +1 -1
  33. package/dist/daemon-connection.js +7 -0
  34. package/dist/daemon-connection.js.map +1 -1
  35. package/dist/failure-kinds.d.ts +25 -3
  36. package/dist/failure-kinds.d.ts.map +1 -1
  37. package/dist/failure-kinds.js +5 -2
  38. package/dist/failure-kinds.js.map +1 -1
  39. package/dist/journal-client.d.ts +2 -6
  40. package/dist/journal-client.d.ts.map +1 -1
  41. package/dist/journal-client.js.map +1 -1
  42. package/dist/model-pricing.d.ts +7 -5
  43. package/dist/model-pricing.d.ts.map +1 -1
  44. package/dist/model-pricing.js +7 -5
  45. package/dist/model-pricing.js.map +1 -1
  46. package/dist/preflight.d.ts +5 -0
  47. package/dist/preflight.d.ts.map +1 -1
  48. package/dist/preflight.js.map +1 -1
  49. package/dist/protocol.d.ts +22 -5
  50. package/dist/protocol.d.ts.map +1 -1
  51. package/dist/spec.d.ts +22 -7
  52. package/dist/spec.d.ts.map +1 -1
  53. package/dist/worker-spend.d.ts +13 -9
  54. package/dist/worker-spend.d.ts.map +1 -1
  55. package/dist/worker-spend.js +22 -8
  56. package/dist/worker-spend.js.map +1 -1
  57. package/package.json +2 -2
  58. package/src/authored-budget.ts +35 -9
  59. package/src/authored-flow-executor.ts +5 -0
  60. package/src/budget-preflight.ts +17 -6
  61. package/src/cli/build.ts +7 -0
  62. package/src/cli/deploy.ts +4 -0
  63. package/src/cli/direct-run.ts +8 -0
  64. package/src/cli/run.ts +94 -11
  65. package/src/cli/step-failure.ts +160 -0
  66. package/src/cli.ts +4 -4
  67. package/src/daemon-connection.ts +8 -0
  68. package/src/failure-kinds.ts +25 -3
  69. package/src/journal-client.ts +2 -1
  70. package/src/model-pricing.ts +7 -5
  71. package/src/preflight.ts +5 -0
  72. package/src/protocol.ts +15 -2
  73. package/src/spec.ts +23 -1
  74. package/src/worker-spend.ts +25 -9
  75. package/dist/cli/deterministic-failure.d.ts +0 -5
  76. package/dist/cli/deterministic-failure.d.ts.map +0 -1
  77. package/dist/cli/deterministic-failure.js +0 -66
  78. package/dist/cli/deterministic-failure.js.map +0 -1
  79. package/src/cli/deterministic-failure.ts +0 -69
@@ -1,20 +1,24 @@
1
+ import type { StepUsage } from './protocol.js';
1
2
  import type { WorkerCliResult } from './worker-cli.js';
2
3
  /**
3
4
  * Attach token/dollar usage to a worker's CLI result. Invalid token counts
4
5
  * (non-integer or negative) are the one remaining failure mode — those are
5
6
  * journaled as `worker_error` with the usage projected from clamped counts.
6
7
  *
7
- * Unpriced models are NOT a failure here: the step journals zero dollars, so it
8
- * never trips a dollar budget, but its reported tokens still count toward any
9
- * token budget. Preflight (see `budgetDiagnostics`) warns that such a step is
10
- * unmetered for dollars.
8
+ * Contract across preflight worker kernel (keep all three aligned):
9
+ * - preflight (`budgetDiagnostics`) warns `budget_unmetered` for a
10
+ * dollar-budgeted step with no frozen price and lets it run;
11
+ * - this function returns priced usage (`dollars`) for a priced model, and
12
+ * unmetered usage (`dollars_unmetered: true`, no `dollars`) otherwise —
13
+ * never `undefined` for a successful decode, so token ceilings always see the
14
+ * step and the journal always says whether its dollars are known;
15
+ * - the kernel (`machine/budget.rs`) enforces tokens on every charge and
16
+ * dollars on metered charges only, and rejects unmetered usage that also
17
+ * claims non-zero dollars.
18
+ * `tests/budget-unmetered-live.test.ts` pins the chain end to end.
11
19
  */
12
20
  export declare function workerSpend(result: WorkerCliResult, model?: string): {
13
21
  result: WorkerCliResult;
14
- usage: {
15
- tokens_in: number;
16
- tokens_out: number;
17
- dollars: string;
18
- } | undefined;
22
+ usage: StepUsage | undefined;
19
23
  };
20
24
  //# sourceMappingURL=worker-spend.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"worker-spend.d.ts","sourceRoot":"","sources":["../src/worker-spend.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,iBAAiB,CAAC;AAQvD;;;;;;;;;GASG;AACH,wBAAgB,WAAW,CAAC,MAAM,EAAE,eAAe,EAAE,KAAK,CAAC,EAAE,MAAM;;;;;;;EAclE"}
1
+ {"version":3,"file":"worker-spend.d.ts","sourceRoot":"","sources":["../src/worker-spend.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,SAAS,EAAE,MAAM,eAAe,CAAC;AAC/C,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,iBAAiB,CAAC;AAgBvD;;;;;;;;;;;;;;;;GAgBG;AACH,wBAAgB,WAAW,CAAC,MAAM,EAAE,eAAe,EAAE,KAAK,CAAC,EAAE,MAAM,GAAG;IAAE,MAAM,EAAE,eAAe,CAAC;IAAC,KAAK,EAAE,SAAS,GAAG,SAAS,CAAA;CAAE,CAc9H"}
@@ -1,19 +1,33 @@
1
1
  import { pricedUsage } from './model-pricing.js';
2
- /** Zero-dollar usage for an unpriced step, only when the CLI reported tokens. */
2
+ /**
3
+ * Usage for a step whose dollar cost is unknown (unpriced or undeclared model).
4
+ *
5
+ * It carries the reported tokens and `dollars_unmetered: true`, and never a
6
+ * `dollars` amount: unknown cost is not journaled as a measured $0. The kernel
7
+ * records the flag on the step's `budget`/`spend` and on the run total, counts
8
+ * the tokens toward token ceilings, and compares only metered dollars against
9
+ * `maxDollars`. A CLI that reported no tokens still marks the step unmetered,
10
+ * with zero tokens — the same token accounting any unreported step gets.
11
+ */
3
12
  function unmeteredUsage(input, output) {
4
- if (input === undefined || output === undefined)
5
- return undefined;
6
- return { tokens_in: input, tokens_out: output, dollars: '0.000000' };
13
+ return { tokens_in: input ?? 0, tokens_out: output ?? 0, dollars_unmetered: true };
7
14
  }
8
15
  /**
9
16
  * Attach token/dollar usage to a worker's CLI result. Invalid token counts
10
17
  * (non-integer or negative) are the one remaining failure mode — those are
11
18
  * journaled as `worker_error` with the usage projected from clamped counts.
12
19
  *
13
- * Unpriced models are NOT a failure here: the step journals zero dollars, so it
14
- * never trips a dollar budget, but its reported tokens still count toward any
15
- * token budget. Preflight (see `budgetDiagnostics`) warns that such a step is
16
- * unmetered for dollars.
20
+ * Contract across preflight worker kernel (keep all three aligned):
21
+ * - preflight (`budgetDiagnostics`) warns `budget_unmetered` for a
22
+ * dollar-budgeted step with no frozen price and lets it run;
23
+ * - this function returns priced usage (`dollars`) for a priced model, and
24
+ * unmetered usage (`dollars_unmetered: true`, no `dollars`) otherwise —
25
+ * never `undefined` for a successful decode, so token ceilings always see the
26
+ * step and the journal always says whether its dollars are known;
27
+ * - the kernel (`machine/budget.rs`) enforces tokens on every charge and
28
+ * dollars on metered charges only, and rejects unmetered usage that also
29
+ * claims non-zero dollars.
30
+ * `tests/budget-unmetered-live.test.ts` pins the chain end to end.
17
31
  */
18
32
  export function workerSpend(result, model) {
19
33
  try {
@@ -1 +1 @@
1
- {"version":3,"file":"worker-spend.js","sourceRoot":"","sources":["../src/worker-spend.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAE,MAAM,oBAAoB,CAAC;AAGjD,iFAAiF;AACjF,SAAS,cAAc,CAAC,KAAyB,EAAE,MAA0B;IAC3E,IAAI,KAAK,KAAK,SAAS,IAAI,MAAM,KAAK,SAAS;QAAE,OAAO,SAAS,CAAC;IAClE,OAAO,EAAE,SAAS,EAAE,KAAK,EAAE,UAAU,EAAE,MAAM,EAAE,OAAO,EAAE,UAAU,EAAE,CAAC;AACvE,CAAC;AAED;;;;;;;;;GASG;AACH,MAAM,UAAU,WAAW,CAAC,MAAuB,EAAE,KAAc;IACjE,IAAI,CAAC;QACH,MAAM,KAAK,GAAG,WAAW,CAAC,KAAK,EAAE,MAAM,CAAC,YAAY,EAAE,MAAM,CAAC,aAAa,CAAC;eACtE,cAAc,CAAC,MAAM,CAAC,YAAY,EAAE,MAAM,CAAC,aAAa,CAAC,CAAC;QAC/D,OAAO,EAAE,MAAM,EAAE,KAAK,EAAE,CAAC;IAC3B,CAAC;IACD,OAAO,KAAK,EAAE,CAAC;QACb,OAAO;YACL,MAAM,EAAE,EAAE,GAAG,MAAM,EAAE,SAAS,EAAE,IAAI,EAAE,WAAW,EAAE,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,qBAAqB,EAAE;YACnH,KAAK,EAAE,WAAW,CAAC,SAAS,EAC1B,MAAM,CAAC,aAAa,CAAC,MAAM,CAAC,YAAY,CAAC,IAAI,MAAM,CAAC,YAAa,IAAI,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,YAAY,CAAC,CAAC,CAAC,CAAC,EAChG,MAAM,CAAC,aAAa,CAAC,MAAM,CAAC,aAAa,CAAC,IAAI,MAAM,CAAC,aAAc,IAAI,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,aAAa,CAAC,CAAC,CAAC,CAAC,CAAC;SACvG,CAAC;IACJ,CAAC;AACH,CAAC"}
1
+ {"version":3,"file":"worker-spend.js","sourceRoot":"","sources":["../src/worker-spend.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAE,MAAM,oBAAoB,CAAC;AAIjD;;;;;;;;;GASG;AACH,SAAS,cAAc,CAAC,KAAyB,EAAE,MAA0B;IAC3E,OAAO,EAAE,SAAS,EAAE,KAAK,IAAI,CAAC,EAAE,UAAU,EAAE,MAAM,IAAI,CAAC,EAAE,iBAAiB,EAAE,IAAI,EAAE,CAAC;AACrF,CAAC;AAED;;;;;;;;;;;;;;;;GAgBG;AACH,MAAM,UAAU,WAAW,CAAC,MAAuB,EAAE,KAAc;IACjE,IAAI,CAAC;QACH,MAAM,KAAK,GAAG,WAAW,CAAC,KAAK,EAAE,MAAM,CAAC,YAAY,EAAE,MAAM,CAAC,aAAa,CAAC;eACtE,cAAc,CAAC,MAAM,CAAC,YAAY,EAAE,MAAM,CAAC,aAAa,CAAC,CAAC;QAC/D,OAAO,EAAE,MAAM,EAAE,KAAK,EAAE,CAAC;IAC3B,CAAC;IACD,OAAO,KAAK,EAAE,CAAC;QACb,OAAO;YACL,MAAM,EAAE,EAAE,GAAG,MAAM,EAAE,SAAS,EAAE,IAAI,EAAE,WAAW,EAAE,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,qBAAqB,EAAE;YACnH,KAAK,EAAE,WAAW,CAAC,SAAS,EAC1B,MAAM,CAAC,aAAa,CAAC,MAAM,CAAC,YAAY,CAAC,IAAI,MAAM,CAAC,YAAa,IAAI,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,YAAY,CAAC,CAAC,CAAC,CAAC,EAChG,MAAM,CAAC,aAAa,CAAC,MAAM,CAAC,aAAa,CAAC,IAAI,MAAM,CAAC,aAAc,IAAI,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,aAAa,CAAC,CAAC,CAAC,CAAC,CAAC;SACvG,CAAC;IACJ,CAAC;AACH,CAAC"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@relayflows/sdk",
3
- "version": "2.0.12",
3
+ "version": "2.0.14",
4
4
  "description": "TypeScript-first authoring SDK for Relayflows. Compiles specs; speaks the journal protocol.",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",
@@ -45,7 +45,7 @@
45
45
  "@modelcontextprotocol/sdk": "^1.30.0",
46
46
  "@relayfile/adapter-core": "0.5.24",
47
47
  "@relayfile/relay-helpers": "0.4.11",
48
- "@relayflows/surface": "2.0.12",
48
+ "@relayflows/surface": "2.0.14",
49
49
  "@types/js-yaml": "^4.0.9",
50
50
  "ai-hist": "0.4.1",
51
51
  "ajv": "^8.17.1",
@@ -2,14 +2,31 @@ import { BudgetSyntaxError, parseBudget, toKernelBudget } from './budget.js';
2
2
  import { AuthoredFlowExecutionError } from './authored-flow-error.js';
3
3
  import type { JournalClient } from './journal-client.js';
4
4
  import type { RunOutcome } from './protocol.js';
5
- import type { KernelBudgetSpec, KernelRunSpec } from './spec.js';
5
+ import type { KernelBudgetSpec, KernelPriorSpend, KernelRunSpec } from './spec.js';
6
+
7
+ /**
8
+ * One journaled charge, in the accumulator's exact integer form.
9
+ *
10
+ * `unmetered` mirrors the journal's `budget.dollars_unmetered`: the charge
11
+ * spent tokens whose dollar cost is unknown, so its `micro` is a lower bound
12
+ * and not a measured amount. It is carried, not derived from `micro`, because
13
+ * an unmetered charge and a genuinely free charge both report zero dollars.
14
+ */
15
+ interface Charge {
16
+ input: bigint;
17
+ output: bigint;
18
+ micro: bigint;
19
+ ms: bigint;
20
+ day: number;
21
+ unmetered: boolean;
22
+ }
6
23
 
7
24
  /** Serialized admission for the internal authored runner's separate step runs. */
8
25
  export class AuthoredBudget {
9
26
  private readonly limit: KernelBudgetSpec | undefined;
10
27
  private failed = false;
11
28
  private tail: Promise<unknown> = Promise.resolve();
12
- private charges: { input: bigint; output: bigint; micro: bigint; ms: bigint; day: number }[] = [];
29
+ private charges: Charge[] = [];
13
30
 
14
31
  constructor(header: unknown) {
15
32
  try {
@@ -32,14 +49,22 @@ export class AuthoredBudget {
32
49
  // Window selection follows journal timestamps, never the SDK host clock.
33
50
  const day = this.charges.at(-1)?.day;
34
51
  const total = this.charges.filter(c => this.limit!.window !== 'day' || c.day === day)
35
- .reduce((s, c) => ({ input: s.input + c.input, output: s.output + c.output, micro: s.micro + c.micro, ms: s.ms + c.ms }),
36
- { input: 0n, output: 0n, micro: 0n, ms: 0n });
52
+ .reduce((s, c) => ({ input: s.input + c.input, output: s.output + c.output, micro: s.micro + c.micro, ms: s.ms + c.ms,
53
+ // Sticky, exactly as the kernel's running total is: once any charge
54
+ // in the window was unmetered, the carried dollars are a lower bound.
55
+ unmetered: s.unmetered || c.unmetered }),
56
+ { input: 0n, output: 0n, micro: 0n, ms: 0n, unmetered: false });
37
57
  const exactNumber = (n: bigint) => { if (n > BigInt(Number.MAX_SAFE_INTEGER)) throw new Error('budget counter overflow'); return Number(n); };
38
- const outcome = await journal.runStart({ ...spec, budget: { ...this.limit, prior_spend: {
58
+ // One explicit conversion to the shared carried-spend shape, so a new
59
+ // accounting field is added here rather than silently dropped inline.
60
+ const priorSpend: KernelPriorSpend = {
39
61
  tokens_in: exactNumber(total.input), tokens_out: exactNumber(total.output),
40
62
  dollars: `${total.micro / 1_000_000n}.${String(total.micro % 1_000_000n).padStart(6, '0')}`,
41
- wallclock_ms: exactNumber(total.ms), ...(this.limit.window === 'day' && day !== undefined ? { day } : {}),
42
- } } }, undefined, admissionKey);
63
+ wallclock_ms: exactNumber(total.ms),
64
+ ...(this.limit.window === 'day' && day !== undefined ? { day } : {}),
65
+ ...(total.unmetered ? { dollars_unmetered: true as const } : {}),
66
+ };
67
+ const outcome = await journal.runStart({ ...spec, budget: { ...this.limit, prior_spend: priorSpend } }, undefined, admissionKey);
43
68
  try {
44
69
  if (outcome.completion_reason === 'budget_exceeded') throw new AuthoredFlowExecutionError('step_failed', 'Flow budget exceeded before the next step.', 'budget_exceeded', outcome.run_id);
45
70
  return await consume(outcome);
@@ -49,7 +74,7 @@ export class AuthoredBudget {
49
74
  const { entries } = await journal.journalRead(outcome.run_id, seq);
50
75
  if (entries.length === 0) break;
51
76
  for (const raw of entries) {
52
- const e = raw as {seq: number; entry_type: string; at_ms: number; payload: {budget?: {tokens_in: number; tokens_out: number; dollars: string}; spend?: {wallclock_ms: number}}};
77
+ const e = raw as {seq: number; entry_type: string; at_ms: number; payload: {budget?: {tokens_in: number; tokens_out: number; dollars: string; dollars_unmetered?: boolean}; spend?: {wallclock_ms: number}}};
53
78
  seq = e.seq + 1;
54
79
  if (!['step.completed', 'memory.injected'].includes(e.entry_type)) continue;
55
80
  const b = e.payload.budget;
@@ -58,7 +83,8 @@ export class AuthoredBudget {
58
83
  if (fraction.length > 6 && /[1-9]/.test(fraction.slice(6))) throw new Error('budget accounting requires microdollar precision');
59
84
  this.charges.push({input: BigInt(b.tokens_in), output: BigInt(b.tokens_out),
60
85
  micro: BigInt(whole!) * 1_000_000n + BigInt(fraction.slice(0, 6).padEnd(6, '0')),
61
- ms: BigInt(e.payload.spend?.wallclock_ms ?? 0), day: Math.floor(e.at_ms / 86_400_000)});
86
+ ms: BigInt(e.payload.spend?.wallclock_ms ?? 0), day: Math.floor(e.at_ms / 86_400_000),
87
+ unmetered: b.dollars_unmetered === true});
62
88
  }
63
89
  }
64
90
  }
@@ -148,6 +148,11 @@ export async function executeAuthoredFlow<Input = undefined>(
148
148
  const onProgress = options.onProgress;
149
149
  const flowPath = options.flowPath ?? join(process.cwd(), 'flow.ts');
150
150
  const waitOptions: RunLifecycleOptions = {
151
+ // Carried so a failed `f.agent` can name the journal that holds its
152
+ // evidence. Each authored worker call runs as its own kernel run, and
153
+ // without the data dir the diagnostic can name the run id but not where
154
+ // on disk to read it.
155
+ ...(options.dataDir !== undefined ? { dataDir: options.dataDir } : {}),
151
156
  ...(options.signal !== undefined ? { signal: options.signal } : {}),
152
157
  ...(options.onWait !== undefined ? { onWait: options.onWait } : {}),
153
158
  };
@@ -3,18 +3,29 @@ import type { PreflightWarning } from './preflight.js';
3
3
  import { hasPricing } from './model-pricing.js';
4
4
  import { resolveAdapterKind } from './adapters/index.js';
5
5
 
6
+ /**
7
+ * The resolved CLI/model pair budget pricing needs, deliberately narrower than
8
+ * `ResolvedCliModel`: pricing depends only on *which* model will run (priced or
9
+ * not) and on the CLI for the Codex wording — not on whether the model came
10
+ * from authoring or an adapter default, which `ResolvedCliModel.source` records
11
+ * for `flows check` reporting. Keeping `source` out means a future resolution
12
+ * rung cannot change pricing by provenance alone.
13
+ */
6
14
  export interface BudgetStepResolution {
7
15
  readonly cli?: string;
8
16
  readonly model?: string;
9
17
  }
10
18
 
11
19
  /**
12
- * Dollar budgets are light enforcement: pricing never refuses a run. A step
13
- * whose model has no frozen price journals zero dollars (see `workerSpend`),
14
- * so it cannot trip `maxDollars`, while its reported tokens still count toward
15
- * any token budget; priced steps still accrue and a crossed limit still stops
16
- * the run in the kernel. This warning names each dollar-unmetered step so the
17
- * gap is reported, not silent.
20
+ * A missing price never refuses a run. Under a frozen dollar budget:
21
+ * - a priced step journals exact dollars and a crossed `maxDollars` stops the
22
+ * run in the kernel;
23
+ * - an unmetered step (no model, or no frozen price) runs and journals its
24
+ * tokens with `dollars_unmetered: true` (see `workerSpend`), so its unknown
25
+ * cost cannot cross `maxDollars` while its tokens still count toward token
26
+ * ceilings.
27
+ * This warning names each unmetered step so the gap is reported, not silent.
28
+ * Preflight `ok` stays true: warnings never refuse.
18
29
  *
19
30
  * Codex selects its own model when none is declared, so a Codex step is
20
31
  * expected to be unmetered and says so rather than asking for a fake price.
package/src/cli/build.ts CHANGED
@@ -62,6 +62,13 @@ export async function runBuild(args: BuildArgs, io: CliIo): Promise<0 | 2> {
62
62
  emitBuildCheckReport(gate.report, args.json, io);
63
63
  return 2;
64
64
  }
65
+ // `ok` means "no refusal", not "no diagnostics". Build defers environment
66
+ // probes, so probe-shaped warnings (`command_unprovable`, ...) are expected
67
+ // here and stay in preflight.json; `budget_unmetered` does not depend on a
68
+ // probe and changes what the dollar budget enforces, so it is printed.
69
+ for (const diagnostic of gate.report.diagnostics) {
70
+ if (diagnostic.kind === 'budget_unmetered') io.stderr(`WARNING [${diagnostic.kind}] ${diagnostic.message}`);
71
+ }
65
72
  io.stdout(await buildFlow(args.value, args.out ?? 'dist/flows', io.stderr));
66
73
  return 0;
67
74
  } catch (error) {
package/src/cli/deploy.ts CHANGED
@@ -31,6 +31,10 @@ export async function runDeploy(args: DeployArgs, io: CliIo): Promise<0 | 1 | 2>
31
31
  for (const diagnostic of checked.report.diagnostics) io.stderr(`${diagnostic.severity.toUpperCase()} [${diagnostic.kind}] ${diagnostic.message}`);
32
32
  return 2;
33
33
  }
34
+ // `ok` may still carry warnings (e.g. `budget_unmetered`); report them.
35
+ for (const diagnostic of checked.report.diagnostics) {
36
+ if (diagnostic.severity === 'warning') io.stderr(`WARNING [${diagnostic.kind}] ${diagnostic.message}`);
37
+ }
34
38
  const target = bucketDirectory(args.to, ref);
35
39
  if (await exists(target)) {
36
40
  await verifyDigest(target, ref.digest);
@@ -14,6 +14,7 @@ import { JournalClient } from '../journal-client.js';
14
14
  import { inputFailureReport } from './check.js';
15
15
  import { checkAuthoredTriggers } from './check-triggers.js';
16
16
  import {
17
+ authoredStepFailure,
17
18
  connect,
18
19
  emptyReport,
19
20
  fromCheckReport,
@@ -190,6 +191,13 @@ export async function runDirectFlow(
190
191
  },
191
192
  };
192
193
  }
194
+ // A step that ran and failed is a run failure, not a protocol failure. The
195
+ // diagnostic carried up from `classifyOutcome` already names the step, its
196
+ // exit code and its output tail; this branch is what lets it reach the
197
+ // terminal. `resumeFlow` takes the same branch, through the same helper.
198
+ if (error instanceof AuthoredFlowExecutionError && error.code === 'step_failed') {
199
+ return authoredStepFailure('run', base, socketPath, error);
200
+ }
193
201
  const runId = error instanceof AuthoredFlowExecutionError ? error.runId : undefined;
194
202
  return protocolFailure('run', base, socketPath, error, runId);
195
203
  } finally {
package/src/cli/run.ts CHANGED
@@ -11,7 +11,7 @@ import { ensureDaemon, type EnsureDaemonOptions } from '../daemon-lifecycle.js';
11
11
  import { isAuthoredFlowPath } from '../direct-input.js';
12
12
  import { daemonRefusal } from './daemon-refusal.js';
13
13
  import type { RunFailureKind, RunWarningKind, StepFailedDetails } from '../failure-kinds.js';
14
- import { deterministicFailureDetails } from './deterministic-failure.js';
14
+ import { inspectionHint, stepFailureDetails } from './step-failure.js';
15
15
  import { JournalClient, JournalProtocolError } from '../journal-client.js';
16
16
  import { attachLocalAgent } from '../local-agent.js';
17
17
  import { LlmWorker } from '../llm-worker.js';
@@ -80,6 +80,14 @@ export interface RunLifecycleOptions {
80
80
  localAgent?: boolean;
81
81
  signal?: AbortSignal;
82
82
  onWait?: (progress: RunProgress) => void;
83
+ /**
84
+ * The daemon data dir, carried so a failure diagnostic can name the journal
85
+ * holding the evidence (`<dataDir>/runs/<runId>.sqlite3`) and emit a
86
+ * `flows replay` invocation that will actually resolve. Each verb sets it
87
+ * from its own `--data-dir`; absent only where no data dir exists, and the
88
+ * diagnostic then omits the path rather than guessing one.
89
+ */
90
+ dataDir?: string;
83
91
  /**
84
92
  * Attach-or-spawn policy for the daemon this command needs
85
93
  * (kernel/DAEMON-LIFECYCLE.md §3). `{ spawn: false }` is `--no-spawn`:
@@ -123,7 +131,7 @@ async function executeCheckedFlow(
123
131
  // advertises its existing pins; the daemon still owns surface matching.
124
132
  if (options.localAgent) localAgent = await attachLocalAgent(client, dataDir, options.onPtyReady);
125
133
  const outcome = await client.runStart(spec, options.reuseFromRunId);
126
- const execution = await classifyOutcome(client, 'run', outcome, base, socketPath, options);
134
+ const execution = await classifyOutcome(client, 'run', outcome, base, socketPath, { ...options, dataDir });
127
135
  if (options.reuseFromRunId !== undefined) {
128
136
  execution.report.reuse = await reuseSummary(client, outcome.run_id, options.reuseFromRunId);
129
137
  }
@@ -207,7 +215,7 @@ export async function resumeFlow(
207
215
  if (await resumeHelperEffect(client, runId, dataDir)) {
208
216
  outcome = await client.runResume(runId, options.allowHumanInfluenced);
209
217
  }
210
- return await classifyOutcome(client, 'resume', outcome, base, socketPath, options);
218
+ return await classifyOutcome(client, 'resume', outcome, base, socketPath, { ...options, dataDir });
211
219
  } catch (error) {
212
220
  if (error instanceof JournalProtocolError && error.code === 'human_influenced_run') {
213
221
  return { exitCode: 2, report: { ...base, runId, socketPath,
@@ -223,6 +231,14 @@ export async function resumeFlow(
223
231
  return { exitCode: 2, report: { ...base, runId, socketPath,
224
232
  diagnostics: [{ severity: 'refusal', kind: error.code, message: error.message }] } };
225
233
  }
234
+ // Same classification the run path gets. Resuming an authored root whose
235
+ // `f.agent` step failed is a step failure, not a protocol failure, and
236
+ // leaving it on `protocolFailure` meant `flows run` printed the evidence
237
+ // while `flows resume` still printed `protocol_error` and
238
+ // `RUN <id> unknown` for the identical failure.
239
+ if (error instanceof AuthoredFlowExecutionError && error.code === 'step_failed') {
240
+ return authoredStepFailure('resume', base, socketPath, error, runId);
241
+ }
226
242
  if (!(error instanceof JournalProtocolError) || error.code !== 'run_not_found') {
227
243
  return protocolFailure('resume', base, socketPath, error, runId);
228
244
  }
@@ -250,6 +266,49 @@ export async function resumeFlow(
250
266
  }
251
267
  }
252
268
 
269
+ /**
270
+ * A step that ran and failed, reported as the run failure it is.
271
+ *
272
+ * Shared by `runDirectFlow` and `resumeFlow` on purpose. The first cut of this
273
+ * fix classified the run path and left resume on `protocolFailure`, so `flows
274
+ * run` became diagnosable while `flows resume` still printed `protocol_error`
275
+ * and `RUN <id> unknown` for the same failed step — and the surface doc claimed
276
+ * both were fixed. One function is what stops the two paths drifting again.
277
+ *
278
+ * `protocolFailure` is wrong here twice over: it blames the daemon for a run it
279
+ * drove correctly, and it produces a report with no `status`, which is the
280
+ * whole of what `RUN <id> unknown` ever meant.
281
+ */
282
+ export function authoredStepFailure(
283
+ command: RunCommand,
284
+ base: CheckReport | RunReport,
285
+ socketPath: string,
286
+ error: AuthoredFlowExecutionError,
287
+ fallbackRunId?: string,
288
+ ): RunExecution {
289
+ // The failing step runs as its own kernel run, so the error's run id is the
290
+ // one whose journal holds the evidence. The resume target is the fallback.
291
+ const runId = error.runId ?? fallbackRunId;
292
+ return {
293
+ exitCode: 1,
294
+ report: {
295
+ ...fromBase(command, base),
296
+ ok: false,
297
+ ...(runId === undefined ? {} : { runId }),
298
+ socketPath,
299
+ status: 'failed',
300
+ completionReason: 'step_failed',
301
+ diagnostics: [...base.diagnostics, {
302
+ severity: 'failure',
303
+ kind: 'step_failed',
304
+ // The `step_failed: ` prefix `AuthoredFlowExecutionError` adds is
305
+ // redundant once the diagnostic is labelled `[step_failed]`.
306
+ message: error.message.replace(/^step_failed: /, ''),
307
+ }],
308
+ },
309
+ };
310
+ }
311
+
253
312
  /**
254
313
  * Get a live daemon, then open the socket to it.
255
314
  *
@@ -409,18 +468,24 @@ export async function classifyOutcome(
409
468
  message: `Run "${current.run_id}" failed with completionReason: ${current.completion_reason}.`,
410
469
  };
411
470
  if (current.completion_reason === 'step_failed') {
471
+ let details: StepFailedDetails | undefined;
412
472
  try {
413
- const details = await deterministicFailureDetails(client, current.run_id);
414
- if (details !== undefined) {
415
- Object.assign(diagnostic, details);
416
- diagnostic.message += ` Step ${JSON.stringify(details.stepId)} exit=${details.exitCode}.`
417
- + (details.stderrTail ? `\nStderr (last 1,024 bytes):\n${details.stderrTail}` : '')
418
- + `\nInspect: ${details.hint}`;
419
- }
473
+ details = await stepFailureDetails(client, current.run_id);
420
474
  } catch (error) {
421
475
  // Inspection must not erase the already known run failure.
422
- diagnostic.message += ` Could not inspect command failure: ${errorMessage(error)}`;
476
+ diagnostic.message += ` Could not inspect the failed step: ${errorMessage(error)}`;
423
477
  }
478
+ if (details !== undefined) {
479
+ Object.assign(diagnostic, details);
480
+ diagnostic.message += renderStepEvidence(details);
481
+ }
482
+ // Appended whatever the inspection found — including nothing. A failure
483
+ // shape this reader does not recognise, or a journal it could not read,
484
+ // must still end with somewhere to go rather than with a dead end.
485
+ const where = inspectionHint(current.run_id, details?.stepId, options.dataDir);
486
+ Object.assign(diagnostic, where);
487
+ diagnostic.message += `\nInspect: ${where.hint}`
488
+ + (where.journalPath === undefined ? '' : `\nJournal: ${where.journalPath}`);
424
489
  }
425
490
  return {
426
491
  exitCode: 1,
@@ -609,6 +674,24 @@ function errorMessage(error: unknown): string {
609
674
  return error instanceof Error ? error.message : 'unknown protocol error';
610
675
  }
611
676
 
677
+ /**
678
+ * The evidence half of a `step_failed` diagnostic; `inspectionHint` adds the
679
+ * rest. Each field is printed only when the journal actually carried it — an
680
+ * agent step has no exit code to report, and inventing `exit=undefined` (which
681
+ * is what the deterministic-only version printed for one) is worse than
682
+ * silence on that field.
683
+ */
684
+ function renderStepEvidence(details: StepFailedDetails): string {
685
+ return ` Step ${JSON.stringify(details.stepId)}`
686
+ + (details.stepType === undefined ? '' : ` (${details.stepType})`)
687
+ + ` completionReason: ${details.completionReason}`
688
+ + (details.exitCode === undefined ? '' : ` exit=${details.exitCode}`)
689
+ + '.'
690
+ + (details.detail === undefined ? '' : `\nDetail: ${details.detail}`)
691
+ + (details.stdoutTail ? `\nStdout (last 1,024 bytes):\n${details.stdoutTail}` : '')
692
+ + (details.stderrTail ? `\nStderr (last 1,024 bytes):\n${details.stderrTail}` : '');
693
+ }
694
+
612
695
  function throwIfCanceled(signal: AbortSignal | undefined, stepId: string): void {
613
696
  if (signal?.aborted === true) throw new Error(`waiting for running step "${stepId}" was canceled`);
614
697
  }
@@ -0,0 +1,160 @@
1
+ import { join } from 'node:path';
2
+ import { DEFAULT_DATA_DIR } from '../daemon-connection.js';
3
+ import type { StepFailedDetails } from '../failure-kinds.js';
4
+ import type { JournalClient } from '../journal-client.js';
5
+
6
+ const TAIL_BYTES = 1_024;
7
+
8
+ /**
9
+ * Read what a failed step left in the journal — for any step type.
10
+ *
11
+ * This was `deterministicFailureDetails`, and its first act was to return
12
+ * `undefined` for any run without a `deterministic` step. That is every
13
+ * `f.agent` and `f.llm` step, because each authored worker call runs as its
14
+ * own single-step kernel run (authored-worker-step.ts). So a failed agent
15
+ * reached the terminal carrying nothing but its taxonomy label — the run said
16
+ * `step_failed` and discarded every account of why, which is the whole reason
17
+ * a local agent failure was undiagnosable.
18
+ *
19
+ * The two step families leave their evidence in different fields, because the
20
+ * kernel preserves `output` on a failed completion only for deterministic
21
+ * steps (`preserve_failure_output`, relayflowd-core/src/machine.rs). For an
22
+ * agent or llm step the worker's `{exit_code, stdout_tail, stderr_tail}` is
23
+ * nulled out of `output` and survives only as the bounded render the daemon
24
+ * captured into `verification.detail` (`worker_failure_detail`,
25
+ * relayflowd/src/engine/remote.rs). Both are read, in that order, and the
26
+ * daemon's render is re-parsed when it carries that same shape: an exit code
27
+ * the daemon stringified on its way into the journal is still an exit code,
28
+ * and printing it as one is the difference between a diagnosis and a blob.
29
+ */
30
+ export async function stepFailureDetails(
31
+ client: JournalClient,
32
+ runId: string,
33
+ ): Promise<StepFailedDetails | undefined> {
34
+ const snapshot = await client.runGet(runId);
35
+ let fromSeq = 1;
36
+ const failures = new Map<string, StepFailedDetails>();
37
+ while (true) {
38
+ const { entries } = await client.journalRead(runId, fromSeq, 100);
39
+ if (entries.length === 0) break;
40
+ for (const raw of entries) {
41
+ const entry = record(raw);
42
+ const seq = entry?.['seq'];
43
+ if (typeof seq !== 'number' || !Number.isSafeInteger(seq) || seq < fromSeq) {
44
+ throw new Error('invalid journal sequence in step failure inspection');
45
+ }
46
+ fromSeq = seq + 1;
47
+ const stepId = entry?.['step_id'];
48
+ if (entry?.['entry_type'] !== 'step.completed' || typeof stepId !== 'string') continue;
49
+ // A later completion supersedes an earlier failed attempt.
50
+ failures.delete(stepId);
51
+ const payload = record(entry['payload']);
52
+ if (payload === undefined) continue;
53
+ const completionReason = payload['completionReason'];
54
+ // A terminal completion that is not a success is the failure, whatever
55
+ // its step type. The old predicate also demanded a non-zero `exit_code`,
56
+ // which no agent completion carries and which a deterministic step that
57
+ // exits 0 and then fails its gate does not carry either — both were
58
+ // silently skipped.
59
+ if (payload['disposition'] !== 'step_done'
60
+ || typeof completionReason !== 'string' || completionReason === 'success') continue;
61
+ const stepType = snapshot.steps[stepId]?.type;
62
+ failures.set(stepId, {
63
+ stepId,
64
+ completionReason,
65
+ ...(stepType === undefined ? {} : { stepType }),
66
+ ...evidence(payload),
67
+ });
68
+ }
69
+ }
70
+ return [...failures.values()].at(-1);
71
+ }
72
+
73
+ /**
74
+ * Where to look, derived from the run id and data dir ALONE.
75
+ *
76
+ * Deliberately independent of the journal read: a failure whose evidence could
77
+ * not be read, or a failure shape nothing here recognises, must still end with
78
+ * somewhere to go rather than with a dead end. `flows replay` is that command —
79
+ * it already exists and already prints the full journal; nothing ever named it
80
+ * at the moment of failure, which is why the surface looked like it had no way
81
+ * to inspect a finished run.
82
+ */
83
+ export function inspectionHint(
84
+ runId: string,
85
+ stepId: string | undefined,
86
+ dataDir: string | undefined,
87
+ ): { hint: string; journalPath?: string } {
88
+ const at = stepId === undefined ? '' : ` --at ${shellQuote(stepId)}`;
89
+ // Only name a non-default data dir: repeating the default back at an
90
+ // operator who never typed it is noise, and `flows replay` defaults to the
91
+ // same value (cli.ts).
92
+ const dir = dataDir === undefined || dataDir === DEFAULT_DATA_DIR
93
+ ? '' : ` --data-dir ${shellQuote(dataDir)}`;
94
+ return {
95
+ hint: `flows replay ${shellQuote(runId)}${at}${dir}`,
96
+ // Left as the operator spelled it rather than resolved: `.relayflowd/...`
97
+ // is what they will recognise in their own working directory.
98
+ ...(dataDir === undefined ? {} : { journalPath: join(dataDir, 'runs', `${runId}.sqlite3`) }),
99
+ };
100
+ }
101
+
102
+ /**
103
+ * Pull the process-shaped fields out of whichever field carried them.
104
+ *
105
+ * `verification.detail` is last because it is the daemon's own render rather
106
+ * than the worker's structured report — but for an agent step it is the only
107
+ * thing that survives, so it is parsed when it parses and kept verbatim when
108
+ * it does not. A truncated render (the daemon caps at 2,000 chars and appends
109
+ * a truncation note) will not parse; that falls through to the raw string,
110
+ * which is still the account of what went wrong.
111
+ */
112
+ function evidence(payload: Record<string, unknown>): Partial<StepFailedDetails> {
113
+ const detail = record(payload['verification'])?.['detail'];
114
+ const structured = record(payload['output'])
115
+ ?? record(payload['trajectory_tail'])
116
+ ?? (typeof detail === 'string' ? parsed(detail) : undefined);
117
+ const exitCode = structured?.['exit_code'];
118
+ const stdout = structured?.['stdout_tail'];
119
+ const stderr = structured?.['stderr_tail'];
120
+ const structuredShape = typeof exitCode === 'number'
121
+ || typeof stdout === 'string' || typeof stderr === 'string';
122
+ return {
123
+ ...(typeof exitCode === 'number' && Number.isSafeInteger(exitCode) ? { exitCode } : {}),
124
+ ...(typeof stdout === 'string' && stdout.length > 0 ? { stdoutTail: tail(stdout) } : {}),
125
+ ...(typeof stderr === 'string' ? { stderrTail: tail(stderr) } : {}),
126
+ // Keep the daemon's account only when it was NOT just a render of the
127
+ // fields above — otherwise the same bytes print twice.
128
+ ...(typeof detail === 'string' && detail.length > 0 && !structuredShape
129
+ ? { detail: tail(detail) } : {}),
130
+ };
131
+ }
132
+
133
+ function parsed(value: string): Record<string, unknown> | undefined {
134
+ try {
135
+ return record(JSON.parse(value));
136
+ } catch {
137
+ return undefined;
138
+ }
139
+ }
140
+
141
+ function record(value: unknown): Record<string, unknown> | undefined {
142
+ return value !== null && typeof value === 'object' && !Array.isArray(value)
143
+ ? value as Record<string, unknown> : undefined;
144
+ }
145
+
146
+ function tail(value: string): string {
147
+ const bytes = Buffer.from(value, 'utf8');
148
+ let start = Math.max(0, bytes.length - TAIL_BYTES);
149
+ // Drop a partial leading code point, avoiding replacement-byte expansion.
150
+ while (start < bytes.length && (bytes[start]! & 0xc0) === 0x80) start += 1;
151
+ // Preserve tabs/newlines; replace binary controls (including ESC and CR),
152
+ // C1 controls and Unicode formatting controls without growing the excerpt.
153
+ return bytes.subarray(start).toString('utf8')
154
+ .replace(/[\p{Cc}\p{Cf}]/gu, character => character === '\n' || character === '\t' ? character : '?');
155
+ }
156
+
157
+ function shellQuote(value: string): string {
158
+ return /^[A-Za-z0-9_-]+$/.test(value) && !value.startsWith('-')
159
+ ? value : `'${value.replace(/'/g, "'\\''")}'`;
160
+ }