@oxygen-agent/cli 1.906.0 → 1.922.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. package/README.md +1 -1
  2. package/dist/command-manifest.js +13 -1
  3. package/dist/index.js +337 -40
  4. package/node_modules/@oxygen/formula/dist/expression.d.ts +21 -0
  5. package/node_modules/@oxygen/formula/dist/expression.js +42 -1
  6. package/node_modules/@oxygen/formula/dist/formula-functions.d.ts +1 -1
  7. package/node_modules/@oxygen/formula/dist/formula-functions.js +10 -1
  8. package/node_modules/@oxygen/shared/dist/billing.d.ts +17 -0
  9. package/node_modules/@oxygen/shared/dist/billing.js +20 -0
  10. package/node_modules/@oxygen/shared/dist/capability-discovery.js +12 -7
  11. package/node_modules/@oxygen/shared/dist/copilot-errors.d.ts +1 -0
  12. package/node_modules/@oxygen/shared/dist/copilot-errors.js +9 -0
  13. package/node_modules/@oxygen/shared/dist/copilot-plan.d.ts +39 -0
  14. package/node_modules/@oxygen/shared/dist/copilot-plan.js +76 -7
  15. package/node_modules/@oxygen/shared/dist/langfuse.d.ts +36 -0
  16. package/node_modules/@oxygen/shared/dist/langfuse.js +83 -0
  17. package/node_modules/@oxygen/shared/dist/linkedin-quota-denial.d.ts +1 -1
  18. package/node_modules/@oxygen/shared/dist/linkedin-quota-denial.js +16 -5
  19. package/node_modules/@oxygen/shared/dist/linkedin-url.d.ts +22 -0
  20. package/node_modules/@oxygen/shared/dist/linkedin-url.js +104 -0
  21. package/node_modules/@oxygen/shared/dist/provider-funding-errors.d.ts +16 -2
  22. package/node_modules/@oxygen/shared/dist/provider-funding-errors.js +19 -9
  23. package/node_modules/@oxygen/shared/dist/table-limits.d.ts +18 -0
  24. package/node_modules/@oxygen/shared/dist/table-limits.js +18 -0
  25. package/node_modules/@oxygen/shared/dist/user-capability-routing.js +23 -2
  26. package/node_modules/@oxygen/shared/dist/version.d.ts +1 -1
  27. package/node_modules/@oxygen/shared/dist/version.js +1 -1
  28. package/node_modules/@oxygen/shared/package.json +10 -0
  29. package/package.json +2 -1
@@ -41,6 +41,27 @@ export type ParseOptions = {
41
41
  export declare function parseFormulaExpression(expression: string, options?: ParseOptions): FormulaAst;
42
42
  /** Validate supported functions and arity on an already-parsed tree. */
43
43
  export declare function validateFormulaAst(ast: FormulaAst): void;
44
+ /**
45
+ * SAVE-TIME ONLY: reject a literal regex pattern that is too long to ever compile.
46
+ *
47
+ * Deliberately NOT part of validateFormulaAst. That runs on the READ path too
48
+ * (formula-column-runner prepares every batch through it) and
49
+ * prepareFormulaForRead swallows the throw and nulls the WHOLE column, so
50
+ * tightening it would silently blank already-stored columns rather than reject
51
+ * the edit. Keeping this save-only means a column that stores today keeps
52
+ * reading today, and only a NEW write is refused.
53
+ *
54
+ * Length is checked, not compilation. A pattern over the cap can never succeed in
55
+ * any branch, so flagging it costs the author nothing -- whereas rejecting an
56
+ * unparseable literal would break `if(cond, regex_match(x, "["), "n/a")`, which
57
+ * works today precisely because if/switch/and/or/coalesce are lazy and never
58
+ * evaluate the untaken side. That stricter semantic is a separate decision.
59
+ *
60
+ * Without this, an over-long pattern stored fine and then failed per row on every
61
+ * run forever -- the production shape was one column failing across dozens of rows
62
+ * for days with received_length 290 against the old 256 cap.
63
+ */
64
+ export declare function validateFormulaRegexLiterals(ast: FormulaAst): void;
44
65
  /** Validate syntax, supported functions, and arity without evaluating any data. */
45
66
  export declare function validateFormulaExpression(expression: string, options?: ParseOptions): void;
46
67
  export declare function walkExpression(node: FormulaAst, visit: (node: FormulaAst) => void): void;
@@ -1,4 +1,4 @@
1
- import { FORMULA_FUNCTION_REGISTRY, checkFormulaFunctionArity, formulaExpressionError, } from "./formula-functions.js";
1
+ import { FORMULA_FUNCTION_REGISTRY, MAX_FORMULA_REGEX_PATTERN_LENGTH, checkFormulaFunctionArity, formulaExpressionError, } from "./formula-functions.js";
2
2
  /**
3
3
  * Functions evaluated against the whole column rather than the current scope.
4
4
  * They are intercepted before registry dispatch (the registry entry carries
@@ -37,6 +37,47 @@ export function validateFormulaAst(ast) {
37
37
  checkFormulaFunctionArity(spec, node.args.length);
38
38
  });
39
39
  }
40
+ /**
41
+ * SAVE-TIME ONLY: reject a literal regex pattern that is too long to ever compile.
42
+ *
43
+ * Deliberately NOT part of validateFormulaAst. That runs on the READ path too
44
+ * (formula-column-runner prepares every batch through it) and
45
+ * prepareFormulaForRead swallows the throw and nulls the WHOLE column, so
46
+ * tightening it would silently blank already-stored columns rather than reject
47
+ * the edit. Keeping this save-only means a column that stores today keeps
48
+ * reading today, and only a NEW write is refused.
49
+ *
50
+ * Length is checked, not compilation. A pattern over the cap can never succeed in
51
+ * any branch, so flagging it costs the author nothing -- whereas rejecting an
52
+ * unparseable literal would break `if(cond, regex_match(x, "["), "n/a")`, which
53
+ * works today precisely because if/switch/and/or/coalesce are lazy and never
54
+ * evaluate the untaken side. That stricter semantic is a separate decision.
55
+ *
56
+ * Without this, an over-long pattern stored fine and then failed per row on every
57
+ * run forever -- the production shape was one column failing across dozens of rows
58
+ * for days with received_length 290 against the old 256 cap.
59
+ */
60
+ export function validateFormulaRegexLiterals(ast) {
61
+ walkExpression(ast, (node) => {
62
+ if (node.type !== "call")
63
+ return;
64
+ const normalized = node.name.toLowerCase();
65
+ if (normalized !== "regex_match" && normalized !== "regex_extract" && normalized !== "regex_replace") {
66
+ return;
67
+ }
68
+ // Every regex function takes its pattern as the second argument.
69
+ const pattern = node.args[1];
70
+ if (!pattern || pattern.type !== "literal" || typeof pattern.value !== "string")
71
+ return;
72
+ if (pattern.value.length <= MAX_FORMULA_REGEX_PATTERN_LENGTH)
73
+ return;
74
+ throw formulaExpressionError("Regular-expression pattern is too long.", {
75
+ function: node.name,
76
+ max_length: MAX_FORMULA_REGEX_PATTERN_LENGTH,
77
+ received_length: pattern.value.length,
78
+ });
79
+ });
80
+ }
40
81
  /** Validate syntax, supported functions, and arity without evaluating any data. */
41
82
  export function validateFormulaExpression(expression, options = {}) {
42
83
  validateFormulaAst(parseFormulaExpression(expression, options));
@@ -37,7 +37,7 @@ export type FormulaFunctionSpec = FormulaFunctionMeta & {
37
37
  evaluate?: (args: unknown[]) => unknown;
38
38
  evaluateLazy?: (thunks: Array<() => unknown>) => unknown;
39
39
  };
40
- export declare const MAX_FORMULA_REGEX_PATTERN_LENGTH = 256;
40
+ export declare const MAX_FORMULA_REGEX_PATTERN_LENGTH = 1000;
41
41
  export declare const MAX_FORMULA_REGEX_INPUT_LENGTH = 20000;
42
42
  export declare function formulaExpressionError(message: string, details: Record<string, unknown>): OxygenError;
43
43
  export declare function isBlankFormulaValue(value: unknown): boolean;
@@ -1,7 +1,16 @@
1
1
  import { OxygenError } from "@oxygen/shared/cli-result";
2
2
  import { walkJsonPath } from "@oxygen/shared/json-path";
3
3
  import { normalizeDomain, normalizeEmail, normalizeLinkedinUrl, } from "./value-normalizers.js";
4
- export const MAX_FORMULA_REGEX_PATTERN_LENGTH = 256;
4
+ // Pattern LENGTH is not a safety bound and never was -- catastrophic backtracking
5
+ // is a function of pattern SHAPE, and `(a+)+$` hangs at 6 characters, so 256 already
6
+ // permitted an unbounded hang while rejecting harmless long literals. The real bound
7
+ // on scan cost is MAX_FORMULA_REGEX_INPUT_LENGTH below, which caps what a pattern is
8
+ // run against. 256 rejected a customer's 290-character alternation on every row of
9
+ // their table, continuously, for days (production 2026-08-28/31). Raised to a value
10
+ // that still stops a pathological paste while leaving ordinary generated alternations
11
+ // room. If real ReDoS protection is ever needed it belongs in the engine (a
12
+ // backtracking budget or re2), not in a character count.
13
+ export const MAX_FORMULA_REGEX_PATTERN_LENGTH = 1_000;
5
14
  export const MAX_FORMULA_REGEX_INPUT_LENGTH = 20_000;
6
15
  // ---------------------------------------------------------------------------
7
16
  // Shared coercion helpers (used by the registry AND the runner's operators)
@@ -33,6 +33,23 @@ export declare const CREDIT_TOPUP_MAX_CREDITS = 1000000;
33
33
  export declare const CREDIT_TOPUP_STEP_CREDITS = 1000;
34
34
  export declare const CREDIT_TOPUP_DEFAULT_CREDITS = 20000;
35
35
  export declare function isValidCreditTopupCredits(credits: number): boolean;
36
+ /**
37
+ * A shortfall expressed as an amount `oxygen billing topup` will actually SELL:
38
+ * rounded up onto the 1,000-credit step and clamped into the purchasable
39
+ * [8,000 .. 1,000,000] band. A next_action (or an alert email) naming a number the
40
+ * top-up route refuses with `invalid_topup_amount` is not a next action, it is a
41
+ * second dead end during the incident it exists to end.
42
+ *
43
+ * Both clamps are honest, not cosmetic: below the floor the customer buys the
44
+ * smallest pack (more than the debt — purchased credits never expire), and above
45
+ * the ceiling they buy the largest single checkout and top up again. Callers carry
46
+ * the raw shortfall alongside so either gap stays visible rather than implied.
47
+ *
48
+ * Shared rather than re-derived per surface: `billing balance` and the renewal
49
+ * failure alert quote this number to the same customer about the same debt, and a
50
+ * second rounding formula is how two customer-facing figures drift apart.
51
+ */
52
+ export declare function creditTopupAmountForShortfall(credits: number): number;
36
53
  export declare function creditTopupUsdCents(credits: number): number | null;
37
54
  export declare const AUTOMATION_ACTION_CREDITS = 0.01;
38
55
  /** Kinds of resource that carry a fixed monthly credit commitment. */
@@ -35,6 +35,26 @@ export function isValidCreditTopupCredits(credits) {
35
35
  && credits <= CREDIT_TOPUP_MAX_CREDITS
36
36
  && credits % CREDIT_TOPUP_STEP_CREDITS === 0;
37
37
  }
38
+ /**
39
+ * A shortfall expressed as an amount `oxygen billing topup` will actually SELL:
40
+ * rounded up onto the 1,000-credit step and clamped into the purchasable
41
+ * [8,000 .. 1,000,000] band. A next_action (or an alert email) naming a number the
42
+ * top-up route refuses with `invalid_topup_amount` is not a next action, it is a
43
+ * second dead end during the incident it exists to end.
44
+ *
45
+ * Both clamps are honest, not cosmetic: below the floor the customer buys the
46
+ * smallest pack (more than the debt — purchased credits never expire), and above
47
+ * the ceiling they buy the largest single checkout and top up again. Callers carry
48
+ * the raw shortfall alongside so either gap stays visible rather than implied.
49
+ *
50
+ * Shared rather than re-derived per surface: `billing balance` and the renewal
51
+ * failure alert quote this number to the same customer about the same debt, and a
52
+ * second rounding formula is how two customer-facing figures drift apart.
53
+ */
54
+ export function creditTopupAmountForShortfall(credits) {
55
+ const stepped = Math.ceil(credits / CREDIT_TOPUP_STEP_CREDITS) * CREDIT_TOPUP_STEP_CREDITS;
56
+ return Math.min(CREDIT_TOPUP_MAX_CREDITS, Math.max(CREDIT_TOPUP_MIN_CREDITS, stepped));
57
+ }
38
58
  export function creditTopupUsdCents(credits) {
39
59
  if (!isValidCreditTopupCredits(credits))
40
60
  return null;
@@ -21,15 +21,20 @@ export const OXYGEN_CAPABILITY_ROUTES = [
21
21
  id: "discovery-and-skills",
22
22
  layer: "Control",
23
23
  primitive: null,
24
- owns: "Bounded capability, command, provider-operation, Recipe, and product-skill discovery with exact hydration on demand.",
24
+ owns: "Bounded capability, command, provider-operation, Recipe, and product-skill discovery, plus INSTANCE discovery — what this workspace actually holds — with exact hydration on demand.",
25
25
  notFor: "Executing GTM work or loading full manifests and schemas before a capability is selected.",
26
- execution: "Search by outcome, choose the owner, then hydrate one exact command, MCP schema, provider descriptor, or skill.",
26
+ execution: "Search by outcome, choose the owner, then hydrate one exact command, MCP schema, provider descriptor, or skill. For instance discovery, read the map and then the primitive that owns the row.",
27
27
  posture: "read_only",
28
- gatewayTools: ["oxygen_capabilities_search", "oxygen_tools_search", "oxygen_recipes_list"],
29
- gatewayCommands: ["capabilities search", "commands search", "skills search", "tools search", "recipes list"],
28
+ gatewayTools: ["oxygen_capabilities_search", "oxygen_tools_search", "oxygen_recipes_list", "oxygen_workspace_map"],
29
+ gatewayCommands: ["capabilities search", "commands search", "skills search", "tools search", "recipes list", "workspace map"],
30
30
  skills: ["oxygen-quickstart", "oxygen-gtm"],
31
- endpointSections: ["skills"],
32
- intentTerms: ["discover", "discovery", "capability", "capabilities", "command", "commands", "skill", "skills", "which tool", "how do i"],
31
+ // `workspace` (the map) belongs here rather than beside `home`: capability
32
+ // discovery answers what OXYGEN can do and instance discovery answers what THIS
33
+ // workspace holds, and they are the same act — bounded, read-only, hydrate one
34
+ // thing on demand. `home` stayed on the onboarding card because the standup is a
35
+ // first-run surface; the map is not.
36
+ endpointSections: ["skills", "workspace"],
37
+ intentTerms: ["discover", "discovery", "capability", "capabilities", "command", "commands", "skill", "skills", "which tool", "how do i", "what do i have", "what is in my workspace", "workspace map"],
33
38
  },
34
39
  {
35
40
  id: "onboarding-and-copilot",
@@ -196,7 +201,7 @@ export const OXYGEN_CAPABILITY_ROUTES = [
196
201
  gatewayTools: ["oxygen_tables_create", "oxygen_columns_add", "oxygen_enrich_column_preview", "oxygen_tables_link_bulk"],
197
202
  gatewayCommands: ["tables create", "columns add", "enrich-column preview", "tables link"],
198
203
  skills: ["oxygen-gtm", "oxygen-table-tidy", "oxygen-diagnostics", "oxygen-clay-migration"],
199
- endpointSections: ["action-columns", "callables", "columns", "company-enrichment", "enrich-column", "enrichment", "projects", "table-action-runs", "table-ingestion-runs", "tables"],
204
+ endpointSections: ["action-columns", "callables", "columns", "company-enrichment", "enrich-column", "enrichment", "projects", "table-action-items", "table-action-runs", "table-ingestion-runs", "tables"],
200
205
  // "link"/"join"/"connect"/"relate" route here for `tables link`. Added after a
201
206
  // blind user eval asked for exactly "link two tables" and was routed to
202
207
  // `tables create` / `columns add` / `enrich-column preview` — none of which
@@ -1,5 +1,6 @@
1
1
  export declare const COPILOT_TURN_TIMEOUT_CODE = "copilot_turn_timeout";
2
2
  export declare const COPILOT_TURN_TIMEOUT_MESSAGE = "This Copilot request timed out and stopped. Any completed actions are still saved\u2014review this session before trying again.";
3
+ export declare const COPILOT_TURN_DEADLINE_EXCEEDED_CODE = "copilot_turn_deadline_exceeded";
3
4
  export type CustomerFacingCopilotError = {
4
5
  code: string | null;
5
6
  message: string | null;
@@ -4,6 +4,14 @@
4
4
  // and already-persisted failures serialize identically across web, CLI, and MCP.
5
5
  export const COPILOT_TURN_TIMEOUT_CODE = "copilot_turn_timeout";
6
6
  export const COPILOT_TURN_TIMEOUT_MESSAGE = "This Copilot request timed out and stopped. Any completed actions are still saved—review this session before trying again.";
7
+ // A turn the worker refused to start because it was ALREADY past its stamped
8
+ // wall-clock deadline when it was reclaimed. Distinct in the durable row from
9
+ // `copilot_turn_timeout` (a turn that ran and then ran out of clock) so an
10
+ // operator can separate "we never started it" from "we could not finish it" in
11
+ // SQL — but deliberately NOT distinct to the customer: the mapping below folds
12
+ // it into the same timeout contract every Control surface already renders, per
13
+ // this module's policy.
14
+ export const COPILOT_TURN_DEADLINE_EXCEEDED_CODE = "copilot_turn_deadline_exceeded";
7
15
  const WORKER_STEP_TIMEOUT_MESSAGE = /\bWorker step '[^']+' exceeded \d+ms deadline\.?/i;
8
16
  /**
9
17
  * Replace an internal worker deadline with the stable Copilot timeout contract.
@@ -15,6 +23,7 @@ export function customerFacingCopilotError(input) {
15
23
  const message = input.message ?? null;
16
24
  const isTimeout = code === "worker_step_timeout" ||
17
25
  code === COPILOT_TURN_TIMEOUT_CODE ||
26
+ code === COPILOT_TURN_DEADLINE_EXCEEDED_CODE ||
18
27
  (message !== null && WORKER_STEP_TIMEOUT_MESSAGE.test(message));
19
28
  return isTimeout
20
29
  ? { code: COPILOT_TURN_TIMEOUT_CODE, message: COPILOT_TURN_TIMEOUT_MESSAGE }
@@ -109,6 +109,8 @@ export type CopilotPlanProjection = {
109
109
  /** True when the turn's wall-clock deadline, not the plan, is the binding bound. */
110
110
  deadlineBound: boolean;
111
111
  };
112
+ /** False when no turn is executing: the plan is history, not work in flight. */
113
+ turnActive: boolean;
112
114
  /** Ledger position of the newest plan_updated, so a caller can tell staleness. */
113
115
  updatedAtSeq: number;
114
116
  updatedAt: string;
@@ -124,6 +126,20 @@ export type ProjectCopilotPlanInput = {
124
126
  events: CopilotPlanSourceEvent[];
125
127
  /** The active turn's wall-clock deadline, used only to clamp the total. */
126
128
  turnDeadlineAt?: string | Date | null;
129
+ /**
130
+ * Whether a turn is executing right now. Defaults to true.
131
+ *
132
+ * A plan OUTLIVES the turn that wrote it. The model routinely stops without
133
+ * marking its last step done, so the session rests in the ledger forever with a
134
+ * step still `in_progress`. Read against a running clock that step accrues
135
+ * elapsed time indefinitely -- production session fd259942 reached 37 DAYS --
136
+ * and the surface goes on offering "time remaining" for work that stopped weeks
137
+ * ago. Both are the same error: treating a historical record as work in flight.
138
+ *
139
+ * So the clock stops with the turn, and a plan nobody is working reports no
140
+ * estimate at all rather than a false one.
141
+ */
142
+ turnActive?: boolean;
127
143
  now?: Date;
128
144
  };
129
145
  /**
@@ -135,3 +151,26 @@ export type ProjectCopilotPlanInput = {
135
151
  * would be worse than one that stays away.
136
152
  */
137
153
  export declare function projectCopilotPlan(input: ProjectCopilotPlanInput): CopilotPlanProjection | null;
154
+ /**
155
+ * "<capability>: <gist>" for a failed call. The gist is the summary cut at its
156
+ * first sentence or clause boundary and then hard-capped, so a schema dump or a
157
+ * multi-sentence explanation cannot become the title; a summary that is only
158
+ * the capability's own name, or empty, leaves the bare name.
159
+ */
160
+ export declare function failedToolSubstepTitle(capability: string, summary: string): string;
161
+ /**
162
+ * How long, in words. One owner, because there were three and all three were wrong.
163
+ *
164
+ * The rail, `oxygen copilot plan` and the MCP widget each carried their own copy
165
+ * of this, and every copy did `Math.floor(s / 60)` minutes with `Math.round(s % 60)`
166
+ * seconds -- which rounds the remainder INDEPENDENTLY of the minutes it was taken
167
+ * from. A 179.6s sub-agent therefore rendered as "2m 60s" on all three surfaces
168
+ * (observed live in session 34826105), and 59.6s rendered as "60s" rather than
169
+ * "1m". Rounding the total FIRST and splitting afterwards cannot produce either.
170
+ *
171
+ * The projection promised one implementation behind three surfaces; the numbers
172
+ * were shared and the words describing them were not, so this is where they meet.
173
+ */
174
+ export declare function formatCopilotPlanSeconds(seconds: number): string;
175
+ /** A duration in ms, or null when there is nothing worth showing. */
176
+ export declare function formatCopilotPlanDuration(ms: number | null | undefined): string | null;
@@ -169,6 +169,13 @@ export function projectCopilotPlan(input) {
169
169
  const planEvents = events.filter((event) => event.kind === "plan_updated");
170
170
  if (planEvents.length === 0)
171
171
  return null;
172
+ // The plan's clock stops when the turn does -- see `turnActive` on the input.
173
+ // `Math.min` because a clock that runs BACKWARDS is worse than one that stops:
174
+ // the last event is normally in the past, but a clock skew must not make an
175
+ // elapsed time negative.
176
+ const turnActive = input.turnActive !== false;
177
+ const lastEvent = events[events.length - 1];
178
+ const clockMs = turnActive || !lastEvent ? nowMs : Math.min(nowMs, toMs(lastEvent.created_at));
172
179
  // -- Pass 1: fold the successive plans into one tracked set -----------------
173
180
  //
174
181
  // Identity is what makes timing possible. Match on `id` first, then on the
@@ -230,7 +237,7 @@ export function projectCopilotPlan(input) {
230
237
  if (tracked.length === 0)
231
238
  return null;
232
239
  // -- Pass 2: attribute observed tool calls to the step that was running -----
233
- attributeToolCalls(events, planEvents, tracked, nowMs);
240
+ attributeToolCalls(events, planEvents, tracked, clockMs);
234
241
  // -- Pass 3: the ETA --------------------------------------------------------
235
242
  const samples = tracked
236
243
  .filter((t) => t.status === "done" && t.startedAtMs !== null && t.endedAtMs !== null)
@@ -250,10 +257,11 @@ export function projectCopilotPlan(input) {
250
257
  // Once a step has started it has an elapsed time: its own end if it finished,
251
258
  // otherwise the clock. A step that never started has none -- which is exactly
252
259
  // why it is also never a measurement sample.
253
- const elapsedMs = entry.startedAtMs === null ? null : (entry.endedAtMs ?? nowMs) - entry.startedAtMs;
260
+ const elapsedMs = entry.startedAtMs === null ? null : (entry.endedAtMs ?? clockMs) - entry.startedAtMs;
254
261
  let etaSeconds = null;
255
262
  let etaBasis = "none";
256
- if (entry.status === "pending" || entry.status === "in_progress") {
263
+ // A step only has time REMAINING while something is running to consume it.
264
+ if (turnActive && (entry.status === "pending" || entry.status === "in_progress")) {
257
265
  let budget = null;
258
266
  if (entry.estimateSeconds !== null && calibration !== null) {
259
267
  budget = entry.estimateSeconds * calibration;
@@ -307,6 +315,7 @@ export function projectCopilotPlan(input) {
307
315
  }
308
316
  const done = steps.filter((step) => step.status === "done").length;
309
317
  return {
318
+ turnActive,
310
319
  summary: summary.length > 0 ? summary : null,
311
320
  steps,
312
321
  progress: { done, total: steps.length, ratio: steps.length === 0 ? 0 : done / steps.length },
@@ -327,7 +336,7 @@ export function projectCopilotPlan(input) {
327
336
  * Attribution uses the plan as of the call's own seq, not the final plan -- a call
328
337
  * made during step 2 belongs to step 2 even after step 4 becomes current.
329
338
  */
330
- function attributeToolCalls(events, planEvents, tracked, nowMs) {
339
+ function attributeToolCalls(events, planEvents, tracked, clockMs) {
331
340
  const byId = new Map(tracked.map((t) => [t.id, t]));
332
341
  const byTitle = new Map(tracked.map((t) => [normalizeTitleKey(t.title), t]));
333
342
  /** Which tracked step was in progress at a given ledger position. */
@@ -387,13 +396,22 @@ function attributeToolCalls(events, planEvents, tracked, nowMs) {
387
396
  if (!target)
388
397
  continue;
389
398
  const rollup = rollups.get(callId);
399
+ const failed = payload.ok === false;
390
400
  target.toolSubsteps.push({
391
401
  id: callId,
392
402
  // `summary` is the tool_call_finished remap the Copilot host writes; it is a
393
403
  // line ("63 columns"), which is what a substep wants. Fall back to the
394
404
  // capability name rather than to a serialized result.
395
- title: readText(payload.summary) || capability,
396
- status: payload.ok === false ? "blocked" : "done",
405
+ //
406
+ // A FAILED call's summary is the error message, which is prose about the
407
+ // failure rather than a name for the step -- a sub-agent's "did not finish
408
+ // (FAILED): This sub-agent run reached its ceiling of 4 model calls."
409
+ // became the substep's whole title. The rail needs what failed and the
410
+ // gist of why, bounded, with the red mark carrying the status.
411
+ title: failed
412
+ ? failedToolSubstepTitle(capability, readText(payload.summary))
413
+ : readText(payload.summary) || capability,
414
+ status: failed ? "blocked" : "done",
397
415
  source: "tool",
398
416
  durationMs: started ? toMs(event.created_at) - started.startedMs : null,
399
417
  startedAt: started ? new Date(started.startedMs).toISOString() : null,
@@ -409,11 +427,30 @@ function attributeToolCalls(events, planEvents, tracked, nowMs) {
409
427
  title: entry.capability,
410
428
  status: "in_progress",
411
429
  source: "tool",
412
- durationMs: nowMs - entry.startedMs,
430
+ durationMs: clockMs - entry.startedMs,
413
431
  startedAt: new Date(entry.startedMs).toISOString(),
414
432
  });
415
433
  }
416
434
  }
435
+ /** How much of a failure's reason a substep title carries before it stops being a label. */
436
+ const FAILED_SUBSTEP_REASON_CHARS = 90;
437
+ /**
438
+ * "<capability>: <gist>" for a failed call. The gist is the summary cut at its
439
+ * first sentence or clause boundary and then hard-capped, so a schema dump or a
440
+ * multi-sentence explanation cannot become the title; a summary that is only
441
+ * the capability's own name, or empty, leaves the bare name.
442
+ */
443
+ export function failedToolSubstepTitle(capability, summary) {
444
+ const trimmed = summary.trim();
445
+ if (!trimmed || trimmed === capability)
446
+ return capability;
447
+ const boundary = trimmed.search(/[.:;](\s|$)|\n/);
448
+ let gist = boundary > 0 ? trimmed.slice(0, boundary) : trimmed;
449
+ if (gist.length > FAILED_SUBSTEP_REASON_CHARS) {
450
+ gist = `${gist.slice(0, FAILED_SUBSTEP_REASON_CHARS - 1).trimEnd()}…`;
451
+ }
452
+ return `${capability}: ${gist}`;
453
+ }
417
454
  /**
418
455
  * Model-declared substeps first, then the observed calls.
419
456
  *
@@ -433,3 +470,35 @@ function mergeSubsteps(entry) {
433
470
  }));
434
471
  return [...declared, ...entry.toolSubsteps];
435
472
  }
473
+ // ---------------------------------------------------------------------------
474
+ // Formatting
475
+ // ---------------------------------------------------------------------------
476
+ /**
477
+ * How long, in words. One owner, because there were three and all three were wrong.
478
+ *
479
+ * The rail, `oxygen copilot plan` and the MCP widget each carried their own copy
480
+ * of this, and every copy did `Math.floor(s / 60)` minutes with `Math.round(s % 60)`
481
+ * seconds -- which rounds the remainder INDEPENDENTLY of the minutes it was taken
482
+ * from. A 179.6s sub-agent therefore rendered as "2m 60s" on all three surfaces
483
+ * (observed live in session 34826105), and 59.6s rendered as "60s" rather than
484
+ * "1m". Rounding the total FIRST and splitting afterwards cannot produce either.
485
+ *
486
+ * The projection promised one implementation behind three surfaces; the numbers
487
+ * were shared and the words describing them were not, so this is where they meet.
488
+ */
489
+ export function formatCopilotPlanSeconds(seconds) {
490
+ if (!Number.isFinite(seconds))
491
+ return "0s";
492
+ const total = Math.max(Math.round(seconds), 0);
493
+ if (total < 60)
494
+ return `${total}s`;
495
+ const minutes = Math.floor(total / 60);
496
+ const rest = total % 60;
497
+ return rest === 0 ? `${minutes}m` : `${minutes}m ${rest}s`;
498
+ }
499
+ /** A duration in ms, or null when there is nothing worth showing. */
500
+ export function formatCopilotPlanDuration(ms) {
501
+ if (typeof ms !== "number" || !Number.isFinite(ms) || ms <= 0)
502
+ return null;
503
+ return ms < 1000 ? `${Math.round(ms)}ms` : formatCopilotPlanSeconds(ms / 1000);
504
+ }
@@ -54,11 +54,42 @@ export type LlmEventBody = {
54
54
  sessionId?: string | null;
55
55
  userId?: string | null;
56
56
  };
57
+ /**
58
+ * An eval/annotation attached to a trace (ADR 0014 files these as exactly what
59
+ * the LLM store exists to enable). Narrower than the API's CreateScoreRequest on
60
+ * purpose: `traceId` is REQUIRED (a score orphaned from its trace is unreadable
61
+ * in the UI), `value` is numeric-only, and `dataType` drops "CATEGORICAL"
62
+ * because that variant requires a *string* value. Widening later stays additive.
63
+ */
64
+ export type LlmScoreBody = {
65
+ /** Deterministic id upserts; omit for a new score each call. */
66
+ id?: string;
67
+ traceId: string;
68
+ name: string;
69
+ value: number;
70
+ dataType?: "BOOLEAN" | "NUMERIC";
71
+ comment?: string;
72
+ };
57
73
  export type LlmTracingClient = {
58
74
  trace(body: LlmTraceBody): void;
59
75
  span(body: LlmSpanBody): void;
60
76
  generation(body: LlmGenerationBody): void;
61
77
  event(body: LlmEventBody): void;
78
+ /**
79
+ * Attach a score to an existing trace.
80
+ *
81
+ * ASYNC, unlike the four above, because it is not the same transport. Spans go
82
+ * through the OTel processor, which batches and is drained by `flush()`; a
83
+ * score has no span, so it is one HTTP call to the scores API. Awaiting it is
84
+ * what keeps it alive on a serverless function that freezes the moment the
85
+ * handler returns — there is nothing for `flush()` to drain on its behalf.
86
+ *
87
+ * Never rejects: resolves `true` when the score was accepted, `false` when it
88
+ * was not (transport down, tracing misconfigured, API refusal). Fail-open like
89
+ * every other path here — telemetry must not break the product write it
90
+ * describes.
91
+ */
92
+ score(body: LlmScoreBody): Promise<boolean>;
62
93
  /** Never rejects; bounded at ~5s. */
63
94
  flush(): Promise<void>;
64
95
  /** Flush + stop background timers. Never rejects; bounded at ~5s. */
@@ -108,6 +139,10 @@ export type LlmEmitter = {
108
139
  flush(): Promise<void>;
109
140
  shutdown(): Promise<void>;
110
141
  };
142
+ export type LlmScorer = {
143
+ /** Resolves true when the score was accepted. Never rejects. */
144
+ score(body: LlmScoreBody): Promise<boolean>;
145
+ };
111
146
  /**
112
147
  * Construct a fail-open Langfuse client, or `null` when tracing is disabled or
113
148
  * misconfigured. Prefer the process-wide `getLlmTracingClient` in app code;
@@ -115,6 +150,7 @@ export type LlmEmitter = {
115
150
  */
116
151
  export declare function createLlmTracingClient(env?: EnvMap, options?: {
117
152
  emitterImpl?: LlmEmitter;
153
+ scorerImpl?: LlmScorer;
118
154
  }): LlmTracingClient | null;
119
155
  export declare function getLlmTracingClient(env?: EnvMap): LlmTracingClient | null;
120
156
  /** Flush the singleton if it exists. Never rejects. Hang off request/cycle ends. */
@@ -126,6 +126,17 @@ function boundJsonField(value) {
126
126
  preview: serialized.slice(0, MAX_JSON_FIELD_CHARS),
127
127
  };
128
128
  }
129
+ // A score's `comment` is free text (a thumbs-down reason is user-authored and
130
+ // unbounded) and the API types it as a STRING, so it cannot take
131
+ // boundJsonField's `{truncated, preview}` envelope. It gets the same
132
+ // MAX_JSON_FIELD_CHARS ceiling and the same "explicit marker, never a silent
133
+ // clip" rule, with the marker counted INSIDE the cap so the bound holds.
134
+ const COMMENT_TRUNCATION_MARKER = "…[truncated]";
135
+ function boundComment(comment) {
136
+ if (comment.length <= MAX_JSON_FIELD_CHARS)
137
+ return comment;
138
+ return comment.slice(0, MAX_JSON_FIELD_CHARS - COMMENT_TRUNCATION_MARKER.length) + COMMENT_TRUNCATION_MARKER;
139
+ }
129
140
  function compact(body) {
130
141
  const out = {};
131
142
  for (const [key, value] of Object.entries(body)) {
@@ -223,6 +234,65 @@ function createOtelEmitter(env, warn) {
223
234
  shutdown: () => init().then((h) => h?.processor.shutdown() ?? Promise.resolve()),
224
235
  };
225
236
  }
237
+ /**
238
+ * The real scorer: one authenticated POST to the Langfuse scores API.
239
+ *
240
+ * `@langfuse/core` is imported LAZILY on first score for the same reason the
241
+ * OTel tree is — a runtime with tracing disabled (the packed CLI, every test)
242
+ * never pays for it.
243
+ *
244
+ * The API client's `environment` option is its BASE URL, not the Langfuse
245
+ * environment tag; the tag is the `environment` FIELD on the score body, and it
246
+ * is resolved from the same helper the span processor uses so a score lands in
247
+ * the same Langfuse environment as the trace it scores. OXYGEN always sets
248
+ * LANGFUSE_BASE_URL (the project is US-region and the EU host 401s these keys);
249
+ * the fallback is the SDK's documented default, which `@langfuse/core` itself
250
+ * does not supply.
251
+ */
252
+ function createApiScorer(env, warn) {
253
+ let handle = null;
254
+ const init = () => {
255
+ handle ??= (async () => {
256
+ try {
257
+ const { LangfuseAPIClient } = await import("@langfuse/core");
258
+ const client = new LangfuseAPIClient({
259
+ environment: env.LANGFUSE_BASE_URL?.trim() || "https://cloud.langfuse.com",
260
+ username: env.LANGFUSE_PUBLIC_KEY,
261
+ password: env.LANGFUSE_SECRET_KEY,
262
+ });
263
+ return client.scores;
264
+ }
265
+ catch (error) {
266
+ warn(error, { stage: "score_init" });
267
+ return null;
268
+ }
269
+ })();
270
+ return handle;
271
+ };
272
+ return {
273
+ score: async (body) => {
274
+ try {
275
+ const api = await init();
276
+ if (!api)
277
+ return false;
278
+ await api.create(compact({
279
+ id: body.id,
280
+ traceId: body.traceId,
281
+ name: body.name,
282
+ value: body.value,
283
+ dataType: body.dataType,
284
+ comment: typeof body.comment === "string" ? boundComment(body.comment) : undefined,
285
+ environment: resolveLlmTracingEnvironment(env),
286
+ }));
287
+ return true;
288
+ }
289
+ catch (error) {
290
+ warn(error, { stage: "score" });
291
+ return false;
292
+ }
293
+ },
294
+ };
295
+ }
226
296
  /**
227
297
  * Construct a fail-open Langfuse client, or `null` when tracing is disabled or
228
298
  * misconfigured. Prefer the process-wide `getLlmTracingClient` in app code;
@@ -244,6 +314,7 @@ export function createLlmTracingClient(env = process.env, options) {
244
314
  });
245
315
  };
246
316
  const emitter = options?.emitterImpl ?? createOtelEmitter(env, warn);
317
+ const scorer = options?.scorerImpl ?? createApiScorer(env, warn);
247
318
  const guarded = (fn, stage) => {
248
319
  try {
249
320
  fn();
@@ -351,6 +422,18 @@ export function createLlmTracingClient(env = process.env, options) {
351
422
  endTime: startTime,
352
423
  });
353
424
  }, "event"),
425
+ // Already fail-open inside the scorer; the extra catch is here so that a
426
+ // scorer which throws SYNCHRONOUSLY (a substituted one in a test, a future
427
+ // implementation) still cannot escape into a product write.
428
+ score: async (body) => {
429
+ try {
430
+ return await scorer.score(body);
431
+ }
432
+ catch (error) {
433
+ warn(error, { stage: "score" });
434
+ return false;
435
+ }
436
+ },
354
437
  // The "never rejects, bounded at ~5s" contract is the CLIENT's, so it is
355
438
  // enforced here rather than inside one emitter — an emitter that throws
356
439
  // synchronously or rejects must still not escape into product code.
@@ -1,5 +1,5 @@
1
1
  /**
2
- * How a LinkedIn quota denial is read by the background jobs that hit it.
2
+ * How a LinkedIn/WhatsApp quota denial is read by the background jobs that hit it.
3
3
  *
4
4
  * The denial itself is raised by the chokepoint in
5
5
  * packages/integrations/src/linkedin-quota.ts; this module is the consumer half,
@@ -1,5 +1,5 @@
1
1
  /**
2
- * How a LinkedIn quota denial is read by the background jobs that hit it.
2
+ * How a LinkedIn/WhatsApp quota denial is read by the background jobs that hit it.
3
3
  *
4
4
  * The denial itself is raised by the chokepoint in
5
5
  * packages/integrations/src/linkedin-quota.ts; this module is the consumer half,
@@ -16,10 +16,21 @@
16
16
  */
17
17
  import { OxygenError } from "./cli-result.js";
18
18
  import { isRecord } from "./type-guards.js";
19
- // The two codes a denial is raised with. Either is a "come back later" signal,
20
- // not a broken caller: a daily cap / closed active window that reopens on its own
21
- // clock, or an account the status webhook will reactivate.
22
- const QUOTA_DENIED_CODES = new Set(["linkedin_rate_limited", "linkedin_account_unavailable"]);
19
+ // The codes a denial is raised with. Each is a "come back later" signal, not a
20
+ // broken caller: a daily cap / closed active window that reopens on its own clock,
21
+ // or an account the status webhook will reactivate.
22
+ //
23
+ // The WhatsApp mirrors are here because the inbox backstop is shared across
24
+ // networks: a WhatsApp daily-limit denial was reaching it, missing this set, and
25
+ // so was logged as a failure AND never parked -- the same hot re-deny loop the
26
+ // LinkedIn codes were added to stop. Both WhatsApp denials carry `resets_at`, so
27
+ // the park lands on the real reset rather than the fallback below.
28
+ const QUOTA_DENIED_CODES = new Set([
29
+ "linkedin_rate_limited",
30
+ "linkedin_account_unavailable",
31
+ "whatsapp_rate_limited",
32
+ "whatsapp_account_unavailable",
33
+ ]);
23
34
  /** Park length when a denial carries no usable `resets_at` hint. */
24
35
  const QUOTA_FALLBACK_BACKOFF_MS = 60 * 60 * 1000;
25
36
  /** Was this thrown error the quota chokepoint refusing the call? */