@oxygen-agent/cli 1.906.0 → 1.922.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/command-manifest.js +13 -1
- package/dist/index.js +337 -40
- package/node_modules/@oxygen/formula/dist/expression.d.ts +21 -0
- package/node_modules/@oxygen/formula/dist/expression.js +42 -1
- package/node_modules/@oxygen/formula/dist/formula-functions.d.ts +1 -1
- package/node_modules/@oxygen/formula/dist/formula-functions.js +10 -1
- package/node_modules/@oxygen/shared/dist/billing.d.ts +17 -0
- package/node_modules/@oxygen/shared/dist/billing.js +20 -0
- package/node_modules/@oxygen/shared/dist/capability-discovery.js +12 -7
- package/node_modules/@oxygen/shared/dist/copilot-errors.d.ts +1 -0
- package/node_modules/@oxygen/shared/dist/copilot-errors.js +9 -0
- package/node_modules/@oxygen/shared/dist/copilot-plan.d.ts +39 -0
- package/node_modules/@oxygen/shared/dist/copilot-plan.js +76 -7
- package/node_modules/@oxygen/shared/dist/langfuse.d.ts +36 -0
- package/node_modules/@oxygen/shared/dist/langfuse.js +83 -0
- package/node_modules/@oxygen/shared/dist/linkedin-quota-denial.d.ts +1 -1
- package/node_modules/@oxygen/shared/dist/linkedin-quota-denial.js +16 -5
- package/node_modules/@oxygen/shared/dist/linkedin-url.d.ts +22 -0
- package/node_modules/@oxygen/shared/dist/linkedin-url.js +104 -0
- package/node_modules/@oxygen/shared/dist/provider-funding-errors.d.ts +16 -2
- package/node_modules/@oxygen/shared/dist/provider-funding-errors.js +19 -9
- package/node_modules/@oxygen/shared/dist/table-limits.d.ts +18 -0
- package/node_modules/@oxygen/shared/dist/table-limits.js +18 -0
- package/node_modules/@oxygen/shared/dist/user-capability-routing.js +23 -2
- package/node_modules/@oxygen/shared/dist/version.d.ts +1 -1
- package/node_modules/@oxygen/shared/dist/version.js +1 -1
- package/node_modules/@oxygen/shared/package.json +10 -0
- package/package.json +2 -1
|
@@ -41,6 +41,27 @@ export type ParseOptions = {
|
|
|
41
41
|
export declare function parseFormulaExpression(expression: string, options?: ParseOptions): FormulaAst;
|
|
42
42
|
/** Validate supported functions and arity on an already-parsed tree. */
|
|
43
43
|
export declare function validateFormulaAst(ast: FormulaAst): void;
|
|
44
|
+
/**
|
|
45
|
+
* SAVE-TIME ONLY: reject a literal regex pattern that is too long to ever compile.
|
|
46
|
+
*
|
|
47
|
+
* Deliberately NOT part of validateFormulaAst. That runs on the READ path too
|
|
48
|
+
* (formula-column-runner prepares every batch through it) and
|
|
49
|
+
* prepareFormulaForRead swallows the throw and nulls the WHOLE column, so
|
|
50
|
+
* tightening it would silently blank already-stored columns rather than reject
|
|
51
|
+
* the edit. Keeping this save-only means a column that stores today keeps
|
|
52
|
+
* reading today, and only a NEW write is refused.
|
|
53
|
+
*
|
|
54
|
+
* Length is checked, not compilation. A pattern over the cap can never succeed in
|
|
55
|
+
* any branch, so flagging it costs the author nothing -- whereas rejecting an
|
|
56
|
+
* unparseable literal would break `if(cond, regex_match(x, "["), "n/a")`, which
|
|
57
|
+
* works today precisely because if/switch/and/or/coalesce are lazy and never
|
|
58
|
+
* evaluate the untaken side. That stricter semantic is a separate decision.
|
|
59
|
+
*
|
|
60
|
+
* Without this, an over-long pattern stored fine and then failed per row on every
|
|
61
|
+
* run forever -- the production shape was one column failing across dozens of rows
|
|
62
|
+
* for days with received_length 290 against the old 256 cap.
|
|
63
|
+
*/
|
|
64
|
+
export declare function validateFormulaRegexLiterals(ast: FormulaAst): void;
|
|
44
65
|
/** Validate syntax, supported functions, and arity without evaluating any data. */
|
|
45
66
|
export declare function validateFormulaExpression(expression: string, options?: ParseOptions): void;
|
|
46
67
|
export declare function walkExpression(node: FormulaAst, visit: (node: FormulaAst) => void): void;
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { FORMULA_FUNCTION_REGISTRY, checkFormulaFunctionArity, formulaExpressionError, } from "./formula-functions.js";
|
|
1
|
+
import { FORMULA_FUNCTION_REGISTRY, MAX_FORMULA_REGEX_PATTERN_LENGTH, checkFormulaFunctionArity, formulaExpressionError, } from "./formula-functions.js";
|
|
2
2
|
/**
|
|
3
3
|
* Functions evaluated against the whole column rather than the current scope.
|
|
4
4
|
* They are intercepted before registry dispatch (the registry entry carries
|
|
@@ -37,6 +37,47 @@ export function validateFormulaAst(ast) {
|
|
|
37
37
|
checkFormulaFunctionArity(spec, node.args.length);
|
|
38
38
|
});
|
|
39
39
|
}
|
|
40
|
+
/**
|
|
41
|
+
* SAVE-TIME ONLY: reject a literal regex pattern that is too long to ever compile.
|
|
42
|
+
*
|
|
43
|
+
* Deliberately NOT part of validateFormulaAst. That runs on the READ path too
|
|
44
|
+
* (formula-column-runner prepares every batch through it) and
|
|
45
|
+
* prepareFormulaForRead swallows the throw and nulls the WHOLE column, so
|
|
46
|
+
* tightening it would silently blank already-stored columns rather than reject
|
|
47
|
+
* the edit. Keeping this save-only means a column that stores today keeps
|
|
48
|
+
* reading today, and only a NEW write is refused.
|
|
49
|
+
*
|
|
50
|
+
* Length is checked, not compilation. A pattern over the cap can never succeed in
|
|
51
|
+
* any branch, so flagging it costs the author nothing -- whereas rejecting an
|
|
52
|
+
* unparseable literal would break `if(cond, regex_match(x, "["), "n/a")`, which
|
|
53
|
+
* works today precisely because if/switch/and/or/coalesce are lazy and never
|
|
54
|
+
* evaluate the untaken side. That stricter semantic is a separate decision.
|
|
55
|
+
*
|
|
56
|
+
* Without this, an over-long pattern stored fine and then failed per row on every
|
|
57
|
+
* run forever -- the production shape was one column failing across dozens of rows
|
|
58
|
+
* for days with received_length 290 against the old 256 cap.
|
|
59
|
+
*/
|
|
60
|
+
export function validateFormulaRegexLiterals(ast) {
|
|
61
|
+
walkExpression(ast, (node) => {
|
|
62
|
+
if (node.type !== "call")
|
|
63
|
+
return;
|
|
64
|
+
const normalized = node.name.toLowerCase();
|
|
65
|
+
if (normalized !== "regex_match" && normalized !== "regex_extract" && normalized !== "regex_replace") {
|
|
66
|
+
return;
|
|
67
|
+
}
|
|
68
|
+
// Every regex function takes its pattern as the second argument.
|
|
69
|
+
const pattern = node.args[1];
|
|
70
|
+
if (!pattern || pattern.type !== "literal" || typeof pattern.value !== "string")
|
|
71
|
+
return;
|
|
72
|
+
if (pattern.value.length <= MAX_FORMULA_REGEX_PATTERN_LENGTH)
|
|
73
|
+
return;
|
|
74
|
+
throw formulaExpressionError("Regular-expression pattern is too long.", {
|
|
75
|
+
function: node.name,
|
|
76
|
+
max_length: MAX_FORMULA_REGEX_PATTERN_LENGTH,
|
|
77
|
+
received_length: pattern.value.length,
|
|
78
|
+
});
|
|
79
|
+
});
|
|
80
|
+
}
|
|
40
81
|
/** Validate syntax, supported functions, and arity without evaluating any data. */
|
|
41
82
|
export function validateFormulaExpression(expression, options = {}) {
|
|
42
83
|
validateFormulaAst(parseFormulaExpression(expression, options));
|
|
@@ -37,7 +37,7 @@ export type FormulaFunctionSpec = FormulaFunctionMeta & {
|
|
|
37
37
|
evaluate?: (args: unknown[]) => unknown;
|
|
38
38
|
evaluateLazy?: (thunks: Array<() => unknown>) => unknown;
|
|
39
39
|
};
|
|
40
|
-
export declare const MAX_FORMULA_REGEX_PATTERN_LENGTH =
|
|
40
|
+
export declare const MAX_FORMULA_REGEX_PATTERN_LENGTH = 1000;
|
|
41
41
|
export declare const MAX_FORMULA_REGEX_INPUT_LENGTH = 20000;
|
|
42
42
|
export declare function formulaExpressionError(message: string, details: Record<string, unknown>): OxygenError;
|
|
43
43
|
export declare function isBlankFormulaValue(value: unknown): boolean;
|
|
@@ -1,7 +1,16 @@
|
|
|
1
1
|
import { OxygenError } from "@oxygen/shared/cli-result";
|
|
2
2
|
import { walkJsonPath } from "@oxygen/shared/json-path";
|
|
3
3
|
import { normalizeDomain, normalizeEmail, normalizeLinkedinUrl, } from "./value-normalizers.js";
|
|
4
|
-
|
|
4
|
+
// Pattern LENGTH is not a safety bound and never was -- catastrophic backtracking
|
|
5
|
+
// is a function of pattern SHAPE, and `(a+)+$` hangs at 6 characters, so 256 already
|
|
6
|
+
// permitted an unbounded hang while rejecting harmless long literals. The real bound
|
|
7
|
+
// on scan cost is MAX_FORMULA_REGEX_INPUT_LENGTH below, which caps what a pattern is
|
|
8
|
+
// run against. 256 rejected a customer's 290-character alternation on every row of
|
|
9
|
+
// their table, continuously, for days (production 2026-08-28/31). Raised to a value
|
|
10
|
+
// that still stops a pathological paste while leaving ordinary generated alternations
|
|
11
|
+
// room. If real ReDoS protection is ever needed it belongs in the engine (a
|
|
12
|
+
// backtracking budget or re2), not in a character count.
|
|
13
|
+
export const MAX_FORMULA_REGEX_PATTERN_LENGTH = 1_000;
|
|
5
14
|
export const MAX_FORMULA_REGEX_INPUT_LENGTH = 20_000;
|
|
6
15
|
// ---------------------------------------------------------------------------
|
|
7
16
|
// Shared coercion helpers (used by the registry AND the runner's operators)
|
|
@@ -33,6 +33,23 @@ export declare const CREDIT_TOPUP_MAX_CREDITS = 1000000;
|
|
|
33
33
|
export declare const CREDIT_TOPUP_STEP_CREDITS = 1000;
|
|
34
34
|
export declare const CREDIT_TOPUP_DEFAULT_CREDITS = 20000;
|
|
35
35
|
export declare function isValidCreditTopupCredits(credits: number): boolean;
|
|
36
|
+
/**
|
|
37
|
+
* A shortfall expressed as an amount `oxygen billing topup` will actually SELL:
|
|
38
|
+
* rounded up onto the 1,000-credit step and clamped into the purchasable
|
|
39
|
+
* [8,000 .. 1,000,000] band. A next_action (or an alert email) naming a number the
|
|
40
|
+
* top-up route refuses with `invalid_topup_amount` is not a next action, it is a
|
|
41
|
+
* second dead end during the incident it exists to end.
|
|
42
|
+
*
|
|
43
|
+
* Both clamps are honest, not cosmetic: below the floor the customer buys the
|
|
44
|
+
* smallest pack (more than the debt — purchased credits never expire), and above
|
|
45
|
+
* the ceiling they buy the largest single checkout and top up again. Callers carry
|
|
46
|
+
* the raw shortfall alongside so either gap stays visible rather than implied.
|
|
47
|
+
*
|
|
48
|
+
* Shared rather than re-derived per surface: `billing balance` and the renewal
|
|
49
|
+
* failure alert quote this number to the same customer about the same debt, and a
|
|
50
|
+
* second rounding formula is how two customer-facing figures drift apart.
|
|
51
|
+
*/
|
|
52
|
+
export declare function creditTopupAmountForShortfall(credits: number): number;
|
|
36
53
|
export declare function creditTopupUsdCents(credits: number): number | null;
|
|
37
54
|
export declare const AUTOMATION_ACTION_CREDITS = 0.01;
|
|
38
55
|
/** Kinds of resource that carry a fixed monthly credit commitment. */
|
|
@@ -35,6 +35,26 @@ export function isValidCreditTopupCredits(credits) {
|
|
|
35
35
|
&& credits <= CREDIT_TOPUP_MAX_CREDITS
|
|
36
36
|
&& credits % CREDIT_TOPUP_STEP_CREDITS === 0;
|
|
37
37
|
}
|
|
38
|
+
/**
|
|
39
|
+
* A shortfall expressed as an amount `oxygen billing topup` will actually SELL:
|
|
40
|
+
* rounded up onto the 1,000-credit step and clamped into the purchasable
|
|
41
|
+
* [8,000 .. 1,000,000] band. A next_action (or an alert email) naming a number the
|
|
42
|
+
* top-up route refuses with `invalid_topup_amount` is not a next action, it is a
|
|
43
|
+
* second dead end during the incident it exists to end.
|
|
44
|
+
*
|
|
45
|
+
* Both clamps are honest, not cosmetic: below the floor the customer buys the
|
|
46
|
+
* smallest pack (more than the debt — purchased credits never expire), and above
|
|
47
|
+
* the ceiling they buy the largest single checkout and top up again. Callers carry
|
|
48
|
+
* the raw shortfall alongside so either gap stays visible rather than implied.
|
|
49
|
+
*
|
|
50
|
+
* Shared rather than re-derived per surface: `billing balance` and the renewal
|
|
51
|
+
* failure alert quote this number to the same customer about the same debt, and a
|
|
52
|
+
* second rounding formula is how two customer-facing figures drift apart.
|
|
53
|
+
*/
|
|
54
|
+
export function creditTopupAmountForShortfall(credits) {
|
|
55
|
+
const stepped = Math.ceil(credits / CREDIT_TOPUP_STEP_CREDITS) * CREDIT_TOPUP_STEP_CREDITS;
|
|
56
|
+
return Math.min(CREDIT_TOPUP_MAX_CREDITS, Math.max(CREDIT_TOPUP_MIN_CREDITS, stepped));
|
|
57
|
+
}
|
|
38
58
|
export function creditTopupUsdCents(credits) {
|
|
39
59
|
if (!isValidCreditTopupCredits(credits))
|
|
40
60
|
return null;
|
|
@@ -21,15 +21,20 @@ export const OXYGEN_CAPABILITY_ROUTES = [
|
|
|
21
21
|
id: "discovery-and-skills",
|
|
22
22
|
layer: "Control",
|
|
23
23
|
primitive: null,
|
|
24
|
-
owns: "Bounded capability, command, provider-operation, Recipe, and product-skill discovery with exact hydration on demand.",
|
|
24
|
+
owns: "Bounded capability, command, provider-operation, Recipe, and product-skill discovery, plus INSTANCE discovery — what this workspace actually holds — with exact hydration on demand.",
|
|
25
25
|
notFor: "Executing GTM work or loading full manifests and schemas before a capability is selected.",
|
|
26
|
-
execution: "Search by outcome, choose the owner, then hydrate one exact command, MCP schema, provider descriptor, or skill.",
|
|
26
|
+
execution: "Search by outcome, choose the owner, then hydrate one exact command, MCP schema, provider descriptor, or skill. For instance discovery, read the map and then the primitive that owns the row.",
|
|
27
27
|
posture: "read_only",
|
|
28
|
-
gatewayTools: ["oxygen_capabilities_search", "oxygen_tools_search", "oxygen_recipes_list"],
|
|
29
|
-
gatewayCommands: ["capabilities search", "commands search", "skills search", "tools search", "recipes list"],
|
|
28
|
+
gatewayTools: ["oxygen_capabilities_search", "oxygen_tools_search", "oxygen_recipes_list", "oxygen_workspace_map"],
|
|
29
|
+
gatewayCommands: ["capabilities search", "commands search", "skills search", "tools search", "recipes list", "workspace map"],
|
|
30
30
|
skills: ["oxygen-quickstart", "oxygen-gtm"],
|
|
31
|
-
|
|
32
|
-
|
|
31
|
+
// `workspace` (the map) belongs here rather than beside `home`: capability
|
|
32
|
+
// discovery answers what OXYGEN can do and instance discovery answers what THIS
|
|
33
|
+
// workspace holds, and they are the same act — bounded, read-only, hydrate one
|
|
34
|
+
// thing on demand. `home` stayed on the onboarding card because the standup is a
|
|
35
|
+
// first-run surface; the map is not.
|
|
36
|
+
endpointSections: ["skills", "workspace"],
|
|
37
|
+
intentTerms: ["discover", "discovery", "capability", "capabilities", "command", "commands", "skill", "skills", "which tool", "how do i", "what do i have", "what is in my workspace", "workspace map"],
|
|
33
38
|
},
|
|
34
39
|
{
|
|
35
40
|
id: "onboarding-and-copilot",
|
|
@@ -196,7 +201,7 @@ export const OXYGEN_CAPABILITY_ROUTES = [
|
|
|
196
201
|
gatewayTools: ["oxygen_tables_create", "oxygen_columns_add", "oxygen_enrich_column_preview", "oxygen_tables_link_bulk"],
|
|
197
202
|
gatewayCommands: ["tables create", "columns add", "enrich-column preview", "tables link"],
|
|
198
203
|
skills: ["oxygen-gtm", "oxygen-table-tidy", "oxygen-diagnostics", "oxygen-clay-migration"],
|
|
199
|
-
endpointSections: ["action-columns", "callables", "columns", "company-enrichment", "enrich-column", "enrichment", "projects", "table-action-runs", "table-ingestion-runs", "tables"],
|
|
204
|
+
endpointSections: ["action-columns", "callables", "columns", "company-enrichment", "enrich-column", "enrichment", "projects", "table-action-items", "table-action-runs", "table-ingestion-runs", "tables"],
|
|
200
205
|
// "link"/"join"/"connect"/"relate" route here for `tables link`. Added after a
|
|
201
206
|
// blind user eval asked for exactly "link two tables" and was routed to
|
|
202
207
|
// `tables create` / `columns add` / `enrich-column preview` — none of which
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
export declare const COPILOT_TURN_TIMEOUT_CODE = "copilot_turn_timeout";
|
|
2
2
|
export declare const COPILOT_TURN_TIMEOUT_MESSAGE = "This Copilot request timed out and stopped. Any completed actions are still saved\u2014review this session before trying again.";
|
|
3
|
+
export declare const COPILOT_TURN_DEADLINE_EXCEEDED_CODE = "copilot_turn_deadline_exceeded";
|
|
3
4
|
export type CustomerFacingCopilotError = {
|
|
4
5
|
code: string | null;
|
|
5
6
|
message: string | null;
|
|
@@ -4,6 +4,14 @@
|
|
|
4
4
|
// and already-persisted failures serialize identically across web, CLI, and MCP.
|
|
5
5
|
export const COPILOT_TURN_TIMEOUT_CODE = "copilot_turn_timeout";
|
|
6
6
|
export const COPILOT_TURN_TIMEOUT_MESSAGE = "This Copilot request timed out and stopped. Any completed actions are still saved—review this session before trying again.";
|
|
7
|
+
// A turn the worker refused to start because it was ALREADY past its stamped
|
|
8
|
+
// wall-clock deadline when it was reclaimed. Distinct in the durable row from
|
|
9
|
+
// `copilot_turn_timeout` (a turn that ran and then ran out of clock) so an
|
|
10
|
+
// operator can separate "we never started it" from "we could not finish it" in
|
|
11
|
+
// SQL — but deliberately NOT distinct to the customer: the mapping below folds
|
|
12
|
+
// it into the same timeout contract every Control surface already renders, per
|
|
13
|
+
// this module's policy.
|
|
14
|
+
export const COPILOT_TURN_DEADLINE_EXCEEDED_CODE = "copilot_turn_deadline_exceeded";
|
|
7
15
|
const WORKER_STEP_TIMEOUT_MESSAGE = /\bWorker step '[^']+' exceeded \d+ms deadline\.?/i;
|
|
8
16
|
/**
|
|
9
17
|
* Replace an internal worker deadline with the stable Copilot timeout contract.
|
|
@@ -15,6 +23,7 @@ export function customerFacingCopilotError(input) {
|
|
|
15
23
|
const message = input.message ?? null;
|
|
16
24
|
const isTimeout = code === "worker_step_timeout" ||
|
|
17
25
|
code === COPILOT_TURN_TIMEOUT_CODE ||
|
|
26
|
+
code === COPILOT_TURN_DEADLINE_EXCEEDED_CODE ||
|
|
18
27
|
(message !== null && WORKER_STEP_TIMEOUT_MESSAGE.test(message));
|
|
19
28
|
return isTimeout
|
|
20
29
|
? { code: COPILOT_TURN_TIMEOUT_CODE, message: COPILOT_TURN_TIMEOUT_MESSAGE }
|
|
@@ -109,6 +109,8 @@ export type CopilotPlanProjection = {
|
|
|
109
109
|
/** True when the turn's wall-clock deadline, not the plan, is the binding bound. */
|
|
110
110
|
deadlineBound: boolean;
|
|
111
111
|
};
|
|
112
|
+
/** False when no turn is executing: the plan is history, not work in flight. */
|
|
113
|
+
turnActive: boolean;
|
|
112
114
|
/** Ledger position of the newest plan_updated, so a caller can tell staleness. */
|
|
113
115
|
updatedAtSeq: number;
|
|
114
116
|
updatedAt: string;
|
|
@@ -124,6 +126,20 @@ export type ProjectCopilotPlanInput = {
|
|
|
124
126
|
events: CopilotPlanSourceEvent[];
|
|
125
127
|
/** The active turn's wall-clock deadline, used only to clamp the total. */
|
|
126
128
|
turnDeadlineAt?: string | Date | null;
|
|
129
|
+
/**
|
|
130
|
+
* Whether a turn is executing right now. Defaults to true.
|
|
131
|
+
*
|
|
132
|
+
* A plan OUTLIVES the turn that wrote it. The model routinely stops without
|
|
133
|
+
* marking its last step done, so the session rests in the ledger forever with a
|
|
134
|
+
* step still `in_progress`. Read against a running clock that step accrues
|
|
135
|
+
* elapsed time indefinitely -- production session fd259942 reached 37 DAYS --
|
|
136
|
+
* and the surface goes on offering "time remaining" for work that stopped weeks
|
|
137
|
+
* ago. Both are the same error: treating a historical record as work in flight.
|
|
138
|
+
*
|
|
139
|
+
* So the clock stops with the turn, and a plan nobody is working reports no
|
|
140
|
+
* estimate at all rather than a false one.
|
|
141
|
+
*/
|
|
142
|
+
turnActive?: boolean;
|
|
127
143
|
now?: Date;
|
|
128
144
|
};
|
|
129
145
|
/**
|
|
@@ -135,3 +151,26 @@ export type ProjectCopilotPlanInput = {
|
|
|
135
151
|
* would be worse than one that stays away.
|
|
136
152
|
*/
|
|
137
153
|
export declare function projectCopilotPlan(input: ProjectCopilotPlanInput): CopilotPlanProjection | null;
|
|
154
|
+
/**
|
|
155
|
+
* "<capability>: <gist>" for a failed call. The gist is the summary cut at its
|
|
156
|
+
* first sentence or clause boundary and then hard-capped, so a schema dump or a
|
|
157
|
+
* multi-sentence explanation cannot become the title; a summary that is only
|
|
158
|
+
* the capability's own name, or empty, leaves the bare name.
|
|
159
|
+
*/
|
|
160
|
+
export declare function failedToolSubstepTitle(capability: string, summary: string): string;
|
|
161
|
+
/**
|
|
162
|
+
* How long, in words. One owner, because there were three and all three were wrong.
|
|
163
|
+
*
|
|
164
|
+
* The rail, `oxygen copilot plan` and the MCP widget each carried their own copy
|
|
165
|
+
* of this, and every copy did `Math.floor(s / 60)` minutes with `Math.round(s % 60)`
|
|
166
|
+
* seconds -- which rounds the remainder INDEPENDENTLY of the minutes it was taken
|
|
167
|
+
* from. A 179.6s sub-agent therefore rendered as "2m 60s" on all three surfaces
|
|
168
|
+
* (observed live in session 34826105), and 59.6s rendered as "60s" rather than
|
|
169
|
+
* "1m". Rounding the total FIRST and splitting afterwards cannot produce either.
|
|
170
|
+
*
|
|
171
|
+
* The projection promised one implementation behind three surfaces; the numbers
|
|
172
|
+
* were shared and the words describing them were not, so this is where they meet.
|
|
173
|
+
*/
|
|
174
|
+
export declare function formatCopilotPlanSeconds(seconds: number): string;
|
|
175
|
+
/** A duration in ms, or null when there is nothing worth showing. */
|
|
176
|
+
export declare function formatCopilotPlanDuration(ms: number | null | undefined): string | null;
|
|
@@ -169,6 +169,13 @@ export function projectCopilotPlan(input) {
|
|
|
169
169
|
const planEvents = events.filter((event) => event.kind === "plan_updated");
|
|
170
170
|
if (planEvents.length === 0)
|
|
171
171
|
return null;
|
|
172
|
+
// The plan's clock stops when the turn does -- see `turnActive` on the input.
|
|
173
|
+
// `Math.min` because a clock that runs BACKWARDS is worse than one that stops:
|
|
174
|
+
// the last event is normally in the past, but a clock skew must not make an
|
|
175
|
+
// elapsed time negative.
|
|
176
|
+
const turnActive = input.turnActive !== false;
|
|
177
|
+
const lastEvent = events[events.length - 1];
|
|
178
|
+
const clockMs = turnActive || !lastEvent ? nowMs : Math.min(nowMs, toMs(lastEvent.created_at));
|
|
172
179
|
// -- Pass 1: fold the successive plans into one tracked set -----------------
|
|
173
180
|
//
|
|
174
181
|
// Identity is what makes timing possible. Match on `id` first, then on the
|
|
@@ -230,7 +237,7 @@ export function projectCopilotPlan(input) {
|
|
|
230
237
|
if (tracked.length === 0)
|
|
231
238
|
return null;
|
|
232
239
|
// -- Pass 2: attribute observed tool calls to the step that was running -----
|
|
233
|
-
attributeToolCalls(events, planEvents, tracked,
|
|
240
|
+
attributeToolCalls(events, planEvents, tracked, clockMs);
|
|
234
241
|
// -- Pass 3: the ETA --------------------------------------------------------
|
|
235
242
|
const samples = tracked
|
|
236
243
|
.filter((t) => t.status === "done" && t.startedAtMs !== null && t.endedAtMs !== null)
|
|
@@ -250,10 +257,11 @@ export function projectCopilotPlan(input) {
|
|
|
250
257
|
// Once a step has started it has an elapsed time: its own end if it finished,
|
|
251
258
|
// otherwise the clock. A step that never started has none -- which is exactly
|
|
252
259
|
// why it is also never a measurement sample.
|
|
253
|
-
const elapsedMs = entry.startedAtMs === null ? null : (entry.endedAtMs ??
|
|
260
|
+
const elapsedMs = entry.startedAtMs === null ? null : (entry.endedAtMs ?? clockMs) - entry.startedAtMs;
|
|
254
261
|
let etaSeconds = null;
|
|
255
262
|
let etaBasis = "none";
|
|
256
|
-
|
|
263
|
+
// A step only has time REMAINING while something is running to consume it.
|
|
264
|
+
if (turnActive && (entry.status === "pending" || entry.status === "in_progress")) {
|
|
257
265
|
let budget = null;
|
|
258
266
|
if (entry.estimateSeconds !== null && calibration !== null) {
|
|
259
267
|
budget = entry.estimateSeconds * calibration;
|
|
@@ -307,6 +315,7 @@ export function projectCopilotPlan(input) {
|
|
|
307
315
|
}
|
|
308
316
|
const done = steps.filter((step) => step.status === "done").length;
|
|
309
317
|
return {
|
|
318
|
+
turnActive,
|
|
310
319
|
summary: summary.length > 0 ? summary : null,
|
|
311
320
|
steps,
|
|
312
321
|
progress: { done, total: steps.length, ratio: steps.length === 0 ? 0 : done / steps.length },
|
|
@@ -327,7 +336,7 @@ export function projectCopilotPlan(input) {
|
|
|
327
336
|
* Attribution uses the plan as of the call's own seq, not the final plan -- a call
|
|
328
337
|
* made during step 2 belongs to step 2 even after step 4 becomes current.
|
|
329
338
|
*/
|
|
330
|
-
function attributeToolCalls(events, planEvents, tracked,
|
|
339
|
+
function attributeToolCalls(events, planEvents, tracked, clockMs) {
|
|
331
340
|
const byId = new Map(tracked.map((t) => [t.id, t]));
|
|
332
341
|
const byTitle = new Map(tracked.map((t) => [normalizeTitleKey(t.title), t]));
|
|
333
342
|
/** Which tracked step was in progress at a given ledger position. */
|
|
@@ -387,13 +396,22 @@ function attributeToolCalls(events, planEvents, tracked, nowMs) {
|
|
|
387
396
|
if (!target)
|
|
388
397
|
continue;
|
|
389
398
|
const rollup = rollups.get(callId);
|
|
399
|
+
const failed = payload.ok === false;
|
|
390
400
|
target.toolSubsteps.push({
|
|
391
401
|
id: callId,
|
|
392
402
|
// `summary` is the tool_call_finished remap the Copilot host writes; it is a
|
|
393
403
|
// line ("63 columns"), which is what a substep wants. Fall back to the
|
|
394
404
|
// capability name rather than to a serialized result.
|
|
395
|
-
|
|
396
|
-
|
|
405
|
+
//
|
|
406
|
+
// A FAILED call's summary is the error message, which is prose about the
|
|
407
|
+
// failure rather than a name for the step -- a sub-agent's "did not finish
|
|
408
|
+
// (FAILED): This sub-agent run reached its ceiling of 4 model calls."
|
|
409
|
+
// became the substep's whole title. The rail needs what failed and the
|
|
410
|
+
// gist of why, bounded, with the red mark carrying the status.
|
|
411
|
+
title: failed
|
|
412
|
+
? failedToolSubstepTitle(capability, readText(payload.summary))
|
|
413
|
+
: readText(payload.summary) || capability,
|
|
414
|
+
status: failed ? "blocked" : "done",
|
|
397
415
|
source: "tool",
|
|
398
416
|
durationMs: started ? toMs(event.created_at) - started.startedMs : null,
|
|
399
417
|
startedAt: started ? new Date(started.startedMs).toISOString() : null,
|
|
@@ -409,11 +427,30 @@ function attributeToolCalls(events, planEvents, tracked, nowMs) {
|
|
|
409
427
|
title: entry.capability,
|
|
410
428
|
status: "in_progress",
|
|
411
429
|
source: "tool",
|
|
412
|
-
durationMs:
|
|
430
|
+
durationMs: clockMs - entry.startedMs,
|
|
413
431
|
startedAt: new Date(entry.startedMs).toISOString(),
|
|
414
432
|
});
|
|
415
433
|
}
|
|
416
434
|
}
|
|
435
|
+
/** How much of a failure's reason a substep title carries before it stops being a label. */
|
|
436
|
+
const FAILED_SUBSTEP_REASON_CHARS = 90;
|
|
437
|
+
/**
|
|
438
|
+
* "<capability>: <gist>" for a failed call. The gist is the summary cut at its
|
|
439
|
+
* first sentence or clause boundary and then hard-capped, so a schema dump or a
|
|
440
|
+
* multi-sentence explanation cannot become the title; a summary that is only
|
|
441
|
+
* the capability's own name, or empty, leaves the bare name.
|
|
442
|
+
*/
|
|
443
|
+
export function failedToolSubstepTitle(capability, summary) {
|
|
444
|
+
const trimmed = summary.trim();
|
|
445
|
+
if (!trimmed || trimmed === capability)
|
|
446
|
+
return capability;
|
|
447
|
+
const boundary = trimmed.search(/[.:;](\s|$)|\n/);
|
|
448
|
+
let gist = boundary > 0 ? trimmed.slice(0, boundary) : trimmed;
|
|
449
|
+
if (gist.length > FAILED_SUBSTEP_REASON_CHARS) {
|
|
450
|
+
gist = `${gist.slice(0, FAILED_SUBSTEP_REASON_CHARS - 1).trimEnd()}…`;
|
|
451
|
+
}
|
|
452
|
+
return `${capability}: ${gist}`;
|
|
453
|
+
}
|
|
417
454
|
/**
|
|
418
455
|
* Model-declared substeps first, then the observed calls.
|
|
419
456
|
*
|
|
@@ -433,3 +470,35 @@ function mergeSubsteps(entry) {
|
|
|
433
470
|
}));
|
|
434
471
|
return [...declared, ...entry.toolSubsteps];
|
|
435
472
|
}
|
|
473
|
+
// ---------------------------------------------------------------------------
|
|
474
|
+
// Formatting
|
|
475
|
+
// ---------------------------------------------------------------------------
|
|
476
|
+
/**
|
|
477
|
+
* How long, in words. One owner, because there were three and all three were wrong.
|
|
478
|
+
*
|
|
479
|
+
* The rail, `oxygen copilot plan` and the MCP widget each carried their own copy
|
|
480
|
+
* of this, and every copy did `Math.floor(s / 60)` minutes with `Math.round(s % 60)`
|
|
481
|
+
* seconds -- which rounds the remainder INDEPENDENTLY of the minutes it was taken
|
|
482
|
+
* from. A 179.6s sub-agent therefore rendered as "2m 60s" on all three surfaces
|
|
483
|
+
* (observed live in session 34826105), and 59.6s rendered as "60s" rather than
|
|
484
|
+
* "1m". Rounding the total FIRST and splitting afterwards cannot produce either.
|
|
485
|
+
*
|
|
486
|
+
* The projection promised one implementation behind three surfaces; the numbers
|
|
487
|
+
* were shared and the words describing them were not, so this is where they meet.
|
|
488
|
+
*/
|
|
489
|
+
export function formatCopilotPlanSeconds(seconds) {
|
|
490
|
+
if (!Number.isFinite(seconds))
|
|
491
|
+
return "0s";
|
|
492
|
+
const total = Math.max(Math.round(seconds), 0);
|
|
493
|
+
if (total < 60)
|
|
494
|
+
return `${total}s`;
|
|
495
|
+
const minutes = Math.floor(total / 60);
|
|
496
|
+
const rest = total % 60;
|
|
497
|
+
return rest === 0 ? `${minutes}m` : `${minutes}m ${rest}s`;
|
|
498
|
+
}
|
|
499
|
+
/** A duration in ms, or null when there is nothing worth showing. */
|
|
500
|
+
export function formatCopilotPlanDuration(ms) {
|
|
501
|
+
if (typeof ms !== "number" || !Number.isFinite(ms) || ms <= 0)
|
|
502
|
+
return null;
|
|
503
|
+
return ms < 1000 ? `${Math.round(ms)}ms` : formatCopilotPlanSeconds(ms / 1000);
|
|
504
|
+
}
|
|
@@ -54,11 +54,42 @@ export type LlmEventBody = {
|
|
|
54
54
|
sessionId?: string | null;
|
|
55
55
|
userId?: string | null;
|
|
56
56
|
};
|
|
57
|
+
/**
|
|
58
|
+
* An eval/annotation attached to a trace (ADR 0014 files these as exactly what
|
|
59
|
+
* the LLM store exists to enable). Narrower than the API's CreateScoreRequest on
|
|
60
|
+
* purpose: `traceId` is REQUIRED (a score orphaned from its trace is unreadable
|
|
61
|
+
* in the UI), `value` is numeric-only, and `dataType` drops "CATEGORICAL"
|
|
62
|
+
* because that variant requires a *string* value. Widening later stays additive.
|
|
63
|
+
*/
|
|
64
|
+
export type LlmScoreBody = {
|
|
65
|
+
/** Deterministic id upserts; omit for a new score each call. */
|
|
66
|
+
id?: string;
|
|
67
|
+
traceId: string;
|
|
68
|
+
name: string;
|
|
69
|
+
value: number;
|
|
70
|
+
dataType?: "BOOLEAN" | "NUMERIC";
|
|
71
|
+
comment?: string;
|
|
72
|
+
};
|
|
57
73
|
export type LlmTracingClient = {
|
|
58
74
|
trace(body: LlmTraceBody): void;
|
|
59
75
|
span(body: LlmSpanBody): void;
|
|
60
76
|
generation(body: LlmGenerationBody): void;
|
|
61
77
|
event(body: LlmEventBody): void;
|
|
78
|
+
/**
|
|
79
|
+
* Attach a score to an existing trace.
|
|
80
|
+
*
|
|
81
|
+
* ASYNC, unlike the four above, because it is not the same transport. Spans go
|
|
82
|
+
* through the OTel processor, which batches and is drained by `flush()`; a
|
|
83
|
+
* score has no span, so it is one HTTP call to the scores API. Awaiting it is
|
|
84
|
+
* what keeps it alive on a serverless function that freezes the moment the
|
|
85
|
+
* handler returns — there is nothing for `flush()` to drain on its behalf.
|
|
86
|
+
*
|
|
87
|
+
* Never rejects: resolves `true` when the score was accepted, `false` when it
|
|
88
|
+
* was not (transport down, tracing misconfigured, API refusal). Fail-open like
|
|
89
|
+
* every other path here — telemetry must not break the product write it
|
|
90
|
+
* describes.
|
|
91
|
+
*/
|
|
92
|
+
score(body: LlmScoreBody): Promise<boolean>;
|
|
62
93
|
/** Never rejects; bounded at ~5s. */
|
|
63
94
|
flush(): Promise<void>;
|
|
64
95
|
/** Flush + stop background timers. Never rejects; bounded at ~5s. */
|
|
@@ -108,6 +139,10 @@ export type LlmEmitter = {
|
|
|
108
139
|
flush(): Promise<void>;
|
|
109
140
|
shutdown(): Promise<void>;
|
|
110
141
|
};
|
|
142
|
+
export type LlmScorer = {
|
|
143
|
+
/** Resolves true when the score was accepted. Never rejects. */
|
|
144
|
+
score(body: LlmScoreBody): Promise<boolean>;
|
|
145
|
+
};
|
|
111
146
|
/**
|
|
112
147
|
* Construct a fail-open Langfuse client, or `null` when tracing is disabled or
|
|
113
148
|
* misconfigured. Prefer the process-wide `getLlmTracingClient` in app code;
|
|
@@ -115,6 +150,7 @@ export type LlmEmitter = {
|
|
|
115
150
|
*/
|
|
116
151
|
export declare function createLlmTracingClient(env?: EnvMap, options?: {
|
|
117
152
|
emitterImpl?: LlmEmitter;
|
|
153
|
+
scorerImpl?: LlmScorer;
|
|
118
154
|
}): LlmTracingClient | null;
|
|
119
155
|
export declare function getLlmTracingClient(env?: EnvMap): LlmTracingClient | null;
|
|
120
156
|
/** Flush the singleton if it exists. Never rejects. Hang off request/cycle ends. */
|
|
@@ -126,6 +126,17 @@ function boundJsonField(value) {
|
|
|
126
126
|
preview: serialized.slice(0, MAX_JSON_FIELD_CHARS),
|
|
127
127
|
};
|
|
128
128
|
}
|
|
129
|
+
// A score's `comment` is free text (a thumbs-down reason is user-authored and
|
|
130
|
+
// unbounded) and the API types it as a STRING, so it cannot take
|
|
131
|
+
// boundJsonField's `{truncated, preview}` envelope. It gets the same
|
|
132
|
+
// MAX_JSON_FIELD_CHARS ceiling and the same "explicit marker, never a silent
|
|
133
|
+
// clip" rule, with the marker counted INSIDE the cap so the bound holds.
|
|
134
|
+
const COMMENT_TRUNCATION_MARKER = "…[truncated]";
|
|
135
|
+
function boundComment(comment) {
|
|
136
|
+
if (comment.length <= MAX_JSON_FIELD_CHARS)
|
|
137
|
+
return comment;
|
|
138
|
+
return comment.slice(0, MAX_JSON_FIELD_CHARS - COMMENT_TRUNCATION_MARKER.length) + COMMENT_TRUNCATION_MARKER;
|
|
139
|
+
}
|
|
129
140
|
function compact(body) {
|
|
130
141
|
const out = {};
|
|
131
142
|
for (const [key, value] of Object.entries(body)) {
|
|
@@ -223,6 +234,65 @@ function createOtelEmitter(env, warn) {
|
|
|
223
234
|
shutdown: () => init().then((h) => h?.processor.shutdown() ?? Promise.resolve()),
|
|
224
235
|
};
|
|
225
236
|
}
|
|
237
|
+
/**
|
|
238
|
+
* The real scorer: one authenticated POST to the Langfuse scores API.
|
|
239
|
+
*
|
|
240
|
+
* `@langfuse/core` is imported LAZILY on first score for the same reason the
|
|
241
|
+
* OTel tree is — a runtime with tracing disabled (the packed CLI, every test)
|
|
242
|
+
* never pays for it.
|
|
243
|
+
*
|
|
244
|
+
* The API client's `environment` option is its BASE URL, not the Langfuse
|
|
245
|
+
* environment tag; the tag is the `environment` FIELD on the score body, and it
|
|
246
|
+
* is resolved from the same helper the span processor uses so a score lands in
|
|
247
|
+
* the same Langfuse environment as the trace it scores. OXYGEN always sets
|
|
248
|
+
* LANGFUSE_BASE_URL (the project is US-region and the EU host 401s these keys);
|
|
249
|
+
* the fallback is the SDK's documented default, which `@langfuse/core` itself
|
|
250
|
+
* does not supply.
|
|
251
|
+
*/
|
|
252
|
+
function createApiScorer(env, warn) {
|
|
253
|
+
let handle = null;
|
|
254
|
+
const init = () => {
|
|
255
|
+
handle ??= (async () => {
|
|
256
|
+
try {
|
|
257
|
+
const { LangfuseAPIClient } = await import("@langfuse/core");
|
|
258
|
+
const client = new LangfuseAPIClient({
|
|
259
|
+
environment: env.LANGFUSE_BASE_URL?.trim() || "https://cloud.langfuse.com",
|
|
260
|
+
username: env.LANGFUSE_PUBLIC_KEY,
|
|
261
|
+
password: env.LANGFUSE_SECRET_KEY,
|
|
262
|
+
});
|
|
263
|
+
return client.scores;
|
|
264
|
+
}
|
|
265
|
+
catch (error) {
|
|
266
|
+
warn(error, { stage: "score_init" });
|
|
267
|
+
return null;
|
|
268
|
+
}
|
|
269
|
+
})();
|
|
270
|
+
return handle;
|
|
271
|
+
};
|
|
272
|
+
return {
|
|
273
|
+
score: async (body) => {
|
|
274
|
+
try {
|
|
275
|
+
const api = await init();
|
|
276
|
+
if (!api)
|
|
277
|
+
return false;
|
|
278
|
+
await api.create(compact({
|
|
279
|
+
id: body.id,
|
|
280
|
+
traceId: body.traceId,
|
|
281
|
+
name: body.name,
|
|
282
|
+
value: body.value,
|
|
283
|
+
dataType: body.dataType,
|
|
284
|
+
comment: typeof body.comment === "string" ? boundComment(body.comment) : undefined,
|
|
285
|
+
environment: resolveLlmTracingEnvironment(env),
|
|
286
|
+
}));
|
|
287
|
+
return true;
|
|
288
|
+
}
|
|
289
|
+
catch (error) {
|
|
290
|
+
warn(error, { stage: "score" });
|
|
291
|
+
return false;
|
|
292
|
+
}
|
|
293
|
+
},
|
|
294
|
+
};
|
|
295
|
+
}
|
|
226
296
|
/**
|
|
227
297
|
* Construct a fail-open Langfuse client, or `null` when tracing is disabled or
|
|
228
298
|
* misconfigured. Prefer the process-wide `getLlmTracingClient` in app code;
|
|
@@ -244,6 +314,7 @@ export function createLlmTracingClient(env = process.env, options) {
|
|
|
244
314
|
});
|
|
245
315
|
};
|
|
246
316
|
const emitter = options?.emitterImpl ?? createOtelEmitter(env, warn);
|
|
317
|
+
const scorer = options?.scorerImpl ?? createApiScorer(env, warn);
|
|
247
318
|
const guarded = (fn, stage) => {
|
|
248
319
|
try {
|
|
249
320
|
fn();
|
|
@@ -351,6 +422,18 @@ export function createLlmTracingClient(env = process.env, options) {
|
|
|
351
422
|
endTime: startTime,
|
|
352
423
|
});
|
|
353
424
|
}, "event"),
|
|
425
|
+
// Already fail-open inside the scorer; the extra catch is here so that a
|
|
426
|
+
// scorer which throws SYNCHRONOUSLY (a substituted one in a test, a future
|
|
427
|
+
// implementation) still cannot escape into a product write.
|
|
428
|
+
score: async (body) => {
|
|
429
|
+
try {
|
|
430
|
+
return await scorer.score(body);
|
|
431
|
+
}
|
|
432
|
+
catch (error) {
|
|
433
|
+
warn(error, { stage: "score" });
|
|
434
|
+
return false;
|
|
435
|
+
}
|
|
436
|
+
},
|
|
354
437
|
// The "never rejects, bounded at ~5s" contract is the CLIENT's, so it is
|
|
355
438
|
// enforced here rather than inside one emitter — an emitter that throws
|
|
356
439
|
// synchronously or rejects must still not escape into product code.
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* How a LinkedIn quota denial is read by the background jobs that hit it.
|
|
2
|
+
* How a LinkedIn/WhatsApp quota denial is read by the background jobs that hit it.
|
|
3
3
|
*
|
|
4
4
|
* The denial itself is raised by the chokepoint in
|
|
5
5
|
* packages/integrations/src/linkedin-quota.ts; this module is the consumer half,
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* How a LinkedIn quota denial is read by the background jobs that hit it.
|
|
2
|
+
* How a LinkedIn/WhatsApp quota denial is read by the background jobs that hit it.
|
|
3
3
|
*
|
|
4
4
|
* The denial itself is raised by the chokepoint in
|
|
5
5
|
* packages/integrations/src/linkedin-quota.ts; this module is the consumer half,
|
|
@@ -16,10 +16,21 @@
|
|
|
16
16
|
*/
|
|
17
17
|
import { OxygenError } from "./cli-result.js";
|
|
18
18
|
import { isRecord } from "./type-guards.js";
|
|
19
|
-
// The
|
|
20
|
-
//
|
|
21
|
-
//
|
|
22
|
-
|
|
19
|
+
// The codes a denial is raised with. Each is a "come back later" signal, not a
|
|
20
|
+
// broken caller: a daily cap / closed active window that reopens on its own clock,
|
|
21
|
+
// or an account the status webhook will reactivate.
|
|
22
|
+
//
|
|
23
|
+
// The WhatsApp mirrors are here because the inbox backstop is shared across
|
|
24
|
+
// networks: a WhatsApp daily-limit denial was reaching it, missing this set, and
|
|
25
|
+
// so was logged as a failure AND never parked -- the same hot re-deny loop the
|
|
26
|
+
// LinkedIn codes were added to stop. Both WhatsApp denials carry `resets_at`, so
|
|
27
|
+
// the park lands on the real reset rather than the fallback below.
|
|
28
|
+
const QUOTA_DENIED_CODES = new Set([
|
|
29
|
+
"linkedin_rate_limited",
|
|
30
|
+
"linkedin_account_unavailable",
|
|
31
|
+
"whatsapp_rate_limited",
|
|
32
|
+
"whatsapp_account_unavailable",
|
|
33
|
+
]);
|
|
23
34
|
/** Park length when a denial carries no usable `resets_at` hint. */
|
|
24
35
|
const QUOTA_FALLBACK_BACKOFF_MS = 60 * 60 * 1000;
|
|
25
36
|
/** Was this thrown error the quota chokepoint refusing the call? */
|