omnirush 0.4.1 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,9 +9,10 @@
9
9
  // Silent no-op when there are no device credentials — collection only
10
10
  // runs for signed-in omnirush identities.
11
11
 
12
+ import os from "node:os";
12
13
  import path from "node:path";
13
14
 
14
- import { gatewayUrlForOrigin, omniDir, resolveOrigin } from "./auth";
15
+ import { deviceMe, gatewayUrlForOrigin, omniDir, resolveOrigin } from "./auth";
15
16
  import { WorkspaceCollector } from "./collector-lib";
16
17
  import { sharedRefresher } from "./refresh";
17
18
  import { recordGatewayUsage, storedUsageRecord } from "./usage";
@@ -21,6 +22,14 @@ import { recordGatewayUsage, storedUsageRecord } from "./usage";
21
22
  // wedged upload still can't wedge exit.
22
23
  const SHUTDOWN_UPLOAD_BUDGET_MS = 120_000;
23
24
 
25
+ /** The pi agent config dir — session transcripts live under <dir>/sessions. */
26
+ function piAgentDir(): string {
27
+ const override =
28
+ process.env.PI_CODING_AGENT_DIR || process.env.OMNIRUSH_CODING_AGENT_DIR || "";
29
+ if (override.trim()) return path.resolve(override.trim());
30
+ return path.join(os.homedir(), ".pi", "agent");
31
+ }
32
+
24
33
  /** pi session ids are UUIDs; still, never let a foreign format stall us. */
25
34
  function sanitizeSessionId(raw: string): string {
26
35
  const id = String(raw || "").replace(/[^A-Za-z0-9._:-]/g, "-").slice(0, 128);
@@ -35,6 +44,41 @@ export default function (pi: any) {
35
44
  gatewayUrl: process.env.OMNIRUSH_GATEWAY_URL || gatewayUrlForOrigin(origin),
36
45
  accessToken,
37
46
  stateDir: omniDir(),
47
+ agentDir: piAgentDir(),
48
+ clientId: (os.hostname() || "unknown").split(/[.\\s]/)[0].slice(0, 64) || null,
49
+ identityProvider:
50
+ accessToken
51
+ ? async () => {
52
+ // Best effort: the signed-in user id rides the trace header
53
+ // (the acceptance contract for trace format spec). 401 ->
54
+ // single-flight refresh -> retry once, like every other
55
+ // Omnirush call; failures leave user_id null.
56
+ const refresher = sharedRefresher();
57
+ const token =
58
+ refresher.auth?.accessToken ||
59
+ (process.env.OMNIRUSH_TOKEN || "").trim();
60
+ if (!token) return { userId: null };
61
+ try {
62
+ let me = await deviceMe(resolveOrigin(process.env), {
63
+ accessToken: token,
64
+ fetchImpl: globalThis.fetch,
65
+ });
66
+ if (me === null) {
67
+ const refreshed = await refresher.refresh(token).catch(() => false);
68
+ const next = refresher.auth?.accessToken;
69
+ if (refreshed && next && next !== token) {
70
+ me = await deviceMe(resolveOrigin(process.env), {
71
+ accessToken: next,
72
+ fetchImpl: globalThis.fetch,
73
+ });
74
+ }
75
+ }
76
+ return { userId: typeof me?.id === "string" ? me.id : null };
77
+ } catch {
78
+ return { userId: null };
79
+ }
80
+ }
81
+ : undefined,
38
82
  refresh: accessToken
39
83
  ? async (tokenUsed: string) => {
40
84
  const refresher = sharedRefresher();
@@ -0,0 +1,144 @@
1
+ // Omnirush retry layer — shared by every Omnirush API call path (model
2
+ // stream via the sota tap, /collect uploads, /device and /device/me).
3
+ //
4
+ // Goal: users ride out a backend deploy or a rate-limit window without
5
+ // seeing a single raw error. Retryable = 429 / 5xx / network failures.
6
+ // 4xx are PROTOCOL responses (401 auth, 428 pending, 400 expired) and are
7
+ // never retried here — their handling lives at the call sites.
8
+ //
9
+ // Plain ESM JavaScript on node builtins only (same constraint as
10
+ // auth.js): loaded directly by node:test and as a sibling import from
11
+ // the TypeScript extensions through jiti.
12
+ //
13
+ // Env knobs (tests + advanced users):
14
+ // OMNIRUSH_RETRY_ATTEMPTS total attempts (default 5, min 1)
15
+ // OMNIRUSH_RETRY_BASE_MS first backoff (default 800)
16
+ // OMNIRUSH_RETRY_MAX_MS per-wait cap before jitter (default 10000)
17
+ //
18
+ // SECURITY: nothing here ever logs a token; error strings carry only
19
+ // statuses and server `detail` fragments sanitized by the callers.
20
+
21
+ /** 429 and 5xx are transient; every 4xx is a protocol answer. */
22
+ export function isRetryableStatus(status) {
23
+ return status === 429 || (status >= 500 && status <= 599);
24
+ }
25
+
26
+ /** Network failures (fetch throws) are transient; aborts are not. */
27
+ export function isRetryableError(error) {
28
+ if (!error) return false;
29
+ if (error.name === "AbortError" || error?.code === "ABORT_ERR") return false;
30
+ if (error.name === "TimeoutError") return true; // AbortSignal.timeout
31
+ return true; // fetch TypeError/DNS/ECONNRESET/... — all transient
32
+ }
33
+
34
+ function intFromEnv(name, fallback, min, max) {
35
+ const raw = Number(process.env[name]);
36
+ if (!Number.isFinite(raw)) return fallback;
37
+ return Math.min(Math.max(Math.round(raw), min), max);
38
+ }
39
+
40
+ export function retryAttempts() {
41
+ return intFromEnv("OMNIRUSH_RETRY_ATTEMPTS", 5, 1, 10);
42
+ }
43
+
44
+ export function retryBaseMs() {
45
+ return intFromEnv("OMNIRUSH_RETRY_BASE_MS", 800, 0, 60_000);
46
+ }
47
+
48
+ export function retryMaxMs() {
49
+ return intFromEnv("OMNIRUSH_RETRY_MAX_MS", 10_000, 0, 120_000);
50
+ }
51
+
52
+ /**
53
+ * Exponential backoff with full jitter: wait = random(0, min(cap,
54
+ * base * 2^(attempt-1))). `retryAfterSec` (Retry-After header) wins when
55
+ * present, capped at maxMs. attempt is 1-based (the attempt that just
56
+ * failed); attempt 1 failure waits random(0, base).
57
+ */
58
+ export function computeBackoffMs(attempt, { baseMs, maxMs, retryAfterSec } = {}) {
59
+ const base = baseMs ?? retryBaseMs();
60
+ const cap = maxMs ?? retryMaxMs();
61
+ if (cap <= 0 || base <= 0) return 0;
62
+ if (Number.isFinite(retryAfterSec) && retryAfterSec > 0) {
63
+ return Math.min(Math.round(retryAfterSec * 1000), Math.max(cap, base));
64
+ }
65
+ const exponential = Math.min(cap, base * 2 ** Math.max(0, attempt - 1));
66
+ return Math.round(Math.random() * exponential);
67
+ }
68
+
69
+ /** Local correlation id for a failure sequence: omr-<8 hex>. */
70
+ export function newCorrelationId(now = Date.now()) {
71
+ const seed = `${now}:${Math.random()}:${process.pid}`;
72
+ let hash = 0x811c9dc5;
73
+ for (let i = 0; i < seed.length; i++) {
74
+ hash ^= seed.charCodeAt(i);
75
+ hash = Math.imul(hash, 0x01000193);
76
+ }
77
+ const tail = (hash >>> 0).toString(16).padStart(8, "0");
78
+ return `omr-${tail.slice(0, 8)}`;
79
+ }
80
+
81
+ /** The final human summary shown after every attempt failed. */
82
+ export function formatUnavailableSummary({ ref, attempts, waitedMs, status } = {}) {
83
+ const waited = Number.isFinite(waitedMs) && waitedMs > 0
84
+ ? ` (waited ${Math.round(waitedMs / 1000)}s)`
85
+ : "";
86
+ const cause = status === 429
87
+ ? "Omnirush is at capacity right now"
88
+ : "Omnirush is briefly unavailable";
89
+ return (
90
+ `${cause} and could not complete this request — tried ${attempts} time${attempts === 1 ? "" : "s"}${waited}. ` +
91
+ `Please retry in a moment; your grant is not affected. [ref ${ref ?? "omr-?????"}]`
92
+ );
93
+ }
94
+
95
+ /**
96
+ * Run `tryFn` with retries. `tryFn(attempt)` returns whatever the caller
97
+ * considers a result; classification is up to the caller via:
98
+ * isRetryable(result, error) — decide whether attempt failed transiently
99
+ * `tryFn` may return { retryAfterSec } alongside its result to honor the
100
+ * header. onRetry({attempt, delayMs, result, error, ref}) observes each
101
+ * waiting round (progress lines). Resolves with the first non-retryable
102
+ * result; when attempts are exhausted it resolves with the LAST result
103
+ * if `throwOnExhausted` is false (default) — callers turn that into
104
+ * their sanitized error — otherwise throws the last error.
105
+ * `sleep` is injectable for tests (default real timer).
106
+ */
107
+ export async function withRetries(tryFn, {
108
+ attempts = retryAttempts(),
109
+ isRetryable,
110
+ onRetry,
111
+ sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
112
+ } = {}) {
113
+ const ref = newCorrelationId();
114
+ const startedAt = Date.now();
115
+ let lastResult;
116
+ let lastError;
117
+ for (let attempt = 1; attempt <= attempts; attempt++) {
118
+ lastResult = undefined;
119
+ lastError = undefined;
120
+ try {
121
+ lastResult = await tryFn(attempt, ref);
122
+ } catch (error) {
123
+ lastError = error;
124
+ }
125
+ const transient = lastError
126
+ ? isRetryableError(lastError)
127
+ : isRetryable?.(lastResult) ?? false;
128
+ if (!transient || attempt === attempts) {
129
+ return {
130
+ ok: !lastError && !transient,
131
+ result: lastResult,
132
+ error: lastError,
133
+ ref,
134
+ attempts: attempt,
135
+ waitedMs: Date.now() - startedAt,
136
+ };
137
+ }
138
+ const delayMs = computeBackoffMs(attempt, { retryAfterSec: lastResult?.retryAfterSec });
139
+ onRetry?.({ attempt, nextAttempt: attempt + 1, attempts, delayMs, ref, result: lastResult, error: lastError });
140
+ if (delayMs > 0) await sleep(delayMs);
141
+ }
142
+ // Unreachable (loop returns), kept for shape completeness.
143
+ return { ok: false, result: lastResult, error: lastError, ref, attempts, waitedMs: Date.now() - startedAt };
144
+ }
@@ -60,3 +60,67 @@ export function formatSotaWarning(expected: string, served: string): string {
60
60
  export function isSotaViolation(expected: string, served: string): boolean {
61
61
  return served !== expected;
62
62
  }
63
+
64
+ // --- gateway error sanitization --------------------------------------------
65
+
66
+ /**
67
+ * Upstream providers leak support URLs and request ids through the
68
+ * gateway ("You have hit your usage limit... help.openai.com...
69
+ * request ID: abc"). That text must never surface to omnirush users —
70
+ * it is wrong (our grants, not the user's OpenAI account), confusing,
71
+ * and leaks our upstream. Every non-OK gateway body is replaced with an
72
+ * omnirush message; the raw body stays available only behind
73
+ * OMNIRUSH_DEBUG=1 (stderr) and is traceable via the local ref.
74
+ */
75
+
76
+ /** Heuristic: does this body look like leaked upstream/provider text? */
77
+ export function looksLikeUpstreamLeak(bodyText: string): boolean {
78
+ if (!bodyText) return false;
79
+ return /help\.openai\.com|platform\.openai\.com|openai\.com|anthropic\.com|request[ _-]?id|please visit|rate[ _-]?limit exceeded|billing/i.test(
80
+ bodyText,
81
+ );
82
+ }
83
+
84
+ /** Friendly per-status message shown when the gateway fails. */
85
+ export function friendlyGatewayMessage(status: number, ref: string, attempts: number, waitedMs: number): string {
86
+ const tried = ` (tried ${attempts} time${attempts === 1 ? "" : "s"}${waitedMs > 0 ? `, waited ${Math.round(waitedMs / 1000)}s` : ""})`;
87
+ if (status === 429) {
88
+ return `Omnirush is at capacity right now${tried} — please retry in a moment; your grant is not affected. [ref ${ref}]`;
89
+ }
90
+ if (status === 401 || status === 403) {
91
+ return `Omnirush could not authenticate this request${tried} — run \`omnirush login\` again. [ref ${ref}]`;
92
+ }
93
+ if (status === 408 || status === 504) {
94
+ return `Omnirush timed out serving this request${tried} — please retry. [ref ${ref}]`;
95
+ }
96
+ return `Omnirush is briefly unavailable and could not complete this request${tried} — please retry in a moment. [ref ${ref}]`;
97
+ }
98
+
99
+ /**
100
+ * Build the SANITIZED error body the agent layer gets to see. It is
101
+ * PLAIN TEXT (not JSON) on purpose: the OpenAI SDK surfaces non-JSON
102
+ * bodies verbatim as the error message, so pi shows exactly our
103
+ * sentence — a JSON body would be re-stringified into the display by
104
+ * the provider error formatter. `debug` (OMNIRUSH_DEBUG=1) passes the
105
+ * original body through instead — full detail for support, never for
106
+ * users.
107
+ */
108
+ export function sanitizeGatewayErrorBody({
109
+ status,
110
+ bodyText,
111
+ ref,
112
+ attempts,
113
+ waitedMs,
114
+ debug,
115
+ }: {
116
+ status: number;
117
+ bodyText: string;
118
+ ref: string;
119
+ attempts: number;
120
+ waitedMs: number;
121
+ debug: boolean;
122
+ }): { bodyText: string; sanitized: boolean } {
123
+ if (debug) return { bodyText, sanitized: false };
124
+ const message = friendlyGatewayMessage(status, ref, attempts, waitedMs);
125
+ return { bodyText: message, sanitized: true };
126
+ }
@@ -11,6 +11,14 @@
11
11
  // - on a 401 from the gateway, runs the single-flight device-token
12
12
  // refresh (rotate both tokens, persisted 0600) and retries the
13
13
  // request exactly once — the GUI gateway-broker pattern.
14
+ // - on 429 / 5xx / network failures, retries with exponential backoff
15
+ // + full jitter (shared retry layer, "attempt 2/5" progress on
16
+ // stderr) so users ride out backend deploys.
17
+ // - SANITIZES every non-OK response body: raw upstream provider text
18
+ // ("help.openai.com… request ID…") never reaches the user — the
19
+ // agent layer gets an omnirush message with a local correlation id
20
+ // ([ref omr-…]); the raw body passes through only with
21
+ // OMNIRUSH_DEBUG=1.
14
22
  //
15
23
  // Registered during the extension factory, so the provider is queued and
16
24
  // becomes the composition base at runner init: every omnirush request is
@@ -18,13 +26,20 @@
18
26
 
19
27
  import fs from "node:fs";
20
28
  import { openAIResponsesApi } from "@earendil-works/pi-ai";
29
+ import { recordGatewayUsage } from "./usage";
21
30
  import { sharedRefresher } from "./refresh";
22
31
  import {
23
32
  createSseDataScanner,
24
33
  formatSotaWarning,
34
+ sanitizeGatewayErrorBody,
25
35
  servedModelFromSseData,
26
36
  } from "./sota-lib";
27
- import { recordGatewayUsage } from "./usage";
37
+ import {
38
+ formatUnavailableSummary,
39
+ isRetryableStatus,
40
+ retryAttempts,
41
+ withRetries,
42
+ } from "./retry";
28
43
 
29
44
  const PROVIDER_ID = "omnirush";
30
45
  const GATEWAY_DEFAULT = "https://omnirush.ai/omnirush/v1";
@@ -121,13 +136,44 @@ function withBearer(init: any, token: string): any {
121
136
  return { ...(init ?? {}), headers };
122
137
  }
123
138
 
139
+ /** Read a non-OK body for sanitization/debug (bounded: first 64 KiB). */
140
+ async function readErrorBody(response: any): Promise<string> {
141
+ try {
142
+ const reader = response?.body?.getReader?.();
143
+ if (!reader) return "";
144
+ const chunks: any[] = [];
145
+ let total = 0;
146
+ for (;;) {
147
+ const { done, value } = await reader.read();
148
+ if (done) break;
149
+ chunks.push(value);
150
+ total += value?.byteLength ?? 0;
151
+ if (total >= 64 * 1024) {
152
+ reader.cancel().catch(() => undefined);
153
+ break;
154
+ }
155
+ }
156
+ return new TextDecoder().decode(Buffer.concat(chunks));
157
+ } catch {
158
+ return "";
159
+ }
160
+ }
161
+
124
162
  /**
125
163
  * Wrap a fetch implementation so that:
126
164
  * - 401 responses trigger one single-flight device-token refresh and a
127
- * single retry with the rotated token (gateway-broker pattern), and
165
+ * single retry with the rotated token (gateway-broker pattern),
166
+ * - 429 / 5xx / network failures retry with exponential backoff + full
167
+ * jitter (progress lines on stderr), and
128
168
  * - SSE response bodies stream through a pass-through tap that observes
129
- * the served model. Bytes are forwarded unmodified and non-SSE
130
- * responses (errors, JSON) are returned as-is.
169
+ * the served model; non-OK response bodies are SANITIZED (upstream
170
+ * provider text never reaches the user; OMNIRUSH_DEBUG=1 passes it
171
+ * through with a stderr copy).
172
+ *
173
+ * Retries only cover the request up to (and including) the response
174
+ * headers — once an OK response body starts streaming it is forwarded
175
+ * untouched; a mid-stream failure surfaces as the provider's own error.
176
+ * Client aborts (Esc) are never retried.
131
177
  */
132
178
  function tapFetch(
133
179
  baseFetch: any,
@@ -135,33 +181,118 @@ function tapFetch(
135
181
  onViolation: (expected: string, served: string) => void,
136
182
  ): any {
137
183
  return async (input: any, init: any) => {
138
- let response = await baseFetch(input, init);
139
- debug(`provider fetch ${typeof input === "string" ? input : input?.url} -> ${response?.status}`);
140
- // Capture the gateway's grant/usage headers (x-omnirush-* plus the
141
- // x-ratelimit-*-tokens grant pair) straight from the raw response —
142
- // guaranteed availability here, independent of pi's event plumbing.
143
- try {
144
- recordGatewayUsage(response?.headers);
145
- } catch {
146
- /* usage capture must never break the request */
147
- }
148
- if (response?.status === 401) {
149
- const tokenUsed = bearerFromInit(init) || bearerFromInit(input);
150
- debug(`401 seen; bearer present: ${Boolean(tokenUsed)}`);
151
- if (tokenUsed) {
152
- const refresher = sharedRefresher();
153
- const refreshed = await refresher.refresh(tokenUsed).catch((error) => {
154
- debug(`refresh threw: ${error?.message ?? error}`);
155
- return false;
156
- });
157
- const next = refresher.auth?.accessToken;
158
- debug(`refresh ok: ${refreshed}, next token present: ${Boolean(next)}`);
159
- if (refreshed && next) {
160
- response = await baseFetch(input, withBearer(init, next));
161
- debug(`retry status: ${response?.status}`);
184
+ const requestUrl = typeof input === "string" ? input : input?.url;
185
+ const doFetch = (token?: string) =>
186
+ baseFetch(input, token ? withBearer(init, token) : init);
187
+
188
+ const outcome = await withRetries(
189
+ async (attempt, ref) => {
190
+ let response = await doFetch();
191
+ debug(`provider fetch ${requestUrl} -> ${response?.status} (attempt ${attempt})`);
192
+ // Capture the gateway's grant/usage headers (x-omnirush-* plus
193
+ // the x-ratelimit-*-tokens grant pair) straight from the raw
194
+ // response — guaranteed availability here, independent of pi's
195
+ // event plumbing.
196
+ try {
197
+ recordGatewayUsage(response?.headers);
198
+ } catch {
199
+ /* usage capture must never break the request */
162
200
  }
201
+ if (response?.status === 401) {
202
+ const tokenUsed = bearerFromInit(init) || bearerFromInit(input);
203
+ debug(`401 seen; bearer present: ${Boolean(tokenUsed)}`);
204
+ if (tokenUsed) {
205
+ const refresher = sharedRefresher();
206
+ const refreshed = await refresher.refresh(tokenUsed).catch((error: any) => {
207
+ debug(`refresh threw: ${error?.message ?? error}`);
208
+ return false;
209
+ });
210
+ const next = refresher.auth?.accessToken;
211
+ debug(`refresh ok: ${refreshed}, next token present: ${Boolean(next)}`);
212
+ if (refreshed && next) {
213
+ response = await doFetch(next);
214
+ debug(`retry-after-refresh status: ${response?.status}`);
215
+ try {
216
+ recordGatewayUsage(response?.headers);
217
+ } catch {
218
+ /* as above */
219
+ }
220
+ }
221
+ }
222
+ }
223
+ const retryAfterSec = Number(response?.headers?.get?.("retry-after")) || undefined;
224
+ return { response, ref, retryAfterSec };
225
+ },
226
+ {
227
+ attempts: retryAttempts(),
228
+ isRetryable: ({ response }: any) => {
229
+ if (init?.signal?.aborted) return false;
230
+ return isRetryableStatus(response?.status);
231
+ },
232
+ onRetry: ({ attempt, attempts, delayMs, error }) => {
233
+ const cause = error
234
+ ? "network error"
235
+ : "gateway busy";
236
+ warnStderr(
237
+ `Omnirush is briefly unavailable — ${cause}; retrying (attempt ${attempt + 1}/${attempts})…`,
238
+ );
239
+ debug(`retry ${attempt + 1}/${attempts} in ${delayMs}ms`);
240
+ },
241
+ },
242
+ );
243
+
244
+ let { response, ref, attempts, waitedMs } = {
245
+ response: outcome.result?.response,
246
+ ref: outcome.ref,
247
+ attempts: outcome.attempts,
248
+ waitedMs: outcome.waitedMs,
249
+ };
250
+
251
+ if (!response) {
252
+ // Client aborts (Esc) must surface as aborts, never as errors.
253
+ if (outcome.error && (outcome.error.name === "AbortError" || outcome.error?.code === "ABORT_ERR")) {
254
+ throw outcome.error;
255
+ }
256
+ // Every attempt threw (network dead). Synthesize a sanitized
257
+ // response so the provider layer shows our message, not a stack.
258
+ const message = formatUnavailableSummary({ ref, attempts, waitedMs });
259
+ warnStderr(`Omnirush: ${message}`);
260
+ return new Response(
261
+ JSON.stringify({ error: { message, type: "omnirush_error", code: "network" } }),
262
+ { status: 503, headers: { "content-type": "application/json" } },
263
+ );
264
+ }
265
+
266
+ if (!response.ok && response.status >= 400) {
267
+ // Error body: sanitize (raw upstream provider text must never
268
+ // surface). OMNIRUSH_DEBUG=1 passes the raw body through and
269
+ // logs it to stderr with the correlation id.
270
+ const raw = await readErrorBody(response);
271
+ const debugRaw = process.env.OMNIRUSH_DEBUG === "1";
272
+ if (debugRaw) {
273
+ warnStderr(`omnirush debug: gateway error ${response.status} [ref ${ref}] body: ${raw.slice(0, 2048)}`);
163
274
  }
275
+ const { bodyText } = sanitizeGatewayErrorBody({
276
+ status: response.status,
277
+ bodyText: raw,
278
+ ref,
279
+ attempts,
280
+ waitedMs,
281
+ debug: debugRaw,
282
+ });
283
+ response = new Response(bodyText, {
284
+ status: response.status,
285
+ statusText: response.statusText,
286
+ headers: new Headers({
287
+ "content-type": "text/plain; charset=utf-8",
288
+ ...(response.headers?.get?.("retry-after")
289
+ ? { "retry-after": response.headers.get("retry-after")! }
290
+ : {}),
291
+ }),
292
+ });
293
+ return response;
164
294
  }
295
+
165
296
  try {
166
297
  const contentType = String(response?.headers?.get?.("content-type") ?? "");
167
298
  if (!response?.body || !contentType.includes("text/event-stream")) {