@milaboratories/pl-client 3.14.4 → 3.14.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -90,6 +90,11 @@ export interface PlClientConfig {
90
90
 
91
91
  /** Value from 0 to 1, determine level of randomness to introduce to the backoff delays sequence. (0 meaning no randomness) */
92
92
  retryJitter: number;
93
+
94
+ /** Upper bound for a single backoff delay, in ms. Without it the exponential
95
+ * sequence keeps growing across all attempts: at the defaults the last sleep
96
+ * alone reaches ~66s (~150s once jitter compounds). */
97
+ retryMaxDelay: number;
93
98
  }
94
99
 
95
100
  export const DEFAULT_REQUEST_TIMEOUT = 5_000;
@@ -106,6 +111,41 @@ export const DEFAULT_RETRY_INITIAL_DELAY = 20; // 20 ms * <jitter> of sleep afte
106
111
  export const DEFAULT_RETRY_EXPONENTIAL_BACKOFF_MULTIPLIER = 1.5; // + 50% on each round
107
112
  export const DEFAULT_RETRY_LINEAR_BACKOFF_STEP = 50; // + 50 ms
108
113
  export const DEFAULT_RETRY_JITTER = 0.3; // 30%
114
+ export const DEFAULT_RETRY_MAX_DELAY = 5_000; // 5 seconds
115
+
116
+ /** Multiplier applied to the observed RTT when deriving the unary deadline. A unary call
117
+ * costs at least one round trip, plus connect/DNS/LB and server work on top, so the
118
+ * deadline needs sizeable headroom over the bare RTT. */
119
+ export const RTT_DEADLINE_MULTIPLIER = 8;
120
+
121
+ /** Ceiling for the RTT-derived unary deadline. Beyond this a call is stuck rather than
122
+ * slow, and waiting longer only delays the retry. */
123
+ export const MAX_ADAPTIVE_REQUEST_TIMEOUT = 60_000;
124
+
125
+ /** Weight of the newest RTT sample in the smoothed estimate. Low enough that one unlucky
126
+ * sample cannot swing the deadline, high enough to follow a real change. */
127
+ export const RTT_SMOOTHING_ALPHA = 0.3;
128
+
129
+ /** Login runs password hashing (and possibly an external IdP round trip) server side, so
130
+ * it is legitimately slow and gets its own deadline instead of the unary default. */
131
+ export const DEFAULT_LOGIN_TIMEOUT = 120_000;
132
+
133
+ /** Unary deadline for a given RTT estimate: floored by the configured timeout so a fast
134
+ * link behaves exactly as before, capped by {@link MAX_ADAPTIVE_REQUEST_TIMEOUT}.
135
+ * An undefined `rttMs` (no ping yet) yields the configured value unchanged. */
136
+ export function deriveUnaryDeadline(configuredMs: number, rttMs: number | undefined): number {
137
+ if (rttMs === undefined) return configuredMs;
138
+ return Math.min(
139
+ MAX_ADAPTIVE_REQUEST_TIMEOUT,
140
+ Math.max(configuredMs, Math.ceil(rttMs * RTT_DEADLINE_MULTIPLIER)),
141
+ );
142
+ }
143
+
144
+ /** Folds a new round-trip sample into an exponentially smoothed estimate. */
145
+ export function smoothRtt(previousMs: number | undefined, sampleMs: number): number {
146
+ if (previousMs === undefined) return sampleMs;
147
+ return previousMs * (1 - RTT_SMOOTHING_ALPHA) + sampleMs * RTT_SMOOTHING_ALPHA;
148
+ }
109
149
 
110
150
  export const DefaultRetryOptions: ExponentialBackoffRetryOptions = {
111
151
  type: "exponentialBackoff",
@@ -113,6 +153,7 @@ export const DefaultRetryOptions: ExponentialBackoffRetryOptions = {
113
153
  initialDelay: DEFAULT_RETRY_INITIAL_DELAY,
114
154
  backoffMultiplier: DEFAULT_RETRY_EXPONENTIAL_BACKOFF_MULTIPLIER,
115
155
  jitter: DEFAULT_RETRY_JITTER,
156
+ maxDelay: DEFAULT_RETRY_MAX_DELAY,
116
157
  };
117
158
 
118
159
  type PlConfigOverrides = Partial<
@@ -161,6 +202,7 @@ export function plAddressToConfig(
161
202
  retryExponentialBackoffMultiplier: DEFAULT_RETRY_EXPONENTIAL_BACKOFF_MULTIPLIER,
162
203
  retryLinearBackoffStep: DEFAULT_RETRY_LINEAR_BACKOFF_STEP,
163
204
  retryJitter: DEFAULT_RETRY_JITTER,
205
+ retryMaxDelay: DEFAULT_RETRY_MAX_DELAY,
164
206
 
165
207
  ...overrides,
166
208
  };
@@ -234,6 +276,7 @@ export function plAddressToConfig(
234
276
  parseInt(url.searchParams.get("retry-linear-backoff-step")) ??
235
277
  DEFAULT_RETRY_LINEAR_BACKOFF_STEP,
236
278
  retryJitter: parseInt(url.searchParams.get("retry-backoff-jitter")) ?? DEFAULT_RETRY_JITTER,
279
+ retryMaxDelay: parseInt(url.searchParams.get("retry-max-delay")) ?? DEFAULT_RETRY_MAX_DELAY,
237
280
 
238
281
  ...overrides,
239
282
  };
@@ -1,7 +1,29 @@
1
1
  import * as tp from "node:timers/promises";
2
- import { isTimeoutOrCancelError } from "./errors";
2
+ import { isTimeoutOrCancelError, isTransientCallFailure } from "./errors";
3
3
  import { test, expect } from "vitest";
4
4
 
5
+ /** Shapes the predicates match on: an RpcError-like from grpc, a RESTError-like from REST. */
6
+ function rpcError(code: string) {
7
+ return { name: "RpcError", code };
8
+ }
9
+
10
+ test("only transport failures count as transient", () => {
11
+ // Worth retrying: the call never reached a verdict.
12
+ expect(isTransientCallFailure(rpcError("UNAVAILABLE"))).toBe(true);
13
+ expect(isTransientCallFailure(rpcError("DEADLINE_EXCEEDED"))).toBe(true);
14
+
15
+ // Real answers from the server. Retrying these would hide a genuine failure and,
16
+ // for auth, silently repeat a rejected credential.
17
+ expect(isTransientCallFailure(rpcError("UNAUTHENTICATED"))).toBe(false);
18
+ expect(isTransientCallFailure(rpcError("PERMISSION_DENIED"))).toBe(false);
19
+ expect(isTransientCallFailure(rpcError("NOT_FOUND"))).toBe(false);
20
+ expect(isTransientCallFailure(rpcError("UNIMPLEMENTED"))).toBe(false);
21
+
22
+ expect(isTransientCallFailure(undefined)).toBe(false);
23
+ expect(isTransientCallFailure(null)).toBe(false);
24
+ expect(isTransientCallFailure(new Error("plain"))).toBe(false);
25
+ });
26
+
5
27
  test("timeout of sleep error type detection", async () => {
6
28
  let noError = false;
7
29
  try {
@@ -82,6 +82,13 @@ export function isTimeoutOrCancelError(err: unknown, nested: boolean = false): b
82
82
  return false;
83
83
  }
84
84
 
85
+ /** Failures worth another attempt on an idempotent call: the peer was unreachable, or the
86
+ * deadline elapsed. Everything else (auth, permission, not-found, unimplemented) is a real
87
+ * answer from the server, and retrying it only wastes the caller's time. */
88
+ export function isTransientCallFailure(err: unknown): boolean {
89
+ return isConnectionProblem(err) || isTimeoutError(err);
90
+ }
91
+
85
92
  export function isUnimplementedError(err: unknown, nested: boolean = false): boolean {
86
93
  if (err === undefined || err === null) return false;
87
94
 
@@ -13,7 +13,14 @@ import type {
13
13
  PlConnectionStatus,
14
14
  PlConnectionStatusListener,
15
15
  } from "./config";
16
- import { plAddressToConfig, type wireProtocol, SUPPORTED_WIRE_PROTOCOLS } from "./config";
16
+ import {
17
+ plAddressToConfig,
18
+ type wireProtocol,
19
+ SUPPORTED_WIRE_PROTOCOLS,
20
+ DEFAULT_LOGIN_TIMEOUT,
21
+ deriveUnaryDeadline,
22
+ smoothRtt,
23
+ } from "./config";
17
24
  import type { GrpcOptions } from "@protobuf-ts/grpc-transport";
18
25
  import { GrpcTransport } from "@protobuf-ts/grpc-transport";
19
26
  import { LLPlTransaction } from "./ll_transaction";
@@ -42,7 +49,7 @@ import {
42
49
  TxAPI_ServerMessage,
43
50
  } from "../proto-grpc/github.com/milaboratory/pl/plapi/plapiproto/api";
44
51
  import type { MiLogger } from "@milaboratories/ts-helpers";
45
- import { isAbortedError, isUnauthenticated } from "./errors";
52
+ import { isAbortedError, isTransientCallFailure, isUnauthenticated } from "./errors";
46
53
  import { Timestamp } from "../proto-grpc/google/protobuf/timestamp";
47
54
 
48
55
  export interface PlCallOps {
@@ -50,6 +57,17 @@ export interface PlCallOps {
50
57
  abortSignal?: AbortSignal;
51
58
  }
52
59
 
60
+ /** Bounded retry for idempotent unary calls. Deliberately short: these calls sit on the
61
+ * connect path and gate the UI, so a few quick attempts beat a long grind. */
62
+ const IDEMPOTENT_RETRY_OPTIONS: RetryOptions = {
63
+ type: "exponentialBackoff",
64
+ maxAttempts: 4,
65
+ initialDelay: 200,
66
+ backoffMultiplier: 2,
67
+ jitter: 0.3,
68
+ maxDelay: 2_000,
69
+ };
70
+
53
71
  // Parses leading "<major>.<minor>.<patch>" from a version string like
54
72
  // "3.1.1" or "3.1.1-rc1" and returns true if the parsed version is >= target.
55
73
  // Returns false for unparseable versions (safer to assume an old backend).
@@ -128,6 +146,15 @@ export class LLPlClient implements WireClientProviderFactory {
128
146
  private _wireProto: wireProtocol = "grpc";
129
147
  private _wireConn!: WireConnection;
130
148
 
149
+ /** The live GrpcOptions object handed to GrpcTransport. The transport keeps it by
150
+ * reference and re-reads it on every call, so mutating `timeout` here retunes
151
+ * subsequent deadlines without tearing down the connection. */
152
+ private _grpcOptions?: GrpcOptions;
153
+
154
+ /** Smoothed round-trip estimate in ms, from ping timings. Undefined until the first
155
+ * successful ping, in which case the configured deadline is used as-is. */
156
+ private _rttMs?: number;
157
+
131
158
  private readonly _restInterceptors: Dispatcher.DispatcherComposeInterceptor[];
132
159
  private readonly _restMiddlewares: Middleware[];
133
160
  private readonly _grpcInterceptors: Interceptor[];
@@ -160,7 +187,10 @@ export class LLPlClient implements WireClientProviderFactory {
160
187
  // Guarantee a ping happened so capability-gated paths (login, refresh) can branch synchronously.
161
188
  // In the autodetect path the loop's last successful ping already populated _serverInfo via the
162
189
  // side-effect in ping(); this fallback covers the path where autodetect is disabled.
163
- if (!pl._serverInfo) await pl.ping();
190
+ // Retried, not bare: this ping is the first contact with the server, so DNS and LB
191
+ // warm-up land here. A single transient failure must not fail the whole connect.
192
+ // (Not inside ping() itself: the autodetect path already wraps it in its own retry.)
193
+ if (!pl._serverInfo) await pl.withIdempotentRetry("ping", () => pl.ping());
164
194
 
165
195
  // Install the process-global signature-strictness flag based on backend version.
166
196
  setResourceSignaturesRequired(pl.supportsResourceSignatures);
@@ -266,6 +296,10 @@ export class LLPlClient implements WireClientProviderFactory {
266
296
  const clientOptions: ClientOptions = {
267
297
  "grpc.keepalive_time_ms": 30_000, // 30 seconds
268
298
  "grpc.service_config_disable_resolution": 1, // Disable DNS TXT lookups for service config
299
+ // Raises the per-stream and connection-level HTTP/2 windows. Node's 64 KB default
300
+ // caps a big single-stream read (project open) at window/RTT rather than link speed,
301
+ // and a lost WINDOW_UPDATE can deadlock it under packet loss.
302
+ "grpc-node.flow_control_window": 16 * 1024 * 1024, // 16 MiB
269
303
  interceptors: this._grpcInterceptors,
270
304
  };
271
305
 
@@ -280,7 +314,7 @@ export class LLPlClient implements WireClientProviderFactory {
280
314
  //
281
315
  const grpcOptions: GrpcOptions = {
282
316
  host: this.conf.hostAndPort,
283
- timeout: this.conf.defaultRequestTimeout,
317
+ timeout: this.unaryDeadline(),
284
318
  channelCredentials: this.conf.ssl
285
319
  ? ChannelCredentials.createSsl()
286
320
  : ChannelCredentials.createInsecure(),
@@ -305,9 +339,48 @@ export class LLPlClient implements WireClientProviderFactory {
305
339
  delete process.env.grpc_proxy;
306
340
  }
307
341
 
342
+ this._grpcOptions = grpcOptions;
308
343
  this._replaceWireConnection({ type: "grpc", Transport: new GrpcTransport(grpcOptions) });
309
344
  }
310
345
 
346
+ /** Unary deadline derived from the observed RTT, floored by the configured value so
347
+ * a fast link behaves exactly as before, and capped by
348
+ * {@link MAX_ADAPTIVE_REQUEST_TIMEOUT}. */
349
+ private unaryDeadline(): number {
350
+ return deriveUnaryDeadline(this.conf.defaultRequestTimeout, this._rttMs);
351
+ }
352
+
353
+ /** Folds a fresh round-trip sample into the estimate and retunes the live unary
354
+ * deadline. Called after every successful ping. */
355
+ private recordRtt(sampleMs: number): void {
356
+ this._rttMs = smoothRtt(this._rttMs, sampleMs);
357
+
358
+ if (this._grpcOptions) {
359
+ const next = this.unaryDeadline();
360
+ if (next !== this._grpcOptions.timeout) {
361
+ this.ops.logger?.info(
362
+ `Unary deadline retuned to ${next}ms (rtt estimate ${Math.round(this._rttMs)}ms)`,
363
+ );
364
+ this._grpcOptions.timeout = next;
365
+ }
366
+ }
367
+ }
368
+
369
+ /** Smoothed round-trip estimate in ms, or undefined before the first ping. */
370
+ public get rttEstimateMs(): number | undefined {
371
+ return this._rttMs;
372
+ }
373
+
374
+ /** Runs an idempotent unary call with a bounded retry on transient transport failures.
375
+ * Safe only for calls with no side effects, since a retry may duplicate the request. */
376
+ private async withIdempotentRetry<T>(name: string, cb: () => Promise<T>): Promise<T> {
377
+ return await retry(cb, IDEMPOTENT_RETRY_OPTIONS, (e: unknown) => {
378
+ if (!isTransientCallFailure(e)) return false;
379
+ this.ops.logger?.info(`${name}: transient failure, retrying. err=${String(e)}`);
380
+ return true;
381
+ });
382
+ }
383
+
311
384
  private _replaceWireConnection(newConn: WireConnection): void {
312
385
  const oldConn = this._wireConn;
313
386
  this._wireConn = newConn;
@@ -558,7 +631,7 @@ export class LLPlClient implements WireClientProviderFactory {
558
631
  expiration: { seconds: ttlSeconds, nanos: 0 },
559
632
  requestedRole: role,
560
633
  },
561
- { meta },
634
+ { meta, timeout: DEFAULT_LOGIN_TIMEOUT },
562
635
  ).response
563
636
  ).token;
564
637
  } else {
@@ -584,14 +657,17 @@ export class LLPlClient implements WireClientProviderFactory {
584
657
 
585
658
  if (cl instanceof GrpcPlApiClient) {
586
659
  return (
587
- await cl.login({
588
- credentials: {
589
- oneofKind: "basic",
590
- basic: { login: user, password },
660
+ await cl.login(
661
+ {
662
+ credentials: {
663
+ oneofKind: "basic",
664
+ basic: { login: user, password },
665
+ },
666
+ expiration: { seconds: ttl, nanos: 0 },
667
+ requestedRole: role,
591
668
  },
592
- expiration: { seconds: ttl, nanos: 0 },
593
- requestedRole: role,
594
- }).response
669
+ { timeout: DEFAULT_LOGIN_TIMEOUT },
670
+ ).response
595
671
  ).token;
596
672
  } else {
597
673
  const resp = cl.POST("/v1/auth/login", {
@@ -620,14 +696,17 @@ export class LLPlClient implements WireClientProviderFactory {
620
696
 
621
697
  if (cl instanceof GrpcPlApiClient) {
622
698
  return (
623
- await cl.login({
624
- credentials: {
625
- oneofKind: "token",
626
- token: { token: bytes },
699
+ await cl.login(
700
+ {
701
+ credentials: {
702
+ oneofKind: "token",
703
+ token: { token: bytes },
704
+ },
705
+ expiration: { seconds: ttl, nanos: 0 },
706
+ requestedRole: role,
627
707
  },
628
- expiration: { seconds: ttl, nanos: 0 },
629
- requestedRole: role,
630
- }).response
708
+ { timeout: DEFAULT_LOGIN_TIMEOUT },
709
+ ).response
631
710
  ).token;
632
711
  } else {
633
712
  const resp = cl.POST("/v1/auth/login", {
@@ -649,12 +728,15 @@ export class LLPlClient implements WireClientProviderFactory {
649
728
  const cl = this.clientProvider.get();
650
729
  if (cl instanceof GrpcPlApiClient) {
651
730
  return (
652
- await cl.login({
653
- credentials: {
654
- oneofKind: "sso",
655
- sso: { tokenResponse },
731
+ await cl.login(
732
+ {
733
+ credentials: {
734
+ oneofKind: "sso",
735
+ sso: { tokenResponse },
736
+ },
656
737
  },
657
- }).response
738
+ { timeout: DEFAULT_LOGIN_TIMEOUT },
739
+ ).response
658
740
  ).token;
659
741
  } else {
660
742
  const resp = cl.POST("/v1/auth/login", {
@@ -707,10 +789,13 @@ export class LLPlClient implements WireClientProviderFactory {
707
789
 
708
790
  if (cl instanceof GrpcPlApiClient) {
709
791
  return (
710
- await cl.refreshToken({
711
- token: currentToken,
712
- expiration: { seconds: ttl, nanos: 0 },
713
- }).response
792
+ await cl.refreshToken(
793
+ {
794
+ token: currentToken,
795
+ expiration: { seconds: ttl, nanos: 0 },
796
+ },
797
+ { timeout: DEFAULT_LOGIN_TIMEOUT },
798
+ ).response
714
799
  ).token;
715
800
  } else {
716
801
  const resp = cl.POST("/v1/auth/refresh", {
@@ -723,6 +808,8 @@ export class LLPlClient implements WireClientProviderFactory {
723
808
  public async ping(): Promise<grpcTypes.MaintenanceAPI_Ping_Response> {
724
809
  const cl = this.clientProvider.get();
725
810
  let resp: grpcTypes.MaintenanceAPI_Ping_Response;
811
+ // Ping is the cheapest call we make, so its duration is our best RTT proxy.
812
+ const startedAt = performance.now();
726
813
  if (cl instanceof GrpcPlApiClient) {
727
814
  resp = (await cl.ping({})).response;
728
815
  } else {
@@ -737,6 +824,7 @@ export class LLPlClient implements WireClientProviderFactory {
737
824
  capabilities: (pingData as any).capabilities ?? [],
738
825
  };
739
826
  }
827
+ this.recordRtt(performance.now() - startedAt);
740
828
  this._serverInfo = resp;
741
829
  return resp;
742
830
  }
@@ -895,6 +983,14 @@ export class LLPlClient implements WireClientProviderFactory {
895
983
 
896
984
  public async getUserRoot(
897
985
  opts: { login?: string; createIfNotExists?: boolean } = {},
986
+ ): Promise<grpcTypes.AuthAPI_GetUserRoot_Response> {
987
+ // Retryable even with createIfNotExists: the call is get-or-create keyed on login, so a
988
+ // second attempt returns the existing root rather than making another one.
989
+ return await this.withIdempotentRetry("getUserRoot", () => this.getUserRootOnce(opts));
990
+ }
991
+
992
+ private async getUserRootOnce(
993
+ opts: { login?: string; createIfNotExists?: boolean } = {},
898
994
  ): Promise<grpcTypes.AuthAPI_GetUserRoot_Response> {
899
995
  const cl = this.clientProvider.get();
900
996
  if (cl instanceof GrpcPlApiClient) {