@maci0/dsh-google-vertex 0.0.0-stage → 0.12.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js ADDED
@@ -0,0 +1,196 @@
1
+ /**
2
+ * dsh-google-vertex: use Google-hosted models from Vertex AI inside DeepSeek
3
+ * Harness, authenticated with a service-account file.
4
+ *
5
+ * Two capabilities, one configuration row: an `ctx.llm` provider adapter for
6
+ * Vertex's Anthropic publisher endpoint (Claude) and one for its Google
7
+ * publisher endpoint (Gemini). Registering them makes both routes selectable in
8
+ * the Web client's model picker, because `buildModelCatalog()` enumerates
9
+ * `ctx.llm.listProviders()` and asks each adapter for `listModels()` /
10
+ * `resolveModel()`.
11
+ *
12
+ * A third: the `google-vertex` settings namespace. The Gemini adapter discovers
13
+ * its catalog at runtime behind a five-minute cache, and a browser half has no
14
+ * other way to reach this process, so the namespace's one write (the Refresh
15
+ * control on this plugin's row page under Plugins) is what drops that cache
16
+ * on demand.
17
+ *
18
+ * The credential is the service-account JSON itself: a path in configuration,
19
+ * or `GOOGLE_APPLICATION_CREDENTIALS` in the launch environment. Nothing is
20
+ * copied into the harness credential store, and the file is read once at mount
21
+ * so a typo fails loudly instead of on the first message.
22
+ *
23
+ * @module dsh-google-vertex
24
+ */
25
+ import Schema from '@deepseek-ai/schemastery';
26
+ import { GoogleVertexAnthropicAdapter } from './adapter.js';
27
+ import { loadServiceAccount } from './auth.js';
28
+ import { ServiceAccountTokens } from './auth.js';
29
+ import { fetchGeminiModels } from './discovery.js';
30
+ import { DEFAULT_GEMINI_CONTEXT_WINDOW, DEFAULT_GEMINI_MAX_TOKENS, DEFAULT_GEMINI_MODELS } from './gemini.js';
31
+ import { GoogleVertexGeminiAdapter } from './gemini_adapter.js';
32
+ import { DEFAULT_LOCATION, DEFAULT_STREAM_IDLE_TIMEOUT_MS, MAX_TIMER_DELAY_MS } from './wire.js';
33
+ /** Plugin name as it appears in the loader. */
34
+ export const name = 'google-vertex';
35
+ /** The `ctx.llm` route serving Google-hosted Anthropic (Claude) models. */
36
+ export const PROVIDER = 'google-vertex-anthropic';
37
+ /** The `ctx.llm` route serving Gemini models. */
38
+ export const GEMINI_PROVIDER = 'google-vertex-gemini';
39
+ /**
40
+ * Settings namespace the browser half's card edits: the join key between the
41
+ * two halves. The card registers into `plugins.row.config` under this namespace,
42
+ * and the settings tab pairs the two without knowing what the namespace means.
43
+ */
44
+ export const GOOGLE_VERTEX_SETTINGS_NAMESPACE = 'google-vertex';
45
+ /** The one service this plugin needs mounted. */
46
+ export const inject = ['llm'];
47
+ /**
48
+ * Claude models Vertex serves, in picker order. Ids are the provider's own
49
+ * aliases, which track the newest dated release of each family and need no
50
+ * edit when Vertex promotes one; the catalog is overridable from configuration
51
+ * for a deployment that pins dated versions or uses a different listing.
52
+ */
53
+ export const DEFAULT_MODELS = [
54
+ { id: 'claude-opus-4-6', name: 'Claude Opus 4.6 (Vertex)' },
55
+ { id: 'claude-sonnet-4-6', name: 'Claude Sonnet 4.6 (Vertex)' },
56
+ { id: 'claude-opus-4-5', name: 'Claude Opus 4.5 (Vertex)' },
57
+ { id: 'claude-sonnet-4-5', name: 'Claude Sonnet 4.5 (Vertex)' },
58
+ { id: 'claude-haiku-4-5', name: 'Claude Haiku 4.5 (Vertex)' },
59
+ ];
60
+ /** Context capacity reported for every Claude model; every listed family serves 200k. */
61
+ export const DEFAULT_CONTEXT_WINDOW = 200_000;
62
+ /**
63
+ * Output cap used when a caller omits one. Well under the models' 64k ceiling:
64
+ * an agent turn rarely needs more, and the harness enforces its own budget.
65
+ */
66
+ export const DEFAULT_MAX_TOKENS = 32_000;
67
+ /**
68
+ * Row schema: defaults live here, so a deployment only states what it changes.
69
+ *
70
+ * `serviceAccountFile`, `project`, `models`, and `geminiModels` carry no
71
+ * `.default()`: the first two fall back to the environment, and an omitted
72
+ * catalog materializes empty, which `resolveConfig` treats exactly like an
73
+ * absent one and replaces with the built-in list.
74
+ *
75
+ * `revalidatedAt` is volatile (the only kind of field the settings document
76
+ * accepts) and carries no default: absence means "never refreshed manually".
77
+ */
78
+ export const Config = Schema.object({
79
+ serviceAccountFile: Schema.string(),
80
+ project: Schema.string(),
81
+ location: Schema.string().default(DEFAULT_LOCATION),
82
+ models: Schema.array(Schema.string()),
83
+ geminiModels: Schema.array(Schema.string()),
84
+ contextWindow: Schema.number().step(1).min(1).default(DEFAULT_CONTEXT_WINDOW),
85
+ maxTokens: Schema.number().step(1).min(1).default(DEFAULT_MAX_TOKENS),
86
+ streamIdleTimeoutMs: Schema.number().min(Number.MIN_VALUE).max(MAX_TIMER_DELAY_MS).default(DEFAULT_STREAM_IDLE_TIMEOUT_MS),
87
+ revalidatedAt: Schema.string().volatile(),
88
+ });
89
+ /**
90
+ * Validate and normalize one configuration row.
91
+ *
92
+ * Invalid values throw rather than being silently defaulted: a typo'd path or
93
+ * project would otherwise present as an opaque provider error mid-turn.
94
+ * @param config - raw row values, as a profile patch or an unwrapped row.
95
+ * @param env - environment consulted for the credential and region defaults.
96
+ * @returns the resolved adapter configuration.
97
+ */
98
+ export function resolveConfig(config = {}, env = process.env) {
99
+ // The exported schema is the one source of the numeric defaults and bounds.
100
+ // The credential path, the project, and the region keep their environment
101
+ // fallbacks, so they are read off the raw row instead.
102
+ const filled = Config(config);
103
+ const serviceAccountFile = filled.serviceAccountFile ?? env['GOOGLE_APPLICATION_CREDENTIALS'];
104
+ if (serviceAccountFile === undefined || serviceAccountFile.trim().length === 0) {
105
+ throw new Error('google-vertex: no service account configured: set serviceAccountFile in this plugin\'s row,'
106
+ + ' or export GOOGLE_APPLICATION_CREDENTIALS to a service-account JSON path');
107
+ }
108
+ const serviceAccount = loadServiceAccount(serviceAccountFile);
109
+ const project = filled.project
110
+ ?? env['GOOGLE_CLOUD_PROJECT']
111
+ ?? env['GCLOUD_PROJECT']
112
+ ?? serviceAccount.project_id;
113
+ if (project === undefined || project.trim().length === 0) {
114
+ throw new Error(`google-vertex: no project configured: set project in this plugin's row, or export`
115
+ + ` GOOGLE_CLOUD_PROJECT, or use a service-account file that carries "project_id"`);
116
+ }
117
+ const location = config.location ?? env['GOOGLE_CLOUD_LOCATION'] ?? DEFAULT_LOCATION;
118
+ if (location.trim().length === 0 || !/^[a-z0-9-]+$/.test(location)) {
119
+ throw new Error(`google-vertex: location "${location}" is not a valid Vertex region`);
120
+ }
121
+ const ids = filled.models === undefined || filled.models.length === 0
122
+ ? DEFAULT_MODELS
123
+ : filled.models.map(id => ({ id, name: id }));
124
+ for (const model of ids) {
125
+ if (model.id.trim().length === 0)
126
+ throw new Error('google-vertex: model ids must be non-empty strings');
127
+ }
128
+ const geminiIds = filled.geminiModels === undefined || filled.geminiModels.length === 0
129
+ ? DEFAULT_GEMINI_MODELS.map(model => model.id)
130
+ : filled.geminiModels;
131
+ const geminiModels = geminiIds.map((id) => {
132
+ if (id.trim().length === 0)
133
+ throw new Error('google-vertex: geminiModels ids must be non-empty strings');
134
+ // A configured id keeps the built-in entry's wording when it is one of them;
135
+ // every model serves the same capacities, which live on the wire config.
136
+ return DEFAULT_GEMINI_MODELS.find(model => model.id === id) ?? { id, name: id };
137
+ });
138
+ const streamIdleTimeoutMs = filled.streamIdleTimeoutMs;
139
+ return {
140
+ serviceAccountFile,
141
+ anthropic: {
142
+ serviceAccount,
143
+ project,
144
+ location,
145
+ models: ids,
146
+ contextWindow: filled.contextWindow,
147
+ maxTokens: filled.maxTokens,
148
+ streamIdleTimeoutMs,
149
+ },
150
+ gemini: {
151
+ serviceAccount,
152
+ project,
153
+ location,
154
+ models: geminiModels,
155
+ maxTokens: DEFAULT_GEMINI_MAX_TOKENS,
156
+ streamIdleTimeoutMs,
157
+ },
158
+ };
159
+ }
160
+ /**
161
+ * Mount both adapters.
162
+ * @param ctx - host context; `ctx.llm` must be mounted (`inject` guarantees it).
163
+ * @param config - this plugin's row, as the schema emits it.
164
+ */
165
+ export function apply(ctx, config) {
166
+ // The volatile stamp stays a live reference: the schema refuses one, and the
167
+ // host never reads the value, so the plain row is what gets resolved.
168
+ const { revalidatedAt: _revalidation, ...row } = config;
169
+ const resolved = resolveConfig(row);
170
+ const { anthropic, gemini } = resolved;
171
+ // Only the Gemini route discovers: Vertex has no Anthropic listing endpoint,
172
+ // so the Claude catalog is the configured (or built-in) list itself. A
173
+ // configured `geminiModels` is likewise served as written, never replaced by
174
+ // the discovered catalog.
175
+ const fetchFn = (input, init) => globalThis.fetch(input, init);
176
+ const tokenSource = new ServiceAccountTokens(anthropic.serviceAccount, { fetch: fetchFn });
177
+ const pinnedGemini = row.geminiModels !== undefined && row.geminiModels.length > 0;
178
+ const discoverGemini = pinnedGemini
179
+ ? undefined
180
+ : () => fetchGeminiModels(gemini.location, tokenSource, fetchFn, AbortSignal.timeout(Math.ceil(gemini.streamIdleTimeoutMs)));
181
+ const anthropicAdapter = new GoogleVertexAnthropicAdapter(anthropic);
182
+ const geminiAdapter = new GoogleVertexGeminiAdapter(gemini, discoverGemini === undefined ? {} : { discover: discoverGemini });
183
+ ctx.llm.registerAdapter([PROVIDER], anthropicAdapter);
184
+ ctx.llm.registerAdapter([GEMINI_PROVIDER], geminiAdapter);
185
+ // A profile edit of `revalidatedAt` drops both cached catalogs.
186
+ if (typeof ctx.on === 'function') {
187
+ ctx.on('loader/volatile-update', () => {
188
+ anthropicAdapter.invalidateModels();
189
+ geminiAdapter.invalidateModels();
190
+ });
191
+ }
192
+ ctx.logger.info(`google-vertex: providers "${PROVIDER}" and "${GEMINI_PROVIDER}" registered for project ${anthropic.project}`
193
+ + ` (location ${anthropic.location}, credentials ${resolved.serviceAccountFile})`
194
+ + `. Claude: ${anthropic.models.map(model => model.id).join(', ')}`
195
+ + `. Gemini: ${gemini.models.map(model => model.id).join(', ')}${pinnedGemini ? '' : ' (+ live discovery)'}`);
196
+ }
@@ -0,0 +1,219 @@
1
+ /**
2
+ * The machinery both Vertex publisher adapters share, plus the Claude route's
3
+ * own adapter: one metadata face the harness calls on every dispatch, and one
4
+ * streaming pipeline carrying the per-read idle watchdog, the token mint, the
5
+ * fetch, the refusal classification, the SSE loop, and the single-terminal-chunk
6
+ * guarantee.
7
+ *
8
+ * Two things separate the Claude route from the pi-ai-backed `google-vertex`
9
+ * route the harness already ships. Its credential is a service-account file,
10
+ * turned into a bearer token per request by {@link ServiceAccountTokens} rather
11
+ * than typed in as an API key. And its models are Claude, which Vertex serves
12
+ * through `publishers/anthropic`, a path and body the Gemini protocol cannot
13
+ * express.
14
+ *
15
+ * Both routes are text-only: `inputModalities: ['text']` makes `LlmRuntime`
16
+ * project images and files to placeholder text before dispatch, which is honest
17
+ * about what these adapters send.
18
+ *
19
+ * The two routes differ in exactly three things: how a request is addressed
20
+ * and built, what the provider's payloads mean, and what a body that ends
21
+ * without a terminal event should be called. Each is an injected callback;
22
+ * everything else lives here once.
23
+ *
24
+ * @module dsh-google-vertex/adapter
25
+ */
26
+ import type { FetchLike, ServiceAccount } from './auth.ts';
27
+ import type { GenerateOptions, LlmAdapterLike, LlmModelInfo, LlmProviderInfo, LlmResolvedModelInfo, PreparedAdapterCall, StreamChunk } from './host.ts';
28
+ /** One advertised model. */
29
+ export interface VertexModel {
30
+ readonly id: string;
31
+ readonly name: string;
32
+ }
33
+ /**
34
+ * The token source the adapter asks for a bearer token before each request.
35
+ * Structural so a test can answer with a fixed token; {@link ServiceAccountTokens}
36
+ * is the production implementation.
37
+ */
38
+ export interface TokenProvider {
39
+ get(signal?: AbortSignal): Promise<string>;
40
+ }
41
+ /**
42
+ * Configuration every Vertex adapter row carries. The Claude route adds only
43
+ * its context window; the model capacities of a Gemini route live on the
44
+ * catalog entries, which is why `models` is typed loosely here.
45
+ */
46
+ interface VertexAdapterConfig {
47
+ readonly serviceAccount: ServiceAccount;
48
+ readonly project: string;
49
+ readonly location: string;
50
+ readonly models: readonly VertexModel[];
51
+ /** Output cap applied when a caller omits `maxTokens`. */
52
+ readonly maxTokens: number;
53
+ /** Bound on the interval between two stream reads, in milliseconds. */
54
+ readonly streamIdleTimeoutMs: number;
55
+ }
56
+ /** Resolved adapter configuration for the Claude route. */
57
+ export interface VertexAnthropicConfig extends VertexAdapterConfig {
58
+ /** Maximum combined request and response context reported for every model. */
59
+ readonly contextWindow: number;
60
+ }
61
+ /**
62
+ * What separates one publisher route from the other, as one parameter set: the
63
+ * display name, the description wording, and the capacities `resolveModel`
64
+ * reports for a model id.
65
+ */
66
+ interface AdapterMetadata<C extends VertexAdapterConfig> {
67
+ /** Display name the model picker shows for this route. */
68
+ readonly providerName: string;
69
+ /** Context window and output cap reported for every model on this route. */
70
+ readonly capacity: {
71
+ contextWindow: number;
72
+ defaultMaxTokens: number;
73
+ };
74
+ /** Model description reported for this route, project and region included. */
75
+ readonly describe: (config: C) => string;
76
+ }
77
+ /**
78
+ * One provider translator, as the shared pump reads it.
79
+ *
80
+ * `handle` is the only method that can emit a terminal chunk: an Anthropic
81
+ * `message_stop` and a Gemini in-band error both do. `terminal` then reports
82
+ * that it, or the watchdog, already ended the turn, so the pump never adds a
83
+ * second finish. A route whose terminal event is not the only way a body ends
84
+ * (Gemini streams whole chunks and simply stops) narrows the truncation
85
+ * question with `sawFinish`: did the provider itself report that the turn was
86
+ * complete? A stream that ends without one is truncated. When absent, the pump
87
+ * falls back to `terminal`.
88
+ */
89
+ interface StreamTranslatorLike {
90
+ handle(event: Record<string, unknown>): Iterable<StreamChunk>;
91
+ readonly terminal: boolean;
92
+ readonly sawFinish?: boolean;
93
+ /** Terminal chunks a provider that closes its body with data needs. */
94
+ finish?(): Iterable<StreamChunk>;
95
+ }
96
+ /** Everything one streaming call varies between the two publisher routes. */
97
+ interface StreamPumpOptions {
98
+ /** The request URL for the chosen model. */
99
+ readonly endpoint: (model: string, config: VertexAdapterConfig) => string;
100
+ /** The request body for the chosen model. */
101
+ readonly body: (options: GenerateOptions, config: VertexAdapterConfig) => unknown;
102
+ /** Failure named when the body ended without the provider's finish. */
103
+ readonly truncatedMessage: (model: string) => string;
104
+ }
105
+ /**
106
+ * One streaming completion, shared by both publisher routes.
107
+ *
108
+ * The bearer token is minted before the request, so a credentials failure is
109
+ * reported as a terminal finish rather than as a thrown error escaping the
110
+ * generator; `LlmRuntime` would turn a throw into a finish too, but would code
111
+ * it `UNKNOWN` and lose the `AUTH` versus `TRANSPORT` classification.
112
+ *
113
+ * Every read is bounded by `streamIdleTimeoutMs`: a provider that stops sending
114
+ * is a terminal `TIMEOUT` rather than a turn that never ends. The watchdog owns
115
+ * its own controller so the stalled read can be torn down; the caller's signal
116
+ * is combined with it when present. The token mint is armed and cleared with
117
+ * the same bound, because it is a network call too.
118
+ *
119
+ * Exactly one terminal chunk leaves this stream. A body that ended without the
120
+ * provider's terminal event is a truncated response, unless the watchdog or the
121
+ * caller ended the turn, which is the more specific account of the same missing
122
+ * event.
123
+ * @param config - the resolved configuration this adapter serves.
124
+ * @param options - the harness request.
125
+ * @param fetch - transport, already defaulted by the adapter.
126
+ * @param tokens - token source, already defaulted by the adapter.
127
+ * @param makeTranslator - builds this route's payload translator for one model.
128
+ * @param pump - this route's endpoint, body, and terminal wording.
129
+ * @yields every chunk the provider's stream completes.
130
+ */
131
+ export declare function streamVertex(config: VertexAdapterConfig, options: GenerateOptions, fetch: FetchLike, tokens: TokenProvider, makeTranslator: (model: string) => StreamTranslatorLike, pump: StreamPumpOptions): AsyncGenerator<StreamChunk>;
132
+ /**
133
+ * Duck-typed base for both Vertex publisher adapters.
134
+ *
135
+ * `LlmRuntime` reaches adapters through plain method calls, so these objects
136
+ * need no harness base class; the plugin's only runtime `@deepseek-ai/*`
137
+ * dependency is `@deepseek-ai/dsh-llm`'s pure `attributionHeaders()` helper.
138
+ * The metadata face below is identical for both routes, so it is written once
139
+ * and parameterized by {@link AdapterMetadata}.
140
+ */
141
+ export declare abstract class VertexPublisherAdapter<C extends VertexAdapterConfig> implements LlmAdapterLike {
142
+ #private;
143
+ /** The config this adapter serves, for the stream pipeline below. */
144
+ protected readonly config: C;
145
+ /** The transport this adapter was built with. */
146
+ protected readonly fetch: FetchLike;
147
+ /** The token source this adapter asks before each request. */
148
+ protected readonly tokens: TokenProvider;
149
+ /**
150
+ * @param metadata - display name, capacities, and description text.
151
+ * @param config - the resolved configuration this adapter serves.
152
+ * @param options - transport, token-source, and discovery overrides.
153
+ */
154
+ constructor(metadata: AdapterMetadata<C>, config: C, options?: {
155
+ fetch?: FetchLike;
156
+ tokens?: TokenProvider;
157
+ discover?: () => Promise<readonly VertexModel[]>;
158
+ });
159
+ /** {@inheritDoc LlmAdapterLike.providerInfo} */
160
+ providerInfo(provider: string): LlmProviderInfo;
161
+ /**
162
+ * No provider-owned retry policy: Vertex's own 429s carry a quota code the
163
+ * harness already classifies, and a per-request backoff belongs to the
164
+ * harness defaults.
165
+ *
166
+ * This and {@link imageRequestPricing} exist because `LlmRuntime` calls them
167
+ * on every dispatch; a duck-typed adapter must supply them or the first
168
+ * registration throws.
169
+ */
170
+ providerRetryPolicy(_provider: string): undefined;
171
+ /** No route charges visual tokens: these adapters are text-only. */
172
+ imageRequestPricing(_provider: string, _model: string): undefined;
173
+ /**
174
+ * The current model catalog, fetched from the provider when discovery is
175
+ * configured, falling back to the static catalog on failure.
176
+ *
177
+ * The result is cached with a five-minute TTL so the model picker does
178
+ * not make a network call on every open.
179
+ */
180
+ listModels(provider: string): Promise<readonly LlmModelInfo[]>;
181
+ /**
182
+ * Drop the cached catalog, so the next `listModels` re-discovers from the
183
+ * provider. Both adapters answer the plugin's manual refresh with this.
184
+ */
185
+ invalidateModels(): void;
186
+ /** {@inheritDoc LlmAdapterLike.resolveModel} */
187
+ resolveModel(provider: string, model: string, _signal?: AbortSignal): Promise<LlmResolvedModelInfo>;
188
+ /** {@inheritDoc LlmAdapterLike.prepareCall} */
189
+ prepareCall(provider: string, model: string, signal?: AbortSignal): Promise<PreparedAdapterCall>;
190
+ /** Stream one completion through this route's publisher endpoint. */
191
+ abstract stream(options: GenerateOptions): AsyncIterable<StreamChunk>;
192
+ }
193
+ /**
194
+ * Duck-typed adapter over Vertex's Anthropic publisher endpoint.
195
+ *
196
+ * The route is text-only and declares no server-executed tools, and every
197
+ * Claude family in the default catalog serves the same 200k context, so a
198
+ * model's capacities are the configured pair rather than a per-model entry.
199
+ */
200
+ export declare class GoogleVertexAnthropicAdapter extends VertexPublisherAdapter<VertexAnthropicConfig> {
201
+ /**
202
+ * @param config - the resolved configuration this adapter serves.
203
+ * @param options - transport, token-source, and discovery overrides for tests.
204
+ */
205
+ constructor(config: VertexAnthropicConfig, options?: {
206
+ fetch?: FetchLike;
207
+ tokens?: TokenProvider;
208
+ discover?: () => Promise<readonly VertexModel[]>;
209
+ });
210
+ /**
211
+ * Stream one completion through `:streamRawPredict`.
212
+ *
213
+ * The shared pump owns the watchdog, the token mint, the SSE loop, and the
214
+ * single terminal chunk; `message_stop` closes the turn mid-body, so this
215
+ * route's finish rides that event.
216
+ */
217
+ stream(options: GenerateOptions): AsyncIterable<StreamChunk>;
218
+ }
219
+ export {};
@@ -0,0 +1,111 @@
1
+ /**
2
+ * Service-account authentication for Google APIs: read one credentials file,
3
+ * sign a JWT with its private key, trade it for an OAuth access token, and
4
+ * cache that token until shortly before it expires.
5
+ *
6
+ * The harness credential plane stores API keys, while a Vertex deployment
7
+ * typically holds a service-account JSON instead: the file is the credential.
8
+ * That is why this module exists rather than a credential reference: nothing in
9
+ * the harness can turn an RSA key into a bearer token on the request path.
10
+ *
11
+ * Everything here is pure enough to test without a network: the token endpoint
12
+ * transport and the clock are injectable. The scope every request needs and the
13
+ * early-refresh margin are fixed, because nothing has ever needed to vary them:
14
+ * one Vertex route and one margin are what this plugin talks to.
15
+ *
16
+ * @module dsh-google-vertex/auth
17
+ */
18
+ /** Scope every Vertex AI request needs; the token carries nothing narrower. */
19
+ export declare const CLOUD_PLATFORM_SCOPE = "https://www.googleapis.com/auth/cloud-platform";
20
+ /** Injectable transport, so tests never touch the network. */
21
+ export type FetchLike = (input: string, init: RequestInit) => Promise<Response>;
22
+ /** The fields of a service-account JSON this module reads. */
23
+ export interface ServiceAccount {
24
+ readonly client_email: string;
25
+ readonly private_key: string;
26
+ readonly private_key_id?: string;
27
+ readonly project_id?: string;
28
+ readonly token_uri?: string;
29
+ }
30
+ /** An auth failure the adapter reports as `AUTH` or `TRANSPORT`. */
31
+ export declare class VertexAuthError extends Error {
32
+ /** Provider-neutral failure code the adapter forwards. */
33
+ readonly code: 'AUTH' | 'TRANSPORT';
34
+ /**
35
+ * @param code - whether the failure is a credential problem or a transport one.
36
+ * @param message - operator-facing detail.
37
+ */
38
+ constructor(code: 'AUTH' | 'TRANSPORT', message: string);
39
+ }
40
+ /**
41
+ * Expand a leading `~` to the process home directory.
42
+ *
43
+ * The path arrives from configuration, which is written by a human the same way
44
+ * a shell is: `~/.secrets/vertex.json` means the home directory, not a literal
45
+ * directory named `~`.
46
+ * @param path - configured path, possibly home-relative.
47
+ * @returns an absolute path.
48
+ */
49
+ export declare function expandHome(path: string): string;
50
+ /**
51
+ * Parse a service-account document, refusing one this module cannot sign with.
52
+ *
53
+ * `name` is the path the text came from, so the failure tells an operator which
54
+ * file to fix rather than only which field is wrong.
55
+ * @param raw - the file's text.
56
+ * @param name - source path used in failure messages.
57
+ * @returns the parsed account.
58
+ * @throws {Error} when the document is not JSON or lacks a signing key.
59
+ */
60
+ export declare function parseServiceAccount(raw: string, name: string): ServiceAccount;
61
+ /**
62
+ * Read and parse one service-account file.
63
+ *
64
+ * Read synchronously on purpose: this runs once at plugin mount, where a typo'd
65
+ * path must fail loudly and immediately rather than as an opaque transport error
66
+ * on the first message.
67
+ * @param path - configured path, possibly home-relative.
68
+ * @returns the parsed account.
69
+ * @throws {Error} when the file is unreadable or unusable.
70
+ */
71
+ export declare function loadServiceAccount(path: string): ServiceAccount;
72
+ /**
73
+ * Sign the JWT-bearer assertion for one service account.
74
+ * @param account - the credentials to sign with.
75
+ * @param nowSeconds - current time from the injected clock.
76
+ * @param scope - OAuth scope the token is requested for.
77
+ * @returns the signed assertion.
78
+ */
79
+ export declare function signedAssertion(account: ServiceAccount, nowSeconds: number, scope: string): string;
80
+ /** Options for {@link ServiceAccountTokens}. */
81
+ interface TokenSourceOptions {
82
+ /** Transport; defaults to the process `fetch`. */
83
+ fetch?: FetchLike;
84
+ /** Clock in milliseconds; defaults to `Date.now`. */
85
+ now?: () => number;
86
+ }
87
+ /**
88
+ * Access tokens for one service account, refreshed on demand.
89
+ *
90
+ * Concurrent callers share one mint in flight, so a first turn with parallel
91
+ * requests does not sign one assertion per request.
92
+ */
93
+ export declare class ServiceAccountTokens {
94
+ #private;
95
+ /**
96
+ * @param account - the parsed service-account credentials.
97
+ * @param options - injectable transport and clock.
98
+ */
99
+ constructor(account: ServiceAccount, options?: TokenSourceOptions);
100
+ /**
101
+ * A currently valid access token, minting one when the cache is cold or stale.
102
+ * @param signal - cancellation for the first caller's mint; waiters attached
103
+ * to an in-flight mint are cancelled only by that mint, which is the
104
+ * accepted ceiling of sharing one token (a per-caller mint would remove it).
105
+ * @returns the bearer token.
106
+ * @throws {VertexAuthError} `AUTH` for a refused credential, `TRANSPORT` when
107
+ * the token endpoint could not be reached or answered unusably.
108
+ */
109
+ get(signal?: AbortSignal): Promise<string>;
110
+ }
111
+ export {};
@@ -0,0 +1,57 @@
1
+ /**
2
+ * Dynamic model discovery for Vertex AI publisher endpoints.
3
+ *
4
+ * Gemini models are fetched from the Vertex Model Garden catalog API
5
+ * (`publishers/google/models`). Anthropic models have no listing endpoint on
6
+ * Vertex, so the adapter serves the catalog from configuration instead.
7
+ *
8
+ * Results are cached for five minutes so the model picker does not
9
+ * make a network call on every open.
10
+ *
11
+ * @module dsh-google-vertex/discovery
12
+ */
13
+ import type { FetchLike } from './auth.ts';
14
+ import type { TokenProvider, VertexModel } from './adapter.ts';
15
+ /** How long a cached model list stays valid, in milliseconds. */
16
+ export declare const DEFAULT_CACHE_TTL_MS: number;
17
+ /**
18
+ * Fetch the Gemini text models from Vertex's `publishers/google/models` list.
19
+ *
20
+ * Follows pagination and keeps `gemini-` ids minus the variants named in
21
+ * {@link NON_TEXT_SEGMENTS}. A built-in id keeps its built-in name; any other
22
+ * is named by its id.
23
+ * @param location - configured Vertex region, or `global`.
24
+ * @param tokens - token source for bearer authentication.
25
+ * @param fetchFn - transport.
26
+ * @param signal - caller cancellation.
27
+ * @returns model ids and display names, in catalog order.
28
+ */
29
+ export declare function fetchGeminiModels(location: string, tokens: TokenProvider, fetchFn: FetchLike, signal?: AbortSignal): Promise<readonly VertexModel[]>;
30
+ /**
31
+ * A cached, TTL-bounded model list that falls back to a static default when
32
+ * the remote fetch fails.
33
+ *
34
+ * Only the Gemini adapter fetches: Vertex has no Anthropic model listing
35
+ * endpoint, so that route serves its configured catalog without one and never
36
+ * calls this.
37
+ */
38
+ export declare class ModelCache {
39
+ #private;
40
+ /**
41
+ * @param fallback - static default returned when the fetch fails or is not
42
+ * attempted.
43
+ */
44
+ constructor(fallback: readonly VertexModel[]);
45
+ /**
46
+ * Return the cached model list, or fetch a fresh one.
47
+ *
48
+ * When a fetch function is provided and the cache is stale, it is called to
49
+ * produce a fresh list. On failure, the fallback is returned. Concurrent
50
+ * callers share one in-flight fetch.
51
+ * @param fetchFn - optional async function that returns a fresh model list.
52
+ * @returns the model list, from cache, fetch, or fallback.
53
+ */
54
+ get(fetchFn?: () => Promise<readonly VertexModel[]>): Promise<readonly VertexModel[]>;
55
+ /** Force the next `get` to re-fetch. */
56
+ invalidate(): void;
57
+ }