@maci0/dsh-google-vertex 0.0.0-stage → 0.12.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +166 -2
- package/cordis.patch.yml +34 -0
- package/icon.svg +6 -0
- package/lib/adapter.js +398 -0
- package/lib/auth.js +246 -0
- package/lib/client.js +206 -0
- package/lib/discovery.js +180 -0
- package/lib/gemini.js +561 -0
- package/lib/gemini_adapter.js +61 -0
- package/lib/host.js +18 -0
- package/lib/index.js +196 -0
- package/lib/types/adapter.d.ts +219 -0
- package/lib/types/auth.d.ts +111 -0
- package/lib/types/discovery.d.ts +57 -0
- package/lib/types/gemini.d.ts +210 -0
- package/lib/types/gemini_adapter.d.ts +54 -0
- package/lib/types/host.d.ts +202 -0
- package/lib/types/index.d.ts +161 -0
- package/lib/types/wire-shared.d.ts +19 -0
- package/lib/types/wire.d.ts +271 -0
- package/lib/wire-shared.js +59 -0
- package/lib/wire.js +722 -0
- package/locale/en.json +6 -0
- package/locale/zh.json +6 -0
- package/package.json +93 -4
package/lib/index.js
ADDED
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* dsh-google-vertex: use Google-hosted models from Vertex AI inside DeepSeek
|
|
3
|
+
* Harness, authenticated with a service-account file.
|
|
4
|
+
*
|
|
5
|
+
* Two capabilities, one configuration row: an `ctx.llm` provider adapter for
|
|
6
|
+
* Vertex's Anthropic publisher endpoint (Claude) and one for its Google
|
|
7
|
+
* publisher endpoint (Gemini). Registering them makes both routes selectable in
|
|
8
|
+
* the Web client's model picker, because `buildModelCatalog()` enumerates
|
|
9
|
+
* `ctx.llm.listProviders()` and asks each adapter for `listModels()` /
|
|
10
|
+
* `resolveModel()`.
|
|
11
|
+
*
|
|
12
|
+
* A third: the `google-vertex` settings namespace. The Gemini adapter discovers
|
|
13
|
+
* its catalog at runtime behind a five-minute cache, and a browser half has no
|
|
14
|
+
* other way to reach this process, so the namespace's one write (the Refresh
|
|
15
|
+
* control on this plugin's row page under Plugins) is what drops that cache
|
|
16
|
+
* on demand.
|
|
17
|
+
*
|
|
18
|
+
* The credential is the service-account JSON itself: a path in configuration,
|
|
19
|
+
* or `GOOGLE_APPLICATION_CREDENTIALS` in the launch environment. Nothing is
|
|
20
|
+
* copied into the harness credential store, and the file is read once at mount
|
|
21
|
+
* so a typo fails loudly instead of on the first message.
|
|
22
|
+
*
|
|
23
|
+
* @module dsh-google-vertex
|
|
24
|
+
*/
|
|
25
|
+
import Schema from '@deepseek-ai/schemastery';
|
|
26
|
+
import { GoogleVertexAnthropicAdapter } from './adapter.js';
|
|
27
|
+
import { loadServiceAccount } from './auth.js';
|
|
28
|
+
import { ServiceAccountTokens } from './auth.js';
|
|
29
|
+
import { fetchGeminiModels } from './discovery.js';
|
|
30
|
+
import { DEFAULT_GEMINI_CONTEXT_WINDOW, DEFAULT_GEMINI_MAX_TOKENS, DEFAULT_GEMINI_MODELS } from './gemini.js';
|
|
31
|
+
import { GoogleVertexGeminiAdapter } from './gemini_adapter.js';
|
|
32
|
+
import { DEFAULT_LOCATION, DEFAULT_STREAM_IDLE_TIMEOUT_MS, MAX_TIMER_DELAY_MS } from './wire.js';
|
|
33
|
+
/** Plugin name as it appears in the loader. */
|
|
34
|
+
export const name = 'google-vertex';
|
|
35
|
+
/** The `ctx.llm` route serving Google-hosted Anthropic (Claude) models. */
|
|
36
|
+
export const PROVIDER = 'google-vertex-anthropic';
|
|
37
|
+
/** The `ctx.llm` route serving Gemini models. */
|
|
38
|
+
export const GEMINI_PROVIDER = 'google-vertex-gemini';
|
|
39
|
+
/**
|
|
40
|
+
* Settings namespace the browser half's card edits: the join key between the
|
|
41
|
+
* two halves. The card registers into `plugins.row.config` under this namespace,
|
|
42
|
+
* and the settings tab pairs the two without knowing what the namespace means.
|
|
43
|
+
*/
|
|
44
|
+
export const GOOGLE_VERTEX_SETTINGS_NAMESPACE = 'google-vertex';
|
|
45
|
+
/** The one service this plugin needs mounted. */
|
|
46
|
+
export const inject = ['llm'];
|
|
47
|
+
/**
|
|
48
|
+
* Claude models Vertex serves, in picker order. Ids are the provider's own
|
|
49
|
+
* aliases, which track the newest dated release of each family and need no
|
|
50
|
+
* edit when Vertex promotes one; the catalog is overridable from configuration
|
|
51
|
+
* for a deployment that pins dated versions or uses a different listing.
|
|
52
|
+
*/
|
|
53
|
+
export const DEFAULT_MODELS = [
|
|
54
|
+
{ id: 'claude-opus-4-6', name: 'Claude Opus 4.6 (Vertex)' },
|
|
55
|
+
{ id: 'claude-sonnet-4-6', name: 'Claude Sonnet 4.6 (Vertex)' },
|
|
56
|
+
{ id: 'claude-opus-4-5', name: 'Claude Opus 4.5 (Vertex)' },
|
|
57
|
+
{ id: 'claude-sonnet-4-5', name: 'Claude Sonnet 4.5 (Vertex)' },
|
|
58
|
+
{ id: 'claude-haiku-4-5', name: 'Claude Haiku 4.5 (Vertex)' },
|
|
59
|
+
];
|
|
60
|
+
/** Context capacity reported for every Claude model; every listed family serves 200k. */
|
|
61
|
+
export const DEFAULT_CONTEXT_WINDOW = 200_000;
|
|
62
|
+
/**
|
|
63
|
+
* Output cap used when a caller omits one. Well under the models' 64k ceiling:
|
|
64
|
+
* an agent turn rarely needs more, and the harness enforces its own budget.
|
|
65
|
+
*/
|
|
66
|
+
export const DEFAULT_MAX_TOKENS = 32_000;
|
|
67
|
+
/**
|
|
68
|
+
* Row schema: defaults live here, so a deployment only states what it changes.
|
|
69
|
+
*
|
|
70
|
+
* `serviceAccountFile`, `project`, `models`, and `geminiModels` carry no
|
|
71
|
+
* `.default()`: the first two fall back to the environment, and an omitted
|
|
72
|
+
* catalog materializes empty, which `resolveConfig` treats exactly like an
|
|
73
|
+
* absent one and replaces with the built-in list.
|
|
74
|
+
*
|
|
75
|
+
* `revalidatedAt` is volatile (the only kind of field the settings document
|
|
76
|
+
* accepts) and carries no default: absence means "never refreshed manually".
|
|
77
|
+
*/
|
|
78
|
+
export const Config = Schema.object({
|
|
79
|
+
serviceAccountFile: Schema.string(),
|
|
80
|
+
project: Schema.string(),
|
|
81
|
+
location: Schema.string().default(DEFAULT_LOCATION),
|
|
82
|
+
models: Schema.array(Schema.string()),
|
|
83
|
+
geminiModels: Schema.array(Schema.string()),
|
|
84
|
+
contextWindow: Schema.number().step(1).min(1).default(DEFAULT_CONTEXT_WINDOW),
|
|
85
|
+
maxTokens: Schema.number().step(1).min(1).default(DEFAULT_MAX_TOKENS),
|
|
86
|
+
streamIdleTimeoutMs: Schema.number().min(Number.MIN_VALUE).max(MAX_TIMER_DELAY_MS).default(DEFAULT_STREAM_IDLE_TIMEOUT_MS),
|
|
87
|
+
revalidatedAt: Schema.string().volatile(),
|
|
88
|
+
});
|
|
89
|
+
/**
|
|
90
|
+
* Validate and normalize one configuration row.
|
|
91
|
+
*
|
|
92
|
+
* Invalid values throw rather than being silently defaulted: a typo'd path or
|
|
93
|
+
* project would otherwise present as an opaque provider error mid-turn.
|
|
94
|
+
* @param config - raw row values, as a profile patch or an unwrapped row.
|
|
95
|
+
* @param env - environment consulted for the credential and region defaults.
|
|
96
|
+
* @returns the resolved adapter configuration.
|
|
97
|
+
*/
|
|
98
|
+
export function resolveConfig(config = {}, env = process.env) {
|
|
99
|
+
// The exported schema is the one source of the numeric defaults and bounds.
|
|
100
|
+
// The credential path, the project, and the region keep their environment
|
|
101
|
+
// fallbacks, so they are read off the raw row instead.
|
|
102
|
+
const filled = Config(config);
|
|
103
|
+
const serviceAccountFile = filled.serviceAccountFile ?? env['GOOGLE_APPLICATION_CREDENTIALS'];
|
|
104
|
+
if (serviceAccountFile === undefined || serviceAccountFile.trim().length === 0) {
|
|
105
|
+
throw new Error('google-vertex: no service account configured: set serviceAccountFile in this plugin\'s row,'
|
|
106
|
+
+ ' or export GOOGLE_APPLICATION_CREDENTIALS to a service-account JSON path');
|
|
107
|
+
}
|
|
108
|
+
const serviceAccount = loadServiceAccount(serviceAccountFile);
|
|
109
|
+
const project = filled.project
|
|
110
|
+
?? env['GOOGLE_CLOUD_PROJECT']
|
|
111
|
+
?? env['GCLOUD_PROJECT']
|
|
112
|
+
?? serviceAccount.project_id;
|
|
113
|
+
if (project === undefined || project.trim().length === 0) {
|
|
114
|
+
throw new Error(`google-vertex: no project configured: set project in this plugin's row, or export`
|
|
115
|
+
+ ` GOOGLE_CLOUD_PROJECT, or use a service-account file that carries "project_id"`);
|
|
116
|
+
}
|
|
117
|
+
const location = config.location ?? env['GOOGLE_CLOUD_LOCATION'] ?? DEFAULT_LOCATION;
|
|
118
|
+
if (location.trim().length === 0 || !/^[a-z0-9-]+$/.test(location)) {
|
|
119
|
+
throw new Error(`google-vertex: location "${location}" is not a valid Vertex region`);
|
|
120
|
+
}
|
|
121
|
+
const ids = filled.models === undefined || filled.models.length === 0
|
|
122
|
+
? DEFAULT_MODELS
|
|
123
|
+
: filled.models.map(id => ({ id, name: id }));
|
|
124
|
+
for (const model of ids) {
|
|
125
|
+
if (model.id.trim().length === 0)
|
|
126
|
+
throw new Error('google-vertex: model ids must be non-empty strings');
|
|
127
|
+
}
|
|
128
|
+
const geminiIds = filled.geminiModels === undefined || filled.geminiModels.length === 0
|
|
129
|
+
? DEFAULT_GEMINI_MODELS.map(model => model.id)
|
|
130
|
+
: filled.geminiModels;
|
|
131
|
+
const geminiModels = geminiIds.map((id) => {
|
|
132
|
+
if (id.trim().length === 0)
|
|
133
|
+
throw new Error('google-vertex: geminiModels ids must be non-empty strings');
|
|
134
|
+
// A configured id keeps the built-in entry's wording when it is one of them;
|
|
135
|
+
// every model serves the same capacities, which live on the wire config.
|
|
136
|
+
return DEFAULT_GEMINI_MODELS.find(model => model.id === id) ?? { id, name: id };
|
|
137
|
+
});
|
|
138
|
+
const streamIdleTimeoutMs = filled.streamIdleTimeoutMs;
|
|
139
|
+
return {
|
|
140
|
+
serviceAccountFile,
|
|
141
|
+
anthropic: {
|
|
142
|
+
serviceAccount,
|
|
143
|
+
project,
|
|
144
|
+
location,
|
|
145
|
+
models: ids,
|
|
146
|
+
contextWindow: filled.contextWindow,
|
|
147
|
+
maxTokens: filled.maxTokens,
|
|
148
|
+
streamIdleTimeoutMs,
|
|
149
|
+
},
|
|
150
|
+
gemini: {
|
|
151
|
+
serviceAccount,
|
|
152
|
+
project,
|
|
153
|
+
location,
|
|
154
|
+
models: geminiModels,
|
|
155
|
+
maxTokens: DEFAULT_GEMINI_MAX_TOKENS,
|
|
156
|
+
streamIdleTimeoutMs,
|
|
157
|
+
},
|
|
158
|
+
};
|
|
159
|
+
}
|
|
160
|
+
/**
|
|
161
|
+
* Mount both adapters.
|
|
162
|
+
* @param ctx - host context; `ctx.llm` must be mounted (`inject` guarantees it).
|
|
163
|
+
* @param config - this plugin's row, as the schema emits it.
|
|
164
|
+
*/
|
|
165
|
+
export function apply(ctx, config) {
|
|
166
|
+
// The volatile stamp stays a live reference: the schema refuses one, and the
|
|
167
|
+
// host never reads the value, so the plain row is what gets resolved.
|
|
168
|
+
const { revalidatedAt: _revalidation, ...row } = config;
|
|
169
|
+
const resolved = resolveConfig(row);
|
|
170
|
+
const { anthropic, gemini } = resolved;
|
|
171
|
+
// Only the Gemini route discovers: Vertex has no Anthropic listing endpoint,
|
|
172
|
+
// so the Claude catalog is the configured (or built-in) list itself. A
|
|
173
|
+
// configured `geminiModels` is likewise served as written, never replaced by
|
|
174
|
+
// the discovered catalog.
|
|
175
|
+
const fetchFn = (input, init) => globalThis.fetch(input, init);
|
|
176
|
+
const tokenSource = new ServiceAccountTokens(anthropic.serviceAccount, { fetch: fetchFn });
|
|
177
|
+
const pinnedGemini = row.geminiModels !== undefined && row.geminiModels.length > 0;
|
|
178
|
+
const discoverGemini = pinnedGemini
|
|
179
|
+
? undefined
|
|
180
|
+
: () => fetchGeminiModels(gemini.location, tokenSource, fetchFn, AbortSignal.timeout(Math.ceil(gemini.streamIdleTimeoutMs)));
|
|
181
|
+
const anthropicAdapter = new GoogleVertexAnthropicAdapter(anthropic);
|
|
182
|
+
const geminiAdapter = new GoogleVertexGeminiAdapter(gemini, discoverGemini === undefined ? {} : { discover: discoverGemini });
|
|
183
|
+
ctx.llm.registerAdapter([PROVIDER], anthropicAdapter);
|
|
184
|
+
ctx.llm.registerAdapter([GEMINI_PROVIDER], geminiAdapter);
|
|
185
|
+
// A profile edit of `revalidatedAt` drops both cached catalogs.
|
|
186
|
+
if (typeof ctx.on === 'function') {
|
|
187
|
+
ctx.on('loader/volatile-update', () => {
|
|
188
|
+
anthropicAdapter.invalidateModels();
|
|
189
|
+
geminiAdapter.invalidateModels();
|
|
190
|
+
});
|
|
191
|
+
}
|
|
192
|
+
ctx.logger.info(`google-vertex: providers "${PROVIDER}" and "${GEMINI_PROVIDER}" registered for project ${anthropic.project}`
|
|
193
|
+
+ ` (location ${anthropic.location}, credentials ${resolved.serviceAccountFile})`
|
|
194
|
+
+ `. Claude: ${anthropic.models.map(model => model.id).join(', ')}`
|
|
195
|
+
+ `. Gemini: ${gemini.models.map(model => model.id).join(', ')}${pinnedGemini ? '' : ' (+ live discovery)'}`);
|
|
196
|
+
}
|
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The machinery both Vertex publisher adapters share, plus the Claude route's
|
|
3
|
+
* own adapter: one metadata face the harness calls on every dispatch, and one
|
|
4
|
+
* streaming pipeline carrying the per-read idle watchdog, the token mint, the
|
|
5
|
+
* fetch, the refusal classification, the SSE loop, and the single-terminal-chunk
|
|
6
|
+
* guarantee.
|
|
7
|
+
*
|
|
8
|
+
* Two things separate the Claude route from the pi-ai-backed `google-vertex`
|
|
9
|
+
* route the harness already ships. Its credential is a service-account file,
|
|
10
|
+
* turned into a bearer token per request by {@link ServiceAccountTokens} rather
|
|
11
|
+
* than typed in as an API key. And its models are Claude, which Vertex serves
|
|
12
|
+
* through `publishers/anthropic`, a path and body the Gemini protocol cannot
|
|
13
|
+
* express.
|
|
14
|
+
*
|
|
15
|
+
* Both routes are text-only: `inputModalities: ['text']` makes `LlmRuntime`
|
|
16
|
+
* project images and files to placeholder text before dispatch, which is honest
|
|
17
|
+
* about what these adapters send.
|
|
18
|
+
*
|
|
19
|
+
* The two routes differ in exactly three things: how a request is addressed
|
|
20
|
+
* and built, what the provider's payloads mean, and what a body that ends
|
|
21
|
+
* without a terminal event should be called. Each is an injected callback;
|
|
22
|
+
* everything else lives here once.
|
|
23
|
+
*
|
|
24
|
+
* @module dsh-google-vertex/adapter
|
|
25
|
+
*/
|
|
26
|
+
import type { FetchLike, ServiceAccount } from './auth.ts';
|
|
27
|
+
import type { GenerateOptions, LlmAdapterLike, LlmModelInfo, LlmProviderInfo, LlmResolvedModelInfo, PreparedAdapterCall, StreamChunk } from './host.ts';
|
|
28
|
+
/** One advertised model. */
|
|
29
|
+
export interface VertexModel {
|
|
30
|
+
readonly id: string;
|
|
31
|
+
readonly name: string;
|
|
32
|
+
}
|
|
33
|
+
/**
|
|
34
|
+
* The token source the adapter asks for a bearer token before each request.
|
|
35
|
+
* Structural so a test can answer with a fixed token; {@link ServiceAccountTokens}
|
|
36
|
+
* is the production implementation.
|
|
37
|
+
*/
|
|
38
|
+
export interface TokenProvider {
|
|
39
|
+
get(signal?: AbortSignal): Promise<string>;
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* Configuration every Vertex adapter row carries. The Claude route adds only
|
|
43
|
+
* its context window; the model capacities of a Gemini route live on the
|
|
44
|
+
* catalog entries, which is why `models` is typed loosely here.
|
|
45
|
+
*/
|
|
46
|
+
interface VertexAdapterConfig {
|
|
47
|
+
readonly serviceAccount: ServiceAccount;
|
|
48
|
+
readonly project: string;
|
|
49
|
+
readonly location: string;
|
|
50
|
+
readonly models: readonly VertexModel[];
|
|
51
|
+
/** Output cap applied when a caller omits `maxTokens`. */
|
|
52
|
+
readonly maxTokens: number;
|
|
53
|
+
/** Bound on the interval between two stream reads, in milliseconds. */
|
|
54
|
+
readonly streamIdleTimeoutMs: number;
|
|
55
|
+
}
|
|
56
|
+
/** Resolved adapter configuration for the Claude route. */
|
|
57
|
+
export interface VertexAnthropicConfig extends VertexAdapterConfig {
|
|
58
|
+
/** Maximum combined request and response context reported for every model. */
|
|
59
|
+
readonly contextWindow: number;
|
|
60
|
+
}
|
|
61
|
+
/**
|
|
62
|
+
* What separates one publisher route from the other, as one parameter set: the
|
|
63
|
+
* display name, the description wording, and the capacities `resolveModel`
|
|
64
|
+
* reports for a model id.
|
|
65
|
+
*/
|
|
66
|
+
interface AdapterMetadata<C extends VertexAdapterConfig> {
|
|
67
|
+
/** Display name the model picker shows for this route. */
|
|
68
|
+
readonly providerName: string;
|
|
69
|
+
/** Context window and output cap reported for every model on this route. */
|
|
70
|
+
readonly capacity: {
|
|
71
|
+
contextWindow: number;
|
|
72
|
+
defaultMaxTokens: number;
|
|
73
|
+
};
|
|
74
|
+
/** Model description reported for this route, project and region included. */
|
|
75
|
+
readonly describe: (config: C) => string;
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* One provider translator, as the shared pump reads it.
|
|
79
|
+
*
|
|
80
|
+
* `handle` is the only method that can emit a terminal chunk: an Anthropic
|
|
81
|
+
* `message_stop` and a Gemini in-band error both do. `terminal` then reports
|
|
82
|
+
* that it, or the watchdog, already ended the turn, so the pump never adds a
|
|
83
|
+
* second finish. A route whose terminal event is not the only way a body ends
|
|
84
|
+
* (Gemini streams whole chunks and simply stops) narrows the truncation
|
|
85
|
+
* question with `sawFinish`: did the provider itself report that the turn was
|
|
86
|
+
* complete? A stream that ends without one is truncated. When absent, the pump
|
|
87
|
+
* falls back to `terminal`.
|
|
88
|
+
*/
|
|
89
|
+
interface StreamTranslatorLike {
|
|
90
|
+
handle(event: Record<string, unknown>): Iterable<StreamChunk>;
|
|
91
|
+
readonly terminal: boolean;
|
|
92
|
+
readonly sawFinish?: boolean;
|
|
93
|
+
/** Terminal chunks a provider that closes its body with data needs. */
|
|
94
|
+
finish?(): Iterable<StreamChunk>;
|
|
95
|
+
}
|
|
96
|
+
/** Everything one streaming call varies between the two publisher routes. */
|
|
97
|
+
interface StreamPumpOptions {
|
|
98
|
+
/** The request URL for the chosen model. */
|
|
99
|
+
readonly endpoint: (model: string, config: VertexAdapterConfig) => string;
|
|
100
|
+
/** The request body for the chosen model. */
|
|
101
|
+
readonly body: (options: GenerateOptions, config: VertexAdapterConfig) => unknown;
|
|
102
|
+
/** Failure named when the body ended without the provider's finish. */
|
|
103
|
+
readonly truncatedMessage: (model: string) => string;
|
|
104
|
+
}
|
|
105
|
+
/**
|
|
106
|
+
* One streaming completion, shared by both publisher routes.
|
|
107
|
+
*
|
|
108
|
+
* The bearer token is minted before the request, so a credentials failure is
|
|
109
|
+
* reported as a terminal finish rather than as a thrown error escaping the
|
|
110
|
+
* generator; `LlmRuntime` would turn a throw into a finish too, but would code
|
|
111
|
+
* it `UNKNOWN` and lose the `AUTH` versus `TRANSPORT` classification.
|
|
112
|
+
*
|
|
113
|
+
* Every read is bounded by `streamIdleTimeoutMs`: a provider that stops sending
|
|
114
|
+
* is a terminal `TIMEOUT` rather than a turn that never ends. The watchdog owns
|
|
115
|
+
* its own controller so the stalled read can be torn down; the caller's signal
|
|
116
|
+
* is combined with it when present. The token mint is armed and cleared with
|
|
117
|
+
* the same bound, because it is a network call too.
|
|
118
|
+
*
|
|
119
|
+
* Exactly one terminal chunk leaves this stream. A body that ended without the
|
|
120
|
+
* provider's terminal event is a truncated response, unless the watchdog or the
|
|
121
|
+
* caller ended the turn, which is the more specific account of the same missing
|
|
122
|
+
* event.
|
|
123
|
+
* @param config - the resolved configuration this adapter serves.
|
|
124
|
+
* @param options - the harness request.
|
|
125
|
+
* @param fetch - transport, already defaulted by the adapter.
|
|
126
|
+
* @param tokens - token source, already defaulted by the adapter.
|
|
127
|
+
* @param makeTranslator - builds this route's payload translator for one model.
|
|
128
|
+
* @param pump - this route's endpoint, body, and terminal wording.
|
|
129
|
+
* @yields every chunk the provider's stream completes.
|
|
130
|
+
*/
|
|
131
|
+
export declare function streamVertex(config: VertexAdapterConfig, options: GenerateOptions, fetch: FetchLike, tokens: TokenProvider, makeTranslator: (model: string) => StreamTranslatorLike, pump: StreamPumpOptions): AsyncGenerator<StreamChunk>;
|
|
132
|
+
/**
|
|
133
|
+
* Duck-typed base for both Vertex publisher adapters.
|
|
134
|
+
*
|
|
135
|
+
* `LlmRuntime` reaches adapters through plain method calls, so these objects
|
|
136
|
+
* need no harness base class; the plugin's only runtime `@deepseek-ai/*`
|
|
137
|
+
* dependency is `@deepseek-ai/dsh-llm`'s pure `attributionHeaders()` helper.
|
|
138
|
+
* The metadata face below is identical for both routes, so it is written once
|
|
139
|
+
* and parameterized by {@link AdapterMetadata}.
|
|
140
|
+
*/
|
|
141
|
+
export declare abstract class VertexPublisherAdapter<C extends VertexAdapterConfig> implements LlmAdapterLike {
|
|
142
|
+
#private;
|
|
143
|
+
/** The config this adapter serves, for the stream pipeline below. */
|
|
144
|
+
protected readonly config: C;
|
|
145
|
+
/** The transport this adapter was built with. */
|
|
146
|
+
protected readonly fetch: FetchLike;
|
|
147
|
+
/** The token source this adapter asks before each request. */
|
|
148
|
+
protected readonly tokens: TokenProvider;
|
|
149
|
+
/**
|
|
150
|
+
* @param metadata - display name, capacities, and description text.
|
|
151
|
+
* @param config - the resolved configuration this adapter serves.
|
|
152
|
+
* @param options - transport, token-source, and discovery overrides.
|
|
153
|
+
*/
|
|
154
|
+
constructor(metadata: AdapterMetadata<C>, config: C, options?: {
|
|
155
|
+
fetch?: FetchLike;
|
|
156
|
+
tokens?: TokenProvider;
|
|
157
|
+
discover?: () => Promise<readonly VertexModel[]>;
|
|
158
|
+
});
|
|
159
|
+
/** {@inheritDoc LlmAdapterLike.providerInfo} */
|
|
160
|
+
providerInfo(provider: string): LlmProviderInfo;
|
|
161
|
+
/**
|
|
162
|
+
* No provider-owned retry policy: Vertex's own 429s carry a quota code the
|
|
163
|
+
* harness already classifies, and a per-request backoff belongs to the
|
|
164
|
+
* harness defaults.
|
|
165
|
+
*
|
|
166
|
+
* This and {@link imageRequestPricing} exist because `LlmRuntime` calls them
|
|
167
|
+
* on every dispatch; a duck-typed adapter must supply them or the first
|
|
168
|
+
* registration throws.
|
|
169
|
+
*/
|
|
170
|
+
providerRetryPolicy(_provider: string): undefined;
|
|
171
|
+
/** No route charges visual tokens: these adapters are text-only. */
|
|
172
|
+
imageRequestPricing(_provider: string, _model: string): undefined;
|
|
173
|
+
/**
|
|
174
|
+
* The current model catalog, fetched from the provider when discovery is
|
|
175
|
+
* configured, falling back to the static catalog on failure.
|
|
176
|
+
*
|
|
177
|
+
* The result is cached with a five-minute TTL so the model picker does
|
|
178
|
+
* not make a network call on every open.
|
|
179
|
+
*/
|
|
180
|
+
listModels(provider: string): Promise<readonly LlmModelInfo[]>;
|
|
181
|
+
/**
|
|
182
|
+
* Drop the cached catalog, so the next `listModels` re-discovers from the
|
|
183
|
+
* provider. Both adapters answer the plugin's manual refresh with this.
|
|
184
|
+
*/
|
|
185
|
+
invalidateModels(): void;
|
|
186
|
+
/** {@inheritDoc LlmAdapterLike.resolveModel} */
|
|
187
|
+
resolveModel(provider: string, model: string, _signal?: AbortSignal): Promise<LlmResolvedModelInfo>;
|
|
188
|
+
/** {@inheritDoc LlmAdapterLike.prepareCall} */
|
|
189
|
+
prepareCall(provider: string, model: string, signal?: AbortSignal): Promise<PreparedAdapterCall>;
|
|
190
|
+
/** Stream one completion through this route's publisher endpoint. */
|
|
191
|
+
abstract stream(options: GenerateOptions): AsyncIterable<StreamChunk>;
|
|
192
|
+
}
|
|
193
|
+
/**
|
|
194
|
+
* Duck-typed adapter over Vertex's Anthropic publisher endpoint.
|
|
195
|
+
*
|
|
196
|
+
* The route is text-only and declares no server-executed tools, and every
|
|
197
|
+
* Claude family in the default catalog serves the same 200k context, so a
|
|
198
|
+
* model's capacities are the configured pair rather than a per-model entry.
|
|
199
|
+
*/
|
|
200
|
+
export declare class GoogleVertexAnthropicAdapter extends VertexPublisherAdapter<VertexAnthropicConfig> {
|
|
201
|
+
/**
|
|
202
|
+
* @param config - the resolved configuration this adapter serves.
|
|
203
|
+
* @param options - transport, token-source, and discovery overrides for tests.
|
|
204
|
+
*/
|
|
205
|
+
constructor(config: VertexAnthropicConfig, options?: {
|
|
206
|
+
fetch?: FetchLike;
|
|
207
|
+
tokens?: TokenProvider;
|
|
208
|
+
discover?: () => Promise<readonly VertexModel[]>;
|
|
209
|
+
});
|
|
210
|
+
/**
|
|
211
|
+
* Stream one completion through `:streamRawPredict`.
|
|
212
|
+
*
|
|
213
|
+
* The shared pump owns the watchdog, the token mint, the SSE loop, and the
|
|
214
|
+
* single terminal chunk; `message_stop` closes the turn mid-body, so this
|
|
215
|
+
* route's finish rides that event.
|
|
216
|
+
*/
|
|
217
|
+
stream(options: GenerateOptions): AsyncIterable<StreamChunk>;
|
|
218
|
+
}
|
|
219
|
+
export {};
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Service-account authentication for Google APIs: read one credentials file,
|
|
3
|
+
* sign a JWT with its private key, trade it for an OAuth access token, and
|
|
4
|
+
* cache that token until shortly before it expires.
|
|
5
|
+
*
|
|
6
|
+
* The harness credential plane stores API keys, while a Vertex deployment
|
|
7
|
+
* typically holds a service-account JSON instead: the file is the credential.
|
|
8
|
+
* That is why this module exists rather than a credential reference: nothing in
|
|
9
|
+
* the harness can turn an RSA key into a bearer token on the request path.
|
|
10
|
+
*
|
|
11
|
+
* Everything here is pure enough to test without a network: the token endpoint
|
|
12
|
+
* transport and the clock are injectable. The scope every request needs and the
|
|
13
|
+
* early-refresh margin are fixed, because nothing has ever needed to vary them:
|
|
14
|
+
* one Vertex route and one margin are what this plugin talks to.
|
|
15
|
+
*
|
|
16
|
+
* @module dsh-google-vertex/auth
|
|
17
|
+
*/
|
|
18
|
+
/** Scope every Vertex AI request needs; the token carries nothing narrower. */
|
|
19
|
+
export declare const CLOUD_PLATFORM_SCOPE = "https://www.googleapis.com/auth/cloud-platform";
|
|
20
|
+
/** Injectable transport, so tests never touch the network. */
|
|
21
|
+
export type FetchLike = (input: string, init: RequestInit) => Promise<Response>;
|
|
22
|
+
/** The fields of a service-account JSON this module reads. */
|
|
23
|
+
export interface ServiceAccount {
|
|
24
|
+
readonly client_email: string;
|
|
25
|
+
readonly private_key: string;
|
|
26
|
+
readonly private_key_id?: string;
|
|
27
|
+
readonly project_id?: string;
|
|
28
|
+
readonly token_uri?: string;
|
|
29
|
+
}
|
|
30
|
+
/** An auth failure the adapter reports as `AUTH` or `TRANSPORT`. */
|
|
31
|
+
export declare class VertexAuthError extends Error {
|
|
32
|
+
/** Provider-neutral failure code the adapter forwards. */
|
|
33
|
+
readonly code: 'AUTH' | 'TRANSPORT';
|
|
34
|
+
/**
|
|
35
|
+
* @param code - whether the failure is a credential problem or a transport one.
|
|
36
|
+
* @param message - operator-facing detail.
|
|
37
|
+
*/
|
|
38
|
+
constructor(code: 'AUTH' | 'TRANSPORT', message: string);
|
|
39
|
+
}
|
|
40
|
+
/**
|
|
41
|
+
* Expand a leading `~` to the process home directory.
|
|
42
|
+
*
|
|
43
|
+
* The path arrives from configuration, which is written by a human the same way
|
|
44
|
+
* a shell is: `~/.secrets/vertex.json` means the home directory, not a literal
|
|
45
|
+
* directory named `~`.
|
|
46
|
+
* @param path - configured path, possibly home-relative.
|
|
47
|
+
* @returns an absolute path.
|
|
48
|
+
*/
|
|
49
|
+
export declare function expandHome(path: string): string;
|
|
50
|
+
/**
|
|
51
|
+
* Parse a service-account document, refusing one this module cannot sign with.
|
|
52
|
+
*
|
|
53
|
+
* `name` is the path the text came from, so the failure tells an operator which
|
|
54
|
+
* file to fix rather than only which field is wrong.
|
|
55
|
+
* @param raw - the file's text.
|
|
56
|
+
* @param name - source path used in failure messages.
|
|
57
|
+
* @returns the parsed account.
|
|
58
|
+
* @throws {Error} when the document is not JSON or lacks a signing key.
|
|
59
|
+
*/
|
|
60
|
+
export declare function parseServiceAccount(raw: string, name: string): ServiceAccount;
|
|
61
|
+
/**
|
|
62
|
+
* Read and parse one service-account file.
|
|
63
|
+
*
|
|
64
|
+
* Read synchronously on purpose: this runs once at plugin mount, where a typo'd
|
|
65
|
+
* path must fail loudly and immediately rather than as an opaque transport error
|
|
66
|
+
* on the first message.
|
|
67
|
+
* @param path - configured path, possibly home-relative.
|
|
68
|
+
* @returns the parsed account.
|
|
69
|
+
* @throws {Error} when the file is unreadable or unusable.
|
|
70
|
+
*/
|
|
71
|
+
export declare function loadServiceAccount(path: string): ServiceAccount;
|
|
72
|
+
/**
|
|
73
|
+
* Sign the JWT-bearer assertion for one service account.
|
|
74
|
+
* @param account - the credentials to sign with.
|
|
75
|
+
* @param nowSeconds - current time from the injected clock.
|
|
76
|
+
* @param scope - OAuth scope the token is requested for.
|
|
77
|
+
* @returns the signed assertion.
|
|
78
|
+
*/
|
|
79
|
+
export declare function signedAssertion(account: ServiceAccount, nowSeconds: number, scope: string): string;
|
|
80
|
+
/** Options for {@link ServiceAccountTokens}. */
|
|
81
|
+
interface TokenSourceOptions {
|
|
82
|
+
/** Transport; defaults to the process `fetch`. */
|
|
83
|
+
fetch?: FetchLike;
|
|
84
|
+
/** Clock in milliseconds; defaults to `Date.now`. */
|
|
85
|
+
now?: () => number;
|
|
86
|
+
}
|
|
87
|
+
/**
|
|
88
|
+
* Access tokens for one service account, refreshed on demand.
|
|
89
|
+
*
|
|
90
|
+
* Concurrent callers share one mint in flight, so a first turn with parallel
|
|
91
|
+
* requests does not sign one assertion per request.
|
|
92
|
+
*/
|
|
93
|
+
export declare class ServiceAccountTokens {
|
|
94
|
+
#private;
|
|
95
|
+
/**
|
|
96
|
+
* @param account - the parsed service-account credentials.
|
|
97
|
+
* @param options - injectable transport and clock.
|
|
98
|
+
*/
|
|
99
|
+
constructor(account: ServiceAccount, options?: TokenSourceOptions);
|
|
100
|
+
/**
|
|
101
|
+
* A currently valid access token, minting one when the cache is cold or stale.
|
|
102
|
+
* @param signal - cancellation for the first caller's mint; waiters attached
|
|
103
|
+
* to an in-flight mint are cancelled only by that mint, which is the
|
|
104
|
+
* accepted ceiling of sharing one token (a per-caller mint would remove it).
|
|
105
|
+
* @returns the bearer token.
|
|
106
|
+
* @throws {VertexAuthError} `AUTH` for a refused credential, `TRANSPORT` when
|
|
107
|
+
* the token endpoint could not be reached or answered unusably.
|
|
108
|
+
*/
|
|
109
|
+
get(signal?: AbortSignal): Promise<string>;
|
|
110
|
+
}
|
|
111
|
+
export {};
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Dynamic model discovery for Vertex AI publisher endpoints.
|
|
3
|
+
*
|
|
4
|
+
* Gemini models are fetched from the Vertex Model Garden catalog API
|
|
5
|
+
* (`publishers/google/models`). Anthropic models have no listing endpoint on
|
|
6
|
+
* Vertex, so the adapter serves the catalog from configuration instead.
|
|
7
|
+
*
|
|
8
|
+
* Results are cached for five minutes so the model picker does not
|
|
9
|
+
* make a network call on every open.
|
|
10
|
+
*
|
|
11
|
+
* @module dsh-google-vertex/discovery
|
|
12
|
+
*/
|
|
13
|
+
import type { FetchLike } from './auth.ts';
|
|
14
|
+
import type { TokenProvider, VertexModel } from './adapter.ts';
|
|
15
|
+
/** How long a cached model list stays valid, in milliseconds. */
|
|
16
|
+
export declare const DEFAULT_CACHE_TTL_MS: number;
|
|
17
|
+
/**
|
|
18
|
+
* Fetch the Gemini text models from Vertex's `publishers/google/models` list.
|
|
19
|
+
*
|
|
20
|
+
* Follows pagination and keeps `gemini-` ids minus the variants named in
|
|
21
|
+
* {@link NON_TEXT_SEGMENTS}. A built-in id keeps its built-in name; any other
|
|
22
|
+
* is named by its id.
|
|
23
|
+
* @param location - configured Vertex region, or `global`.
|
|
24
|
+
* @param tokens - token source for bearer authentication.
|
|
25
|
+
* @param fetchFn - transport.
|
|
26
|
+
* @param signal - caller cancellation.
|
|
27
|
+
* @returns model ids and display names, in catalog order.
|
|
28
|
+
*/
|
|
29
|
+
export declare function fetchGeminiModels(location: string, tokens: TokenProvider, fetchFn: FetchLike, signal?: AbortSignal): Promise<readonly VertexModel[]>;
|
|
30
|
+
/**
|
|
31
|
+
* A cached, TTL-bounded model list that falls back to a static default when
|
|
32
|
+
* the remote fetch fails.
|
|
33
|
+
*
|
|
34
|
+
* Only the Gemini adapter fetches: Vertex has no Anthropic model listing
|
|
35
|
+
* endpoint, so that route serves its configured catalog without one and never
|
|
36
|
+
* calls this.
|
|
37
|
+
*/
|
|
38
|
+
export declare class ModelCache {
|
|
39
|
+
#private;
|
|
40
|
+
/**
|
|
41
|
+
* @param fallback - static default returned when the fetch fails or is not
|
|
42
|
+
* attempted.
|
|
43
|
+
*/
|
|
44
|
+
constructor(fallback: readonly VertexModel[]);
|
|
45
|
+
/**
|
|
46
|
+
* Return the cached model list, or fetch a fresh one.
|
|
47
|
+
*
|
|
48
|
+
* When a fetch function is provided and the cache is stale, it is called to
|
|
49
|
+
* produce a fresh list. On failure, the fallback is returned. Concurrent
|
|
50
|
+
* callers share one in-flight fetch.
|
|
51
|
+
* @param fetchFn - optional async function that returns a fresh model list.
|
|
52
|
+
* @returns the model list, from cache, fetch, or fallback.
|
|
53
|
+
*/
|
|
54
|
+
get(fetchFn?: () => Promise<readonly VertexModel[]>): Promise<readonly VertexModel[]>;
|
|
55
|
+
/** Force the next `get` to re-fetch. */
|
|
56
|
+
invalidate(): void;
|
|
57
|
+
}
|