@maci0/dsh-chatjimmy 0.11.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/API.md +145 -0
- package/LICENSE +21 -0
- package/README.md +127 -0
- package/cordis.patch.yml +13 -0
- package/icon.svg +6 -0
- package/lib/adapter.js +421 -0
- package/lib/client.js +398 -0
- package/lib/host.js +21 -0
- package/lib/index.js +172 -0
- package/lib/protocol.js +172 -0
- package/lib/types/adapter.d.ts +78 -0
- package/lib/types/host.d.ts +177 -0
- package/lib/types/index.d.ts +125 -0
- package/lib/types/protocol.d.ts +101 -0
- package/locale/en.json +6 -0
- package/locale/zh.json +6 -0
- package/package.json +95 -0
package/lib/protocol.js
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Wire translation between the harness request vocabulary and the
|
|
3
|
+
* chatjimmy.ai HTTP API reconstructed in `API.md`.
|
|
4
|
+
*
|
|
5
|
+
* Everything here is pure so it can be tested without a harness or a network.
|
|
6
|
+
*
|
|
7
|
+
* @module dsh-chatjimmy/protocol
|
|
8
|
+
*/
|
|
9
|
+
/** Opening marker of the trailing generation-stats block. */
|
|
10
|
+
export const STATS_OPEN = '<|stats|>';
|
|
11
|
+
/** Closing marker of the trailing generation-stats block. */
|
|
12
|
+
export const STATS_CLOSE = '<|/stats|>';
|
|
13
|
+
/** Flatten a block tree to the plain text the wire can carry. */
|
|
14
|
+
function blockText(block) {
|
|
15
|
+
switch (block.type) {
|
|
16
|
+
case 'text':
|
|
17
|
+
case 'reasoning':
|
|
18
|
+
return typeof block.text === 'string' ? block.text : '';
|
|
19
|
+
case 'tool-call':
|
|
20
|
+
// The service has no tool protocol. Render the call as prose so a
|
|
21
|
+
// cross-provider history still reads as a conversation.
|
|
22
|
+
return `[tool call] ${String(block.name)}(${String(block.arguments)})`;
|
|
23
|
+
default:
|
|
24
|
+
// Images and files are already projected to text for a text-only route
|
|
25
|
+
// by LlmRuntime; anything else unknown is dropped rather than guessed.
|
|
26
|
+
return '';
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
/** Flatten a block list to plain text. */
|
|
30
|
+
export function flatten(blocks) {
|
|
31
|
+
return blocks.map(blockText).join('');
|
|
32
|
+
}
|
|
33
|
+
/**
|
|
34
|
+
* Build the exact request body for one model call: history is flattened to
|
|
35
|
+
* text, every system-role message is hoisted into the single `systemPrompt`
|
|
36
|
+
* slot, and the caller's `system` text leads it.
|
|
37
|
+
*/
|
|
38
|
+
export function buildChatRequest(options, config) {
|
|
39
|
+
const systemParts = [];
|
|
40
|
+
const messages = [];
|
|
41
|
+
for (const message of options.messages) {
|
|
42
|
+
const text = flatten(message.content);
|
|
43
|
+
if (message.role === 'system' || message.role === 'developer') {
|
|
44
|
+
if (text.length > 0)
|
|
45
|
+
systemParts.push(text);
|
|
46
|
+
continue;
|
|
47
|
+
}
|
|
48
|
+
if (text.length === 0)
|
|
49
|
+
continue;
|
|
50
|
+
// The wire accepts user/assistant turns only, so a tool result rides the
|
|
51
|
+
// user turn (the harness's own adapters do the same for a provider with no
|
|
52
|
+
// tool role), labelled so it does not read as the user's own words.
|
|
53
|
+
if (message.role === 'tool')
|
|
54
|
+
messages.push({ role: 'user', content: `[tool result] ${text}` });
|
|
55
|
+
else
|
|
56
|
+
messages.push({ role: message.role, content: text });
|
|
57
|
+
}
|
|
58
|
+
if (options.system !== undefined && options.system.length > 0)
|
|
59
|
+
systemParts.unshift(options.system);
|
|
60
|
+
return {
|
|
61
|
+
messages,
|
|
62
|
+
chatOptions: {
|
|
63
|
+
selectedModel: options.model.length > 0 ? options.model : config.model,
|
|
64
|
+
systemPrompt: systemParts.join('\n\n'),
|
|
65
|
+
topK: config.topK,
|
|
66
|
+
},
|
|
67
|
+
attachment: null,
|
|
68
|
+
};
|
|
69
|
+
}
|
|
70
|
+
/** Parse the stats payload, returning undefined for malformed JSON. */
|
|
71
|
+
export function parseStats(raw) {
|
|
72
|
+
try {
|
|
73
|
+
const value = JSON.parse(raw);
|
|
74
|
+
return typeof value === 'object' && value !== null ? value : undefined;
|
|
75
|
+
}
|
|
76
|
+
catch {
|
|
77
|
+
return undefined;
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* True when a stats `reason` reports the backend's own context-limit refusal.
|
|
82
|
+
* @param reason - the stats `reason` field, when present.
|
|
83
|
+
*/
|
|
84
|
+
export function isContextLimitReason(reason) {
|
|
85
|
+
return typeof reason === 'string' && /max\s+context\s+limit\s+\d+\s+reached/i.test(reason);
|
|
86
|
+
}
|
|
87
|
+
/** A token counter, or undefined when the stats did not carry a usable one. */
|
|
88
|
+
function counter(value) {
|
|
89
|
+
return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined;
|
|
90
|
+
}
|
|
91
|
+
/**
|
|
92
|
+
* Map backend token counters onto harness usage.
|
|
93
|
+
* @param stats - parsed stats, when the stream carried them.
|
|
94
|
+
* @returns disjoint harness counts (the provider reports no cache split), or
|
|
95
|
+
* undefined when the stats carry neither prompt nor output counter. The total
|
|
96
|
+
* is the provider's own, else the sum when both parts are known, else omitted.
|
|
97
|
+
*/
|
|
98
|
+
export function mapUsage(stats) {
|
|
99
|
+
const input = counter(stats?.prefill_tokens);
|
|
100
|
+
const output = counter(stats?.decode_tokens);
|
|
101
|
+
if (input === undefined && output === undefined)
|
|
102
|
+
return undefined;
|
|
103
|
+
const total = counter(stats?.total_tokens)
|
|
104
|
+
?? (input === undefined || output === undefined ? undefined : input + output);
|
|
105
|
+
return { inputTokens: input ?? 0, outputTokens: output ?? 0, ...total === undefined ? {} : { totalTokens: total } };
|
|
106
|
+
}
|
|
107
|
+
/**
|
|
108
|
+
* Splits the generated text from the trailing `<|stats|>…<|/stats|>` block
|
|
109
|
+
* without ever emitting a partial marker.
|
|
110
|
+
*
|
|
111
|
+
* The backend appends the block to the same byte stream as the completion, so a
|
|
112
|
+
* reader that forwarded chunks verbatim would leak `{"prefill_tokens":…}` into
|
|
113
|
+
* the model's visible answer. Text is held back only as far as a marker could
|
|
114
|
+
* still be forming, so time-to-first-token is unaffected.
|
|
115
|
+
*/
|
|
116
|
+
export class StatsStreamFilter {
|
|
117
|
+
#buffer = '';
|
|
118
|
+
#stats;
|
|
119
|
+
#closed = false;
|
|
120
|
+
/**
|
|
121
|
+
* Absorb one decoded chunk.
|
|
122
|
+
* @param text - newly decoded text.
|
|
123
|
+
* @returns the prefix that is certainly completion text.
|
|
124
|
+
*/
|
|
125
|
+
push(text) {
|
|
126
|
+
if (this.#closed || text.length === 0)
|
|
127
|
+
return '';
|
|
128
|
+
this.#buffer += text;
|
|
129
|
+
const open = this.#buffer.indexOf(STATS_OPEN);
|
|
130
|
+
if (open < 0) {
|
|
131
|
+
// No marker yet: hold back only what a split marker could consume.
|
|
132
|
+
const keep = STATS_OPEN.length - 1;
|
|
133
|
+
if (this.#buffer.length <= keep)
|
|
134
|
+
return '';
|
|
135
|
+
const emit = this.#buffer.slice(0, this.#buffer.length - keep);
|
|
136
|
+
this.#buffer = this.#buffer.slice(-keep);
|
|
137
|
+
return emit;
|
|
138
|
+
}
|
|
139
|
+
const emit = this.#buffer.slice(0, open);
|
|
140
|
+
const rest = this.#buffer.slice(open);
|
|
141
|
+
const close = rest.indexOf(STATS_CLOSE);
|
|
142
|
+
if (close < 0) {
|
|
143
|
+
this.#buffer = rest;
|
|
144
|
+
return emit;
|
|
145
|
+
}
|
|
146
|
+
this.#stats = parseStats(rest.slice(STATS_OPEN.length, close));
|
|
147
|
+
this.#buffer = '';
|
|
148
|
+
this.#closed = true;
|
|
149
|
+
return emit;
|
|
150
|
+
}
|
|
151
|
+
/**
|
|
152
|
+
* Release whatever the stream ended with.
|
|
153
|
+
* @returns residual completion text; empty when the body ended on the marker.
|
|
154
|
+
*/
|
|
155
|
+
flush() {
|
|
156
|
+
if (this.#closed || this.#buffer.length === 0)
|
|
157
|
+
return '';
|
|
158
|
+
// A full opening marker was seen, so nothing still buffered is completion
|
|
159
|
+
// text: a block the body cut short must not be printed as the answer.
|
|
160
|
+
if (this.#buffer.startsWith(STATS_OPEN)) {
|
|
161
|
+
this.#buffer = '';
|
|
162
|
+
return '';
|
|
163
|
+
}
|
|
164
|
+
const emit = this.#buffer;
|
|
165
|
+
this.#buffer = '';
|
|
166
|
+
return emit;
|
|
167
|
+
}
|
|
168
|
+
/** Stats the stream carried, when the sentinel was complete. */
|
|
169
|
+
get stats() {
|
|
170
|
+
return this.#stats;
|
|
171
|
+
}
|
|
172
|
+
}
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Provider adapter for the chatjimmy.ai chat API (see `API.md` at the
|
|
3
|
+
* repository root for the reconstructed wire contract).
|
|
4
|
+
*
|
|
5
|
+
* The service is text-in/text-out: it accepts no tool schemas, no images, no
|
|
6
|
+
* sampling parameters, and returns one plain-text stream. This adapter is
|
|
7
|
+
* therefore honest about being a text-only route: it advertises
|
|
8
|
+
* `inputModalities: ['text']` so `LlmRuntime` projects files and images to
|
|
9
|
+
* placeholder text before dispatch, and it ignores every tool field rather than
|
|
10
|
+
* pretending the model can call them.
|
|
11
|
+
*
|
|
12
|
+
* @module dsh-chatjimmy/adapter
|
|
13
|
+
*/
|
|
14
|
+
import { CONTEXT_WINDOW_EXCEEDED_CODE } from '@deepseek-ai/dsh-llm';
|
|
15
|
+
import type { ResolvedRetryPolicy } from '@deepseek-ai/dsh-llm';
|
|
16
|
+
import { type ChatJimmyConfig } from './protocol.ts';
|
|
17
|
+
import type { GenerateOptions, LlmAdapterLike, LlmModelInfo, LlmProviderInfo, LlmResolvedModelInfo, PreparedAdapterCall, StreamChunk } from './host.ts';
|
|
18
|
+
/** Stable failure used when the backend answers with a zero-byte stream body. */
|
|
19
|
+
export { CONTEXT_WINDOW_EXCEEDED_CODE };
|
|
20
|
+
/** Injectable fetch, so the adapter is testable without a network. */
|
|
21
|
+
export type FetchLike = (input: string, init: RequestInit) => Promise<Response>;
|
|
22
|
+
/**
|
|
23
|
+
* Duck-typed adapter over `POST /api/chat`.
|
|
24
|
+
*
|
|
25
|
+
* `LlmRuntime` reaches adapters through plain method calls, so this object
|
|
26
|
+
* needs no harness base class. The plugin's only runtime dependency on
|
|
27
|
+
* `@deepseek-ai/*` is `@deepseek-ai/dsh-llm`'s pure helpers
|
|
28
|
+
* (`attributionHeaders()` and `resolveRetryPolicy()`), never its error classes
|
|
29
|
+
* or adapter base class.
|
|
30
|
+
*/
|
|
31
|
+
export declare class ChatJimmyAdapter implements LlmAdapterLike {
|
|
32
|
+
#private;
|
|
33
|
+
/**
|
|
34
|
+
* @param config - the resolved configuration this adapter serves, or a
|
|
35
|
+
* provider that resolves it on every read. The plugin passes the live
|
|
36
|
+
* provider, so a settings edit (the Plugins card, `/alias`-style patch
|
|
37
|
+
* writes) reaches the next request without remounting the adapter; a plain
|
|
38
|
+
* value keeps the snapshot behavior for direct callers and tests.
|
|
39
|
+
* @param fetchImpl - transport override for tests.
|
|
40
|
+
*/
|
|
41
|
+
constructor(config: ChatJimmyConfig | (() => ChatJimmyConfig), fetchImpl?: FetchLike);
|
|
42
|
+
/** {@inheritDoc LlmAdapterLike.providerInfo} */
|
|
43
|
+
providerInfo(provider: string): LlmProviderInfo;
|
|
44
|
+
/**
|
|
45
|
+
* The provider-owned retry policy from configuration, already resolved, or
|
|
46
|
+
* `undefined` to leave the harness's normal defaults in place.
|
|
47
|
+
*
|
|
48
|
+
* This and {@link imageRequestPricing} exist because `LlmRuntime` calls them
|
|
49
|
+
* on every dispatch. A harness `LlmAdapter` subclass inherits them; a
|
|
50
|
+
* duck-typed adapter must supply them or the very first registration throws
|
|
51
|
+
* `adapter.providerRetryPolicy is not a function`.
|
|
52
|
+
*/
|
|
53
|
+
providerRetryPolicy(_provider: string): ResolvedRetryPolicy | undefined;
|
|
54
|
+
/** No route charges visual tokens: this adapter is text-only. */
|
|
55
|
+
imageRequestPricing(_provider: string, _model: string): undefined;
|
|
56
|
+
/**
|
|
57
|
+
* The advertised catalog. `/api/models` serves exactly one entry, so it is
|
|
58
|
+
* mirrored from configuration instead of costing a round trip on every model
|
|
59
|
+
* picker open. The id stays advisory: any id is accepted on the wire.
|
|
60
|
+
*/
|
|
61
|
+
listModels(provider: string): Promise<readonly LlmModelInfo[]>;
|
|
62
|
+
/** {@inheritDoc LlmAdapterLike.resolveModel} */
|
|
63
|
+
resolveModel(provider: string, model: string, _signal?: AbortSignal): Promise<LlmResolvedModelInfo>;
|
|
64
|
+
/** {@inheritDoc LlmAdapterLike.prepareCall} */
|
|
65
|
+
prepareCall(provider: string, model: string, signal?: AbortSignal): Promise<PreparedAdapterCall>;
|
|
66
|
+
/**
|
|
67
|
+
* Stream one completion. Everything the service sends is completion text up
|
|
68
|
+
* to the trailing stats block, which {@link StatsStreamFilter} removes; the
|
|
69
|
+
* block is then reported as harness usage.
|
|
70
|
+
*
|
|
71
|
+
* A zero-byte body with HTTP 200 is the service's signature for a request
|
|
72
|
+
* that overflowed the context window: the response headers are already
|
|
73
|
+
* committed as `text/event-stream`, so the backend's refusal cannot be
|
|
74
|
+
* delivered as an error status. That case is reported as
|
|
75
|
+
* `CONTEXT_WINDOW_EXCEEDED` rather than as an empty completion.
|
|
76
|
+
*/
|
|
77
|
+
stream(options: GenerateOptions): AsyncIterable<StreamChunk>;
|
|
78
|
+
}
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The slice of the DeepSeek Harness host surface this plugin uses, declared
|
|
3
|
+
* structurally.
|
|
4
|
+
*
|
|
5
|
+
* Like the other plugins under `~/dsh-plugins`, this package declares the host
|
|
6
|
+
* surface it consumes rather than depending on the harness's own classes: its
|
|
7
|
+
* one runtime `@deepseek-ai/*` dependency is `@deepseek-ai/dsh-llm`'s pure
|
|
8
|
+
* `attributionHeaders()` / `resolveRetryPolicy()` helpers. The harness reaches
|
|
9
|
+
* its adapters through plain method calls on the registered object (there is
|
|
10
|
+
* no `instanceof LlmAdapter` check anywhere in `LlmRuntime`), so a duck-typed
|
|
11
|
+
* adapter is a supported shape, not a workaround.
|
|
12
|
+
*
|
|
13
|
+
* Each declaration here is deliberately narrowed to what this adapter reads or
|
|
14
|
+
* emits. Mirroring a foreign API in full is not documentation: a field we never
|
|
15
|
+
* touch is a field whose absence goes unnoticed, which is how the missing
|
|
16
|
+
* `providerRetryPolicy` reached an integration test instead of a compiler.
|
|
17
|
+
* Widen a declaration when the adapter starts using it, not before.
|
|
18
|
+
*
|
|
19
|
+
* @module dsh-chatjimmy/host
|
|
20
|
+
*/
|
|
21
|
+
/** Disposer returned by every host registration. */
|
|
22
|
+
type Disposable = () => void;
|
|
23
|
+
/** Content block the harness may hand us in request history. */
|
|
24
|
+
export type ContentBlock = {
|
|
25
|
+
readonly type: 'text';
|
|
26
|
+
readonly text: string;
|
|
27
|
+
} | {
|
|
28
|
+
readonly type: 'reasoning';
|
|
29
|
+
readonly text: string;
|
|
30
|
+
} | {
|
|
31
|
+
readonly type: 'tool-call';
|
|
32
|
+
readonly id: string;
|
|
33
|
+
readonly name: string;
|
|
34
|
+
readonly arguments: string;
|
|
35
|
+
} | {
|
|
36
|
+
readonly type: string;
|
|
37
|
+
readonly [key: string]: unknown;
|
|
38
|
+
};
|
|
39
|
+
/** One message in a fully-assembled request. */
|
|
40
|
+
interface Message {
|
|
41
|
+
readonly id: string;
|
|
42
|
+
/**
|
|
43
|
+
* Every role the harness really sends. Narrowing this to the three the wire
|
|
44
|
+
* accepts is what let `tool` and `developer` turns reach the service
|
|
45
|
+
* verbatim, so the union stays complete and `buildChatRequest` does the
|
|
46
|
+
* projection the wire needs.
|
|
47
|
+
*/
|
|
48
|
+
readonly role: 'system' | 'developer' | 'user' | 'assistant' | 'tool';
|
|
49
|
+
readonly content: readonly ContentBlock[];
|
|
50
|
+
/** Id of the call a `tool`-role message answers. */
|
|
51
|
+
readonly toolCallId?: string;
|
|
52
|
+
}
|
|
53
|
+
/** A single model request, narrowed to the fields this adapter reads. */
|
|
54
|
+
export interface GenerateOptions {
|
|
55
|
+
readonly model: string;
|
|
56
|
+
readonly messages: readonly Message[];
|
|
57
|
+
readonly system?: string;
|
|
58
|
+
/**
|
|
59
|
+
* Stop sequences. The service has no wire slot for them, so the adapter
|
|
60
|
+
* refuses a request that sets any (`UNSUPPORTED_OPTION`) instead of
|
|
61
|
+
* generating past where the caller asked it to stop. Tool schemas,
|
|
62
|
+
* `temperature`, and `maxTokens` are declared capability limits in
|
|
63
|
+
* `README.md` and stay dropped.
|
|
64
|
+
*/
|
|
65
|
+
readonly stop?: readonly string[];
|
|
66
|
+
readonly signal?: AbortSignal;
|
|
67
|
+
}
|
|
68
|
+
/** Token accounting for one model call. */
|
|
69
|
+
export interface TokenUsage {
|
|
70
|
+
inputTokens: number;
|
|
71
|
+
outputTokens: number;
|
|
72
|
+
totalTokens?: number;
|
|
73
|
+
}
|
|
74
|
+
/** Stable provider-neutral failure shape. */
|
|
75
|
+
export interface LlmFailure {
|
|
76
|
+
readonly message: string;
|
|
77
|
+
readonly code: string;
|
|
78
|
+
readonly status?: number;
|
|
79
|
+
/** Positive provider-requested retry delay in milliseconds. */
|
|
80
|
+
readonly providerRetryAfterMs?: number;
|
|
81
|
+
}
|
|
82
|
+
/** Why a model response stopped. */
|
|
83
|
+
export type FinishReason = {
|
|
84
|
+
readonly kind: 'stop';
|
|
85
|
+
} | {
|
|
86
|
+
readonly kind: 'max-tokens';
|
|
87
|
+
} | {
|
|
88
|
+
readonly kind: 'error';
|
|
89
|
+
readonly failure: LlmFailure;
|
|
90
|
+
} | {
|
|
91
|
+
readonly kind: 'aborted';
|
|
92
|
+
readonly failure: LlmFailure;
|
|
93
|
+
};
|
|
94
|
+
/** The two chunk shapes this adapter emits, plus the terminal pair. */
|
|
95
|
+
export type StreamChunk = {
|
|
96
|
+
readonly type: 'block-start';
|
|
97
|
+
readonly index: number;
|
|
98
|
+
readonly blockType: string;
|
|
99
|
+
} | {
|
|
100
|
+
readonly type: 'text-delta';
|
|
101
|
+
readonly index: number;
|
|
102
|
+
readonly text: string;
|
|
103
|
+
} | {
|
|
104
|
+
readonly type: 'block-end';
|
|
105
|
+
readonly index: number;
|
|
106
|
+
readonly block: ContentBlock;
|
|
107
|
+
} | {
|
|
108
|
+
readonly type: 'usage';
|
|
109
|
+
readonly usage: TokenUsage;
|
|
110
|
+
} | {
|
|
111
|
+
readonly type: 'finish';
|
|
112
|
+
readonly reason: FinishReason;
|
|
113
|
+
};
|
|
114
|
+
/** Display metadata for one adapter-owned provider route. */
|
|
115
|
+
export interface LlmProviderInfo {
|
|
116
|
+
readonly id: string;
|
|
117
|
+
readonly name: string;
|
|
118
|
+
}
|
|
119
|
+
/** One adapter-advertised model. */
|
|
120
|
+
export interface LlmModelInfo {
|
|
121
|
+
readonly provider: string;
|
|
122
|
+
readonly id: string;
|
|
123
|
+
readonly name: string;
|
|
124
|
+
readonly description?: string;
|
|
125
|
+
readonly inputModalities?: readonly string[];
|
|
126
|
+
}
|
|
127
|
+
/** Exact-route model metadata resolved by its owning adapter. */
|
|
128
|
+
export interface LlmResolvedModelInfo extends LlmModelInfo {
|
|
129
|
+
readonly context?: {
|
|
130
|
+
readonly contextWindow: number;
|
|
131
|
+
};
|
|
132
|
+
}
|
|
133
|
+
/** What `prepareCall` binds: exact model metadata plus one-generation dispatch. */
|
|
134
|
+
export interface PreparedAdapterCall {
|
|
135
|
+
readonly model: LlmResolvedModelInfo;
|
|
136
|
+
readonly stream: (options: GenerateOptions) => AsyncIterable<StreamChunk>;
|
|
137
|
+
}
|
|
138
|
+
/**
|
|
139
|
+
* The adapter face `ctx.llm.registerAdapter()` consumes. Every member is called
|
|
140
|
+
* by the harness at runtime; nothing about the object must be a harness class.
|
|
141
|
+
*
|
|
142
|
+
* The list is the full public surface of the harness `LlmAdapter` base class,
|
|
143
|
+
* including the members that only have defaults there. `LlmRuntime` reads
|
|
144
|
+
* `providerRetryPolicy` while registering the adapter and `imageRequestPricing`
|
|
145
|
+
* during a token-meter measurement, so a duck-typed adapter that omits either
|
|
146
|
+
* throws instead of falling back to the base-class default.
|
|
147
|
+
*/
|
|
148
|
+
export interface LlmAdapterLike {
|
|
149
|
+
providerInfo(provider: string): LlmProviderInfo;
|
|
150
|
+
providerRetryPolicy(provider: string): unknown;
|
|
151
|
+
imageRequestPricing(provider: string, model: string): unknown;
|
|
152
|
+
listModels(provider: string): Promise<readonly LlmModelInfo[]>;
|
|
153
|
+
resolveModel(provider: string, model: string, signal?: AbortSignal): Promise<LlmResolvedModelInfo>;
|
|
154
|
+
prepareCall(provider: string, model: string, signal?: AbortSignal): Promise<PreparedAdapterCall>;
|
|
155
|
+
stream(options: GenerateOptions): AsyncIterable<StreamChunk>;
|
|
156
|
+
}
|
|
157
|
+
/** The `ctx.llm` seam, narrowed to the one call this plugin makes. */
|
|
158
|
+
interface LlmServiceLike {
|
|
159
|
+
registerAdapter(providers: string[], adapter: LlmAdapterLike): Disposable;
|
|
160
|
+
}
|
|
161
|
+
/** The host context slice this plugin touches. */
|
|
162
|
+
export interface HostContext {
|
|
163
|
+
readonly llm: LlmServiceLike;
|
|
164
|
+
readonly logger: {
|
|
165
|
+
info(message: unknown): void;
|
|
166
|
+
warn(message: unknown): void;
|
|
167
|
+
};
|
|
168
|
+
/**
|
|
169
|
+
* Subscribe to a host event. This plugin watches `loader/volatile-update`,
|
|
170
|
+
* which is what a settings write emits once the live row references moved.
|
|
171
|
+
* @param event - the event name.
|
|
172
|
+
* @param listener - the callback.
|
|
173
|
+
* @returns the disposer that removes this listener.
|
|
174
|
+
*/
|
|
175
|
+
on(event: 'loader/volatile-update', listener: () => void): Disposable;
|
|
176
|
+
}
|
|
177
|
+
export {};
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* dsh-chatjimmy: use the chatjimmy.ai model inside DeepSeek Harness.
|
|
3
|
+
*
|
|
4
|
+
* One capability: an `ctx.llm` provider adapter for the reconstructed chat API
|
|
5
|
+
* (see `API.md`). Registering it makes the route selectable in the Web client's
|
|
6
|
+
* model picker, because `buildModelCatalog()` enumerates `ctx.llm.listProviders()`
|
|
7
|
+
* and asks each adapter for `listModels()` / `resolveModel()`.
|
|
8
|
+
*
|
|
9
|
+
* See README.md for the known limits: the service has no tool-calling, no
|
|
10
|
+
* image input, and a 6144-token total context.
|
|
11
|
+
*
|
|
12
|
+
* @module dsh-chatjimmy
|
|
13
|
+
*/
|
|
14
|
+
import type { Volatile } from '@deepseek-ai/cordis';
|
|
15
|
+
import Schema from '@deepseek-ai/schemastery';
|
|
16
|
+
import type { RetryPolicyConfig } from '@deepseek-ai/dsh-llm';
|
|
17
|
+
import type { ChatJimmyConfig } from './protocol.ts';
|
|
18
|
+
import type { HostContext } from './host.ts';
|
|
19
|
+
/** Plugin name as it appears in the loader. */
|
|
20
|
+
export declare const name = "chatjimmy";
|
|
21
|
+
/** The `ctx.llm` provider route this adapter owns. */
|
|
22
|
+
export declare const PROVIDER = "chatjimmy";
|
|
23
|
+
/** The one service this plugin needs mounted. */
|
|
24
|
+
export declare const inject: string[];
|
|
25
|
+
/**
|
|
26
|
+
* Configuration this plugin's row resolves to, as `apply` receives it.
|
|
27
|
+
*
|
|
28
|
+
* Every field a user may edit is `volatile()`, and the loader hands a volatile
|
|
29
|
+
* field a live reference rather than a value: the schema's own output type,
|
|
30
|
+
* `Volatile<T>`. Reading `.get()` at use time is what makes an edit from the
|
|
31
|
+
* Plugins card reach the next request without remounting the route.
|
|
32
|
+
*/
|
|
33
|
+
export interface Config {
|
|
34
|
+
/** Deployment origin. Defaults to `https://chatjimmy.ai`. Volatile: editable from the Plugins card. */
|
|
35
|
+
readonly baseUrl: Volatile<string>;
|
|
36
|
+
/** Model id sent as `chatOptions.selectedModel`. Defaults to `llama3.1-8B`. Volatile. */
|
|
37
|
+
readonly model: Volatile<string>;
|
|
38
|
+
/** Forwarded as `chatOptions.topK`. The site's own client sends 8. Volatile. */
|
|
39
|
+
readonly topK: Volatile<number>;
|
|
40
|
+
/**
|
|
41
|
+
* Total context window in tokens used for call-config validation. Measured
|
|
42
|
+
* against the live service at 6144 (prompt + completion). Volatile.
|
|
43
|
+
*/
|
|
44
|
+
readonly contextWindow: Volatile<number>;
|
|
45
|
+
/**
|
|
46
|
+
* Per-read stream idle watchdog in milliseconds; a stream that produces
|
|
47
|
+
* nothing for this long ends with the `TIMEOUT` failure. Volatile.
|
|
48
|
+
*/
|
|
49
|
+
readonly streamIdleTimeoutMs: Volatile<number>;
|
|
50
|
+
/**
|
|
51
|
+
* Provider-owned retry policy for this route, in the harness
|
|
52
|
+
* `RetryPolicyConfig` shape (`{ mode: 'normal' | 'always', … }`). Absent
|
|
53
|
+
* leaves the harness's own normal defaults. Patch-only: a policy is an
|
|
54
|
+
* operator's decision, not a form field.
|
|
55
|
+
*/
|
|
56
|
+
readonly retryPolicy?: RetryPolicyConfig;
|
|
57
|
+
}
|
|
58
|
+
/** Raw row values, as a profile patch states them and as tests pass them. */
|
|
59
|
+
export type Options = {
|
|
60
|
+
[K in keyof Config]?: Config[K] extends Volatile<infer T> ? T : Config[K];
|
|
61
|
+
};
|
|
62
|
+
/**
|
|
63
|
+
* Row schema as Cordis resolves it: defaults live here, so a deployment only
|
|
64
|
+
* states what it changes.
|
|
65
|
+
*
|
|
66
|
+
* Every field a user may edit is `volatile()`: the settings document accepts
|
|
67
|
+
* only volatile paths, and the browser half's card edits exactly these. The
|
|
68
|
+
* adapter resolves the row per read, so an edit lands on the next request
|
|
69
|
+
* instead of waiting for a remount. `retryPolicy` stays ordinary
|
|
70
|
+
* configuration: patch-only, as its docs say.
|
|
71
|
+
*/
|
|
72
|
+
export declare const Config: Schema<Schemastery.ObjectS<NoInfer<{
|
|
73
|
+
baseUrl: Schema<string, string, "volatile-defined">;
|
|
74
|
+
model: Schema<string, string, "volatile-defined">;
|
|
75
|
+
topK: Schema<number, number, "volatile-defined">;
|
|
76
|
+
contextWindow: Schema<number, number, "volatile-defined">;
|
|
77
|
+
streamIdleTimeoutMs: Schema<number, number, "volatile-defined">;
|
|
78
|
+
retryPolicy: import("@deepseek-ai/schemastery").default<RetryPolicyConfig>;
|
|
79
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
80
|
+
baseUrl: Schema<string, string, "volatile-defined">;
|
|
81
|
+
model: Schema<string, string, "volatile-defined">;
|
|
82
|
+
topK: Schema<number, number, "volatile-defined">;
|
|
83
|
+
contextWindow: Schema<number, number, "volatile-defined">;
|
|
84
|
+
streamIdleTimeoutMs: Schema<number, number, "volatile-defined">;
|
|
85
|
+
retryPolicy: import("@deepseek-ai/schemastery").default<RetryPolicyConfig>;
|
|
86
|
+
}>>, "plain">;
|
|
87
|
+
/**
|
|
88
|
+
* Validate and normalize one configuration row.
|
|
89
|
+
*
|
|
90
|
+
* The row is fed back through the exported `Config` schema, which is the one
|
|
91
|
+
* source of the defaults and the numeric bounds. Cordis already ran the same
|
|
92
|
+
* schema before `apply`, so this only makes `resolveConfig` usable on its own.
|
|
93
|
+
* What the schema cannot express is checked here: invalid values throw rather
|
|
94
|
+
* than being silently defaulted, because a typo'd base URL would otherwise
|
|
95
|
+
* present as an opaque transport failure on the first message.
|
|
96
|
+
*
|
|
97
|
+
* @param config - raw row configuration.
|
|
98
|
+
* @returns the resolved adapter configuration.
|
|
99
|
+
*/
|
|
100
|
+
export declare function resolveConfig(config?: Options): ChatJimmyConfig;
|
|
101
|
+
/**
|
|
102
|
+
* Read the live row out of the references the loader resolved.
|
|
103
|
+
*
|
|
104
|
+
* Every editable field arrives as a `Volatile<T>`; this is the one place that
|
|
105
|
+
* turns them back into the plain values the schema and the adapter understand,
|
|
106
|
+
* so a caller cannot forget one.
|
|
107
|
+
*
|
|
108
|
+
* @param config - the resolved row.
|
|
109
|
+
* @returns plain row values, with absent references left undefined.
|
|
110
|
+
*/
|
|
111
|
+
export declare function liveOptions(config: Config): Options;
|
|
112
|
+
/**
|
|
113
|
+
* Mount the adapter.
|
|
114
|
+
*
|
|
115
|
+
* The adapter is handed the resolver itself, not one resolved row: every
|
|
116
|
+
* configurable field is `volatile()`, so a settings write from the Plugins
|
|
117
|
+
* card changes what the next request uses without remounting the provider
|
|
118
|
+
* route (which would drop the model picker's selection). The row is resolved
|
|
119
|
+
* once here anyway, so an unusable one still fails at mount the way a
|
|
120
|
+
* non-volatile row would.
|
|
121
|
+
*
|
|
122
|
+
* @param ctx - host context; `ctx.llm` must be mounted (`inject` guarantees it).
|
|
123
|
+
* @param config - this plugin's row configuration.
|
|
124
|
+
*/
|
|
125
|
+
export declare function apply(ctx: HostContext, config: Config): void;
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Wire translation between the harness request vocabulary and the
|
|
3
|
+
* chatjimmy.ai HTTP API reconstructed in `API.md`.
|
|
4
|
+
*
|
|
5
|
+
* Everything here is pure so it can be tested without a harness or a network.
|
|
6
|
+
*
|
|
7
|
+
* @module dsh-chatjimmy/protocol
|
|
8
|
+
*/
|
|
9
|
+
import type { RetryPolicyConfig } from '@deepseek-ai/dsh-llm';
|
|
10
|
+
import type { ContentBlock, GenerateOptions, TokenUsage } from './host.ts';
|
|
11
|
+
/** Opening marker of the trailing generation-stats block. */
|
|
12
|
+
export declare const STATS_OPEN = "<|stats|>";
|
|
13
|
+
/** Closing marker of the trailing generation-stats block. */
|
|
14
|
+
export declare const STATS_CLOSE = "<|/stats|>";
|
|
15
|
+
/** One wire message chatjimmy accepts. */
|
|
16
|
+
interface WireMessage {
|
|
17
|
+
role: 'user' | 'assistant';
|
|
18
|
+
content: string;
|
|
19
|
+
}
|
|
20
|
+
/** The body `POST /api/chat` expects. */
|
|
21
|
+
interface ChatRequestBody {
|
|
22
|
+
messages: WireMessage[];
|
|
23
|
+
chatOptions: {
|
|
24
|
+
selectedModel: string;
|
|
25
|
+
systemPrompt: string;
|
|
26
|
+
topK: number;
|
|
27
|
+
};
|
|
28
|
+
attachment: null;
|
|
29
|
+
}
|
|
30
|
+
/** The stats fields this adapter reads; the payload carries more. */
|
|
31
|
+
export interface ChatStats {
|
|
32
|
+
prefill_tokens?: number;
|
|
33
|
+
decode_tokens?: number;
|
|
34
|
+
total_tokens?: number;
|
|
35
|
+
done_reason?: string;
|
|
36
|
+
reason?: string;
|
|
37
|
+
[key: string]: unknown;
|
|
38
|
+
}
|
|
39
|
+
/** Resolved adapter configuration. */
|
|
40
|
+
export interface ChatJimmyConfig {
|
|
41
|
+
baseUrl: string;
|
|
42
|
+
model: string;
|
|
43
|
+
topK: number;
|
|
44
|
+
contextWindow: number;
|
|
45
|
+
/** Per-read idle watchdog: a stream that produces nothing for this long fails with `TIMEOUT`. */
|
|
46
|
+
streamIdleTimeoutMs: number;
|
|
47
|
+
/**
|
|
48
|
+
* Provider-owned retry policy, reported to the harness at registration. Absent
|
|
49
|
+
* means the harness's own normal defaults.
|
|
50
|
+
*/
|
|
51
|
+
retryPolicy?: RetryPolicyConfig;
|
|
52
|
+
}
|
|
53
|
+
/** Flatten a block list to plain text. */
|
|
54
|
+
export declare function flatten(blocks: readonly ContentBlock[]): string;
|
|
55
|
+
/**
|
|
56
|
+
* Build the exact request body for one model call: history is flattened to
|
|
57
|
+
* text, every system-role message is hoisted into the single `systemPrompt`
|
|
58
|
+
* slot, and the caller's `system` text leads it.
|
|
59
|
+
*/
|
|
60
|
+
export declare function buildChatRequest(options: GenerateOptions, config: ChatJimmyConfig): ChatRequestBody;
|
|
61
|
+
/** Parse the stats payload, returning undefined for malformed JSON. */
|
|
62
|
+
export declare function parseStats(raw: string): ChatStats | undefined;
|
|
63
|
+
/**
|
|
64
|
+
* True when a stats `reason` reports the backend's own context-limit refusal.
|
|
65
|
+
* @param reason - the stats `reason` field, when present.
|
|
66
|
+
*/
|
|
67
|
+
export declare function isContextLimitReason(reason: unknown): boolean;
|
|
68
|
+
/**
|
|
69
|
+
* Map backend token counters onto harness usage.
|
|
70
|
+
* @param stats - parsed stats, when the stream carried them.
|
|
71
|
+
* @returns disjoint harness counts (the provider reports no cache split), or
|
|
72
|
+
* undefined when the stats carry neither prompt nor output counter. The total
|
|
73
|
+
* is the provider's own, else the sum when both parts are known, else omitted.
|
|
74
|
+
*/
|
|
75
|
+
export declare function mapUsage(stats: ChatStats | undefined): TokenUsage | undefined;
|
|
76
|
+
/**
|
|
77
|
+
* Splits the generated text from the trailing `<|stats|>…<|/stats|>` block
|
|
78
|
+
* without ever emitting a partial marker.
|
|
79
|
+
*
|
|
80
|
+
* The backend appends the block to the same byte stream as the completion, so a
|
|
81
|
+
* reader that forwarded chunks verbatim would leak `{"prefill_tokens":…}` into
|
|
82
|
+
* the model's visible answer. Text is held back only as far as a marker could
|
|
83
|
+
* still be forming, so time-to-first-token is unaffected.
|
|
84
|
+
*/
|
|
85
|
+
export declare class StatsStreamFilter {
|
|
86
|
+
#private;
|
|
87
|
+
/**
|
|
88
|
+
* Absorb one decoded chunk.
|
|
89
|
+
* @param text - newly decoded text.
|
|
90
|
+
* @returns the prefix that is certainly completion text.
|
|
91
|
+
*/
|
|
92
|
+
push(text: string): string;
|
|
93
|
+
/**
|
|
94
|
+
* Release whatever the stream ended with.
|
|
95
|
+
* @returns residual completion text; empty when the body ended on the marker.
|
|
96
|
+
*/
|
|
97
|
+
flush(): string;
|
|
98
|
+
/** Stats the stream carried, when the sentinel was complete. */
|
|
99
|
+
get stats(): ChatStats | undefined;
|
|
100
|
+
}
|
|
101
|
+
export {};
|
package/locale/en.json
ADDED