dsh-local-ai 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/LICENSE +201 -0
- package/README.es.md +211 -0
- package/README.hi.md +211 -0
- package/README.md +211 -0
- package/README.pt.md +211 -0
- package/README.zh.md +211 -0
- package/THIRD_PARTY_NOTICES.md +20 -0
- package/cordis.patch.yml +44 -0
- package/lib/index.js +1334 -0
- package/lib/types/adapter.d.ts +39 -0
- package/lib/types/adapter.d.ts.map +1 -0
- package/lib/types/adapter.js +190 -0
- package/lib/types/adapter.js.map +1 -0
- package/lib/types/config.d.ts +109 -0
- package/lib/types/config.d.ts.map +1 -0
- package/lib/types/config.js +161 -0
- package/lib/types/config.js.map +1 -0
- package/lib/types/health.d.ts +64 -0
- package/lib/types/health.d.ts.map +1 -0
- package/lib/types/health.js +92 -0
- package/lib/types/health.js.map +1 -0
- package/lib/types/index.d.ts +43 -0
- package/lib/types/index.d.ts.map +1 -0
- package/lib/types/index.js +232 -0
- package/lib/types/index.js.map +1 -0
- package/lib/types/ollama.d.ts +91 -0
- package/lib/types/ollama.d.ts.map +1 -0
- package/lib/types/ollama.js +184 -0
- package/lib/types/ollama.js.map +1 -0
- package/lib/types/route.d.ts +52 -0
- package/lib/types/route.d.ts.map +1 -0
- package/lib/types/route.js +119 -0
- package/lib/types/route.js.map +1 -0
- package/lib/types/sanitize.d.ts +58 -0
- package/lib/types/sanitize.d.ts.map +1 -0
- package/lib/types/sanitize.js +110 -0
- package/lib/types/sanitize.js.map +1 -0
- package/lib/types/serialize.d.ts +65 -0
- package/lib/types/serialize.d.ts.map +1 -0
- package/lib/types/serialize.js +149 -0
- package/lib/types/serialize.js.map +1 -0
- package/lib/types/translate.d.ts +57 -0
- package/lib/types/translate.d.ts.map +1 -0
- package/lib/types/translate.js +169 -0
- package/lib/types/translate.js.map +1 -0
- package/lib/types/version.d.ts +6 -0
- package/lib/types/version.d.ts.map +1 -0
- package/lib/types/version.js +6 -0
- package/lib/types/version.js.map +1 -0
- package/package.json +141 -0
- package/src/adapter.ts +158 -0
- package/src/config.ts +243 -0
- package/src/health.ts +133 -0
- package/src/index.ts +279 -0
- package/src/ollama.ts +274 -0
- package/src/route.ts +127 -0
- package/src/sanitize.ts +114 -0
- package/src/serialize.ts +178 -0
- package/src/translate.ts +205 -0
- package/src/version.ts +5 -0
package/src/sanitize.ts
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Display/log sanitization for `dsh-local-ai`. Every value shown to the model
|
|
3
|
+
* or to the user (tool results, the `/ollama` command, error messages) passes
|
|
4
|
+
* through one of these pure functions first, so an endpoint address or a local
|
|
5
|
+
* path can never leak credentials, secret query parameters, or unbounded text.
|
|
6
|
+
*
|
|
7
|
+
* All functions are pure: they depend only on their arguments, never on
|
|
8
|
+
* process state (the caller supplies a home directory for path redaction).
|
|
9
|
+
* @module dsh-local-ai/sanitize
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
/** Placeholder substituted for a redacted secret or credential. */
|
|
13
|
+
export const REDACTED = '[REDACTED]'
|
|
14
|
+
|
|
15
|
+
/** Control characters (C0 + DEL) stripped from every sanitized value. */
|
|
16
|
+
const CONTROL_CHARS = /[\u0000-\u001f\u007f]/gu
|
|
17
|
+
|
|
18
|
+
/** Secret-shaped query-parameter keys removed from endpoint URLs. */
|
|
19
|
+
const SECRET_KEY_PATTERN = /key|token|secret|password|credential|auth/iu
|
|
20
|
+
|
|
21
|
+
/** Built-in secret literal patterns redacted from arbitrary text. */
|
|
22
|
+
const BUILTIN_SECRET_PATTERNS = [
|
|
23
|
+
/\bsk-[A-Za-z0-9]{16,}\b/u,
|
|
24
|
+
/\bghp_[A-Za-z0-9]{20,}\b/u,
|
|
25
|
+
/\bgho_[A-Za-z0-9]{20,}\b/u,
|
|
26
|
+
/\bAKIA[0-9A-Z]{16}\b/u,
|
|
27
|
+
/\bBearer\s+[A-Za-z0-9._~+/=-]{8,}\b/u,
|
|
28
|
+
/-----BEGIN [A-Z ]*PRIVATE KEY-----[A-Za-z0-9+/=\s]*-----END [A-Z ]*PRIVATE KEY-----/u,
|
|
29
|
+
] as const
|
|
30
|
+
|
|
31
|
+
/** Remove C0/DEL control characters from a string. */
|
|
32
|
+
export function stripControl(value: string): string {
|
|
33
|
+
return value.replace(CONTROL_CHARS, '')
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Truncate a string to `maxChars`, appending an ellipsis when it was cut.
|
|
38
|
+
* A non-positive `maxChars` yields the empty string.
|
|
39
|
+
* @param value - the string to bound.
|
|
40
|
+
* @param maxChars - maximum returned length, including the ellipsis.
|
|
41
|
+
* @returns the bounded string.
|
|
42
|
+
*/
|
|
43
|
+
export function truncate(value: string, maxChars: number): string {
|
|
44
|
+
if (value.length <= maxChars) return value
|
|
45
|
+
if (maxChars <= 1) return maxChars <= 0 ? '' : '…'
|
|
46
|
+
return `${value.slice(0, maxChars - 1)}…`
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Sanitize an endpoint address for display: strip the URL userinfo
|
|
51
|
+
* (`user:pass@`), drop query parameters whose key looks like a secret, strip
|
|
52
|
+
* control characters, and bound the length. Values that are not parseable as
|
|
53
|
+
* a URL are still stripped and truncated.
|
|
54
|
+
* @param value - the raw endpoint (URL string or anything stringifiable).
|
|
55
|
+
* @param maxChars - maximum returned length.
|
|
56
|
+
* @returns the sanitized endpoint text.
|
|
57
|
+
*/
|
|
58
|
+
export function sanitizeEndpoint(value: unknown, maxChars = 2048): string {
|
|
59
|
+
const text = stripControl(typeof value === 'string' ? value : String(value))
|
|
60
|
+
let out: string
|
|
61
|
+
try {
|
|
62
|
+
const url = new URL(text)
|
|
63
|
+
url.username = ''
|
|
64
|
+
url.password = ''
|
|
65
|
+
for (const key of [...url.searchParams.keys()]) {
|
|
66
|
+
if (SECRET_KEY_PATTERN.test(key)) url.searchParams.delete(key)
|
|
67
|
+
}
|
|
68
|
+
out = url.href
|
|
69
|
+
} catch {
|
|
70
|
+
out = text
|
|
71
|
+
}
|
|
72
|
+
return truncate(out, maxChars)
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Sanitize a local path for display: strip control characters, redact a
|
|
77
|
+
* leading home directory to `~`, and bound the length.
|
|
78
|
+
* @param value - the raw path (string or anything stringifiable).
|
|
79
|
+
* @param home - the user's home directory to redact; omit to skip redaction.
|
|
80
|
+
* @param maxChars - maximum returned length.
|
|
81
|
+
* @returns the sanitized path text.
|
|
82
|
+
*/
|
|
83
|
+
export function sanitizePath(value: unknown, home = '', maxChars = 1024): string {
|
|
84
|
+
const text = stripControl(typeof value === 'string' ? value : String(value))
|
|
85
|
+
const redacted = home.length > 0 && text.startsWith(home)
|
|
86
|
+
? `~${text.slice(home.length)}`
|
|
87
|
+
: text
|
|
88
|
+
return truncate(redacted, maxChars)
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* Redact built-in secret literals (API keys, GitHub tokens, AWS keys, bearer
|
|
93
|
+
* credentials, PEM private keys) from arbitrary text. Control characters are
|
|
94
|
+
* stripped first.
|
|
95
|
+
* @param value - the raw text (string or anything stringifiable).
|
|
96
|
+
* @returns the text with secret literals replaced by {@link REDACTED}.
|
|
97
|
+
*/
|
|
98
|
+
export function redactSecrets(value: unknown): string {
|
|
99
|
+
const text = stripControl(typeof value === 'string' ? value : String(value))
|
|
100
|
+
let out = text
|
|
101
|
+
for (const pattern of BUILTIN_SECRET_PATTERNS) out = out.replace(pattern, REDACTED)
|
|
102
|
+
return out
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Sanitize arbitrary display text: redact secrets, strip control characters,
|
|
107
|
+
* and bound the length.
|
|
108
|
+
* @param value - the raw text (string or anything stringifiable).
|
|
109
|
+
* @param maxChars - maximum returned length.
|
|
110
|
+
* @returns the sanitized text.
|
|
111
|
+
*/
|
|
112
|
+
export function sanitizeText(value: unknown, maxChars = 4000): string {
|
|
113
|
+
return truncate(redactSecrets(value), maxChars)
|
|
114
|
+
}
|
package/src/serialize.ts
ADDED
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Serialize harness messages and requests into the Ollama `/api/chat` wire
|
|
3
|
+
* vocabulary. User text is joined; assistant text becomes `content` and tool
|
|
4
|
+
* calls become `tool_calls` (with arguments parsed from the raw JSON string to
|
|
5
|
+
* the object Ollama expects); tool results become separate `tool` messages.
|
|
6
|
+
* Core image blocks are rejected explicitly because this route is text-only;
|
|
7
|
+
* unknown declaration-merged block types retain the documented extension
|
|
8
|
+
* fallback (ignored for content, retained as text where text is expected).
|
|
9
|
+
* @module dsh-local-ai/serialize
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { contentHasImage, LlmError } from '@deepseek-ai/dsh-llm'
|
|
13
|
+
import type { ContentBlock, GenerateOptions, Message, ToolSchema } from '@deepseek-ai/dsh-llm'
|
|
14
|
+
import type { ResolvedConfig } from './config.ts'
|
|
15
|
+
|
|
16
|
+
/** One Ollama chat message on the wire. */
|
|
17
|
+
export interface OllamaWireMessage {
|
|
18
|
+
role: 'system' | 'user' | 'assistant' | 'tool'
|
|
19
|
+
content: string
|
|
20
|
+
tool_calls?: Array<{ function: { name: string; arguments: Record<string, unknown> } }>
|
|
21
|
+
tool_name?: string
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/** The Ollama `/api/chat` request body. */
|
|
25
|
+
export interface OllamaWireRequest {
|
|
26
|
+
model: string
|
|
27
|
+
messages: OllamaWireMessage[]
|
|
28
|
+
stream: boolean
|
|
29
|
+
options?: Record<string, unknown>
|
|
30
|
+
tools?: Array<{ type: 'function'; function: ToolSchema }>
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** Join the text blocks of a message (used for user/tool-result content). */
|
|
34
|
+
function flattenText(blocks: readonly ContentBlock[]): string {
|
|
35
|
+
return blocks
|
|
36
|
+
.filter(block => block.type === 'text')
|
|
37
|
+
.map(block => block.text)
|
|
38
|
+
.join('')
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Reject core image content before any text-flattening path can silently erase it. */
|
|
42
|
+
function assertTextOnly(blocks: readonly ContentBlock[]): void {
|
|
43
|
+
if (contentHasImage(blocks)) {
|
|
44
|
+
throw new LlmError('The Ollama adapter does not support image content.', 'UNSUPPORTED_CONTENT')
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Parse a tool-call argument string into the object Ollama expects. The raw
|
|
50
|
+
* string is guaranteed by the harness contract to be JSON; a malformed value
|
|
51
|
+
* from a hand-built call degrades to a single `value` field rather than
|
|
52
|
+
* bricking the whole session.
|
|
53
|
+
* @param raw - the raw JSON string produced by the model.
|
|
54
|
+
* @returns the parsed object, or a `{ value }` fallback.
|
|
55
|
+
*/
|
|
56
|
+
export function parseToolArguments(raw: string): Record<string, unknown> {
|
|
57
|
+
try {
|
|
58
|
+
const parsed = JSON.parse(raw) as unknown
|
|
59
|
+
if (typeof parsed === 'object' && parsed !== null && !Array.isArray(parsed)) {
|
|
60
|
+
return parsed as Record<string, unknown>
|
|
61
|
+
}
|
|
62
|
+
return { value: raw }
|
|
63
|
+
} catch {
|
|
64
|
+
return { value: raw }
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** Serialize one assistant message (text + tool calls). */
|
|
69
|
+
function serializeAssistant(message: Message): OllamaWireMessage {
|
|
70
|
+
const text = flattenText(message.content)
|
|
71
|
+
const toolCalls = message.content
|
|
72
|
+
.filter(block => block.type === 'tool-call')
|
|
73
|
+
.map(block => ({
|
|
74
|
+
function: { name: block.name, arguments: parseToolArguments(block.arguments) },
|
|
75
|
+
}))
|
|
76
|
+
|
|
77
|
+
return {
|
|
78
|
+
role: 'assistant',
|
|
79
|
+
content: text,
|
|
80
|
+
...toolCalls.length > 0 ? { tool_calls: toolCalls } : {},
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/** Resolve a tool-result block's name from the assistant tool calls that precede it. */
|
|
85
|
+
function toolNameOf(callId: string, namesByCallId: ReadonlyMap<string, string>): string | undefined {
|
|
86
|
+
return namesByCallId.get(callId)
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Serialize the conversation. `tool-result` blocks become standalone
|
|
91
|
+
* `{role: 'tool'}` messages; the harness puts each tool result in its own
|
|
92
|
+
* user-role message, so a mixed user message contributes its text first and
|
|
93
|
+
* its tool results as separate wire messages after. Assistant tool calls are
|
|
94
|
+
* indexed first so their results can carry the tool name Ollama needs.
|
|
95
|
+
* @param messages - the harness conversation, in order.
|
|
96
|
+
* @returns the wire messages; order preserved, each tool result expanded into its own entry.
|
|
97
|
+
*/
|
|
98
|
+
export function serializeMessages(messages: Message[]): OllamaWireMessage[] {
|
|
99
|
+
const namesByCallId = new Map<string, string>()
|
|
100
|
+
for (const message of messages) {
|
|
101
|
+
if (message.role !== 'assistant') continue
|
|
102
|
+
for (const block of message.content) {
|
|
103
|
+
if (block.type === 'tool-call') namesByCallId.set(String(block.id), block.name)
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
const wire: OllamaWireMessage[] = []
|
|
108
|
+
for (const message of messages) {
|
|
109
|
+
assertTextOnly(message.content)
|
|
110
|
+
if (message.role === 'system') {
|
|
111
|
+
wire.push({ role: 'system', content: flattenText(message.content) })
|
|
112
|
+
continue
|
|
113
|
+
}
|
|
114
|
+
if (message.role === 'assistant') {
|
|
115
|
+
wire.push(serializeAssistant(message))
|
|
116
|
+
continue
|
|
117
|
+
}
|
|
118
|
+
// user role: tool results ride in user messages in the harness
|
|
119
|
+
// vocabulary, but Ollama wants them as role:'tool' messages.
|
|
120
|
+
const toolResults = message.content.filter(block => block.type === 'tool-result')
|
|
121
|
+
const text = flattenText(message.content)
|
|
122
|
+
if (text.length > 0 || toolResults.length === 0) {
|
|
123
|
+
wire.push({ role: 'user', content: text })
|
|
124
|
+
}
|
|
125
|
+
for (const result of toolResults) {
|
|
126
|
+
const name = toolNameOf(String(result.toolCallId), namesByCallId)
|
|
127
|
+
wire.push({
|
|
128
|
+
role: 'tool',
|
|
129
|
+
// Empty tool output still needs SOME content on the wire.
|
|
130
|
+
content: flattenText(result.content) || '(no output)',
|
|
131
|
+
...name === undefined ? {} : { tool_name: name },
|
|
132
|
+
})
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
return wire
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* Build the full wire request. Always streaming (`stream: true`); optional
|
|
140
|
+
* fields are omitted rather than sent as null, so Ollama defaults apply.
|
|
141
|
+
* `temperature` resolves request → model mapping → plugin default; `num_predict`
|
|
142
|
+
* is the harness-materialized `maxTokens`.
|
|
143
|
+
* @param options - the harness request (model, history, system, tools, sampling).
|
|
144
|
+
* @param resolved - the resolved plugin config.
|
|
145
|
+
* @returns the `/api/chat` request body.
|
|
146
|
+
*/
|
|
147
|
+
export function serializeRequest(
|
|
148
|
+
options: GenerateOptions,
|
|
149
|
+
resolved: ResolvedConfig,
|
|
150
|
+
): OllamaWireRequest {
|
|
151
|
+
const mapping = resolved.models.find(entry => entry.name === options.model)
|
|
152
|
+
const model = mapping?.model ?? options.model
|
|
153
|
+
|
|
154
|
+
const messages: OllamaWireMessage[] = []
|
|
155
|
+
if (options.system !== undefined) {
|
|
156
|
+
messages.push({ role: 'system', content: options.system })
|
|
157
|
+
}
|
|
158
|
+
messages.push(...serializeMessages(options.messages))
|
|
159
|
+
|
|
160
|
+
const temperature = options.temperature ?? mapping?.temperature ?? resolved.temperature
|
|
161
|
+
const ollamaOptions: Record<string, unknown> = {}
|
|
162
|
+
if (temperature !== undefined) ollamaOptions.temperature = temperature
|
|
163
|
+
if (options.maxTokens !== undefined) ollamaOptions.num_predict = options.maxTokens
|
|
164
|
+
if (options.stop !== undefined && options.stop.length > 0) ollamaOptions.stop = options.stop
|
|
165
|
+
|
|
166
|
+
const tools = options.tools?.map(tool => ({
|
|
167
|
+
type: 'function' as const,
|
|
168
|
+
function: tool,
|
|
169
|
+
}))
|
|
170
|
+
|
|
171
|
+
return {
|
|
172
|
+
model,
|
|
173
|
+
messages,
|
|
174
|
+
stream: true,
|
|
175
|
+
...Object.keys(ollamaOptions).length > 0 ? { options: ollamaOptions } : {},
|
|
176
|
+
...tools !== undefined && tools.length > 0 ? { tools } : {},
|
|
177
|
+
}
|
|
178
|
+
}
|
package/src/translate.ts
ADDED
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Translate Ollama NDJSON chat chunks into the harness `StreamChunk` protocol.
|
|
3
|
+
* Ollama streams one JSON object per line: `message.content` and
|
|
4
|
+
* `message.thinking` are incremental deltas, while `message.tool_calls`
|
|
5
|
+
* carries the CUMULATIVE arguments object, so tool-call deltas are computed by
|
|
6
|
+
* longest-common-prefix diffing. Usage and the finish reason are deferred to
|
|
7
|
+
* the `done: true` chunk, guaranteeing no chunk follows `finish`.
|
|
8
|
+
* @module dsh-local-ai/translate
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { CallId, EMPTY_RESPONSE_CODE, LlmError } from '@deepseek-ai/dsh-llm'
|
|
12
|
+
import type { ContentBlock, FinishReason, StreamChunk, TokenUsage } from '@deepseek-ai/dsh-llm'
|
|
13
|
+
|
|
14
|
+
/** One Ollama streaming chat chunk on the wire. */
|
|
15
|
+
export interface OllamaWireChunk {
|
|
16
|
+
model?: string
|
|
17
|
+
message?: {
|
|
18
|
+
role?: string
|
|
19
|
+
content?: string
|
|
20
|
+
thinking?: string
|
|
21
|
+
tool_calls?: Array<{ function?: { name?: string; arguments?: Record<string, unknown> } }>
|
|
22
|
+
}
|
|
23
|
+
done?: boolean
|
|
24
|
+
done_reason?: string
|
|
25
|
+
prompt_eval_count?: number
|
|
26
|
+
eval_count?: number
|
|
27
|
+
error?: string
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/** One open block under assembly. */
|
|
31
|
+
interface OpenBlock {
|
|
32
|
+
index: number
|
|
33
|
+
kind: 'text' | 'reasoning' | 'tool-call'
|
|
34
|
+
text: string
|
|
35
|
+
/** tool-call only */
|
|
36
|
+
callId?: string
|
|
37
|
+
name?: string
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* Map the Ollama `done_reason` vocabulary to the harness FinishReason.
|
|
42
|
+
* @param reason - the wire `done_reason` string.
|
|
43
|
+
* @returns the mapped reason; unrecognized values become `{kind: 'error'}`.
|
|
44
|
+
*/
|
|
45
|
+
export function mapFinishReason(reason: string): FinishReason {
|
|
46
|
+
switch (reason) {
|
|
47
|
+
case 'stop': return { kind: 'stop' }
|
|
48
|
+
case 'tool_calls': return { kind: 'tool-calls' }
|
|
49
|
+
case 'length': return { kind: 'max-tokens' }
|
|
50
|
+
default:
|
|
51
|
+
return {
|
|
52
|
+
kind: 'error',
|
|
53
|
+
failure: { message: `model stopped: ${reason}`, code: reason.toUpperCase() },
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** Assemble the final ContentBlock for one open block. */
|
|
59
|
+
function closeBlock(block: OpenBlock): ContentBlock {
|
|
60
|
+
switch (block.kind) {
|
|
61
|
+
case 'text': return { type: 'text', text: block.text }
|
|
62
|
+
case 'reasoning': return { type: 'reasoning', text: block.text }
|
|
63
|
+
case 'tool-call': return {
|
|
64
|
+
type: 'tool-call',
|
|
65
|
+
id: CallId(block.callId ?? ''),
|
|
66
|
+
name: block.name ?? '',
|
|
67
|
+
arguments: block.text,
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Compute the append delta from a cumulative JSON string, so the harness's
|
|
74
|
+
* delta-concatenating assembler reconstructs the full arguments. Ollama grows
|
|
75
|
+
* the arguments object monotonically, so the delta is everything past the
|
|
76
|
+
* longest common prefix with the previously seen string.
|
|
77
|
+
* @param previous - the previously seen cumulative JSON (or `''`).
|
|
78
|
+
* @param next - the new cumulative JSON.
|
|
79
|
+
* @returns the fragment to append.
|
|
80
|
+
*/
|
|
81
|
+
export function argumentsDelta(previous: string, next: string): string {
|
|
82
|
+
if (next.startsWith(previous)) return next.slice(previous.length)
|
|
83
|
+
let index = 0
|
|
84
|
+
while (index < previous.length && index < next.length && previous[index] === next[index]) index += 1
|
|
85
|
+
return next.slice(index)
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* Consume Ollama NDJSON lines and yield StreamChunks. Text and reasoning deltas
|
|
90
|
+
* stream as they arrive; tool-call deltas are diffed from the cumulative wire
|
|
91
|
+
* arguments; `block-end`, `usage`, and `finish` are deferred to the `done`
|
|
92
|
+
* chunk. A `stop` finish with no opened blocks maps to an `EMPTY_RESPONSE`
|
|
93
|
+
* error finish.
|
|
94
|
+
* @param lines - newline-delimited Ollama chat payloads.
|
|
95
|
+
* @returns deltas as they arrive; `block-end`s, `usage`, and `finish` deferred to `done`.
|
|
96
|
+
*/
|
|
97
|
+
export async function* translate(lines: AsyncIterable<string>): AsyncGenerator<StreamChunk> {
|
|
98
|
+
let nextIndex = 0
|
|
99
|
+
let textBlock: OpenBlock | undefined
|
|
100
|
+
let reasoningBlock: OpenBlock | undefined
|
|
101
|
+
const toolBlocks = new Map<number, OpenBlock>()
|
|
102
|
+
const toolArguments = new Map<number, string>()
|
|
103
|
+
const order: OpenBlock[] = []
|
|
104
|
+
|
|
105
|
+
function open(kind: OpenBlock['kind']): OpenBlock {
|
|
106
|
+
const block: OpenBlock = { index: nextIndex++, kind, text: '' }
|
|
107
|
+
order.push(block)
|
|
108
|
+
return block
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
for await (const line of lines) {
|
|
112
|
+
if (line.length === 0) continue
|
|
113
|
+
|
|
114
|
+
let chunk: OllamaWireChunk
|
|
115
|
+
try {
|
|
116
|
+
chunk = JSON.parse(line) as OllamaWireChunk
|
|
117
|
+
} catch {
|
|
118
|
+
throw new LlmError(`malformed Ollama NDJSON payload: ${line.slice(0, 120)}`, 'MALFORMED_RESPONSE')
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
if (typeof chunk.error === 'string' && chunk.error.length > 0) {
|
|
122
|
+
throw new LlmError(chunk.error, 'PROVIDER')
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
const message = chunk.message
|
|
126
|
+
if (message !== undefined) {
|
|
127
|
+
const reasoning = message.thinking
|
|
128
|
+
if (typeof reasoning === 'string' && reasoning.length > 0) {
|
|
129
|
+
if (!reasoningBlock) {
|
|
130
|
+
reasoningBlock = open('reasoning')
|
|
131
|
+
yield { type: 'block-start', index: reasoningBlock.index, blockType: 'reasoning' }
|
|
132
|
+
}
|
|
133
|
+
reasoningBlock.text += reasoning
|
|
134
|
+
yield { type: 'reasoning-delta', index: reasoningBlock.index, text: reasoning }
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
const content = message.content
|
|
138
|
+
if (typeof content === 'string' && content.length > 0) {
|
|
139
|
+
if (!textBlock) {
|
|
140
|
+
textBlock = open('text')
|
|
141
|
+
yield { type: 'block-start', index: textBlock.index, blockType: 'text' }
|
|
142
|
+
}
|
|
143
|
+
textBlock.text += content
|
|
144
|
+
yield { type: 'text-delta', index: textBlock.index, text: content }
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
const toolCalls = message.tool_calls ?? []
|
|
148
|
+
for (let callIndex = 0; callIndex < toolCalls.length; callIndex++) {
|
|
149
|
+
const call = toolCalls[callIndex]
|
|
150
|
+
let block = toolBlocks.get(callIndex)
|
|
151
|
+
if (!block) {
|
|
152
|
+
block = open('tool-call')
|
|
153
|
+
toolBlocks.set(callIndex, block)
|
|
154
|
+
toolArguments.set(callIndex, '')
|
|
155
|
+
yield { type: 'block-start', index: block.index, blockType: 'tool-call' }
|
|
156
|
+
}
|
|
157
|
+
if (call?.function?.name !== undefined && block.name === undefined) {
|
|
158
|
+
block.name = call.function.name
|
|
159
|
+
}
|
|
160
|
+
const cumulative = JSON.stringify(call?.function?.arguments ?? {})
|
|
161
|
+
const previous = toolArguments.get(callIndex) ?? ''
|
|
162
|
+
if (cumulative !== previous) {
|
|
163
|
+
const fragment = argumentsDelta(previous, cumulative)
|
|
164
|
+
toolArguments.set(callIndex, cumulative)
|
|
165
|
+
block.text += fragment
|
|
166
|
+
yield {
|
|
167
|
+
type: 'tool-call-delta',
|
|
168
|
+
index: block.index,
|
|
169
|
+
id: CallId(block.callId ?? ''),
|
|
170
|
+
...block.name !== undefined ? { name: block.name } : {},
|
|
171
|
+
argumentsDelta: fragment,
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
if (chunk.done === true) {
|
|
178
|
+
for (const block of order) {
|
|
179
|
+
yield { type: 'block-end', index: block.index, block: closeBlock(block) }
|
|
180
|
+
}
|
|
181
|
+
const promptCount = chunk.prompt_eval_count
|
|
182
|
+
const evalCount = chunk.eval_count
|
|
183
|
+
if (promptCount !== undefined || evalCount !== undefined) {
|
|
184
|
+
const usage: TokenUsage = {
|
|
185
|
+
inputTokens: promptCount ?? 0,
|
|
186
|
+
outputTokens: evalCount ?? 0,
|
|
187
|
+
}
|
|
188
|
+
yield { type: 'usage', usage }
|
|
189
|
+
}
|
|
190
|
+
const reason = mapFinishReason(chunk.done_reason ?? 'stop')
|
|
191
|
+
yield {
|
|
192
|
+
type: 'finish',
|
|
193
|
+
reason: reason.kind === 'stop' && order.length === 0
|
|
194
|
+
? {
|
|
195
|
+
kind: 'error',
|
|
196
|
+
failure: { message: 'model returned a completed response with no content', code: EMPTY_RESPONSE_CODE },
|
|
197
|
+
}
|
|
198
|
+
: reason,
|
|
199
|
+
}
|
|
200
|
+
return
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
throw new LlmError('Ollama NDJSON stream ended without a done chunk', 'STREAM_CLOSED')
|
|
205
|
+
}
|