apple-llm 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +94 -0
- package/README.md +251 -112
- package/dist/ai-sdk.cjs +281 -0
- package/dist/ai-sdk.d.cts +54 -0
- package/dist/ai-sdk.d.ts +54 -0
- package/dist/ai-sdk.js +281 -0
- package/dist/chunk-GM325EMJ.js +2584 -0
- package/dist/chunk-NRDZIP5G.cjs +2588 -0
- package/dist/chunk-OQATUZWF.cjs +81 -0
- package/dist/chunk-ZB4RDEPW.js +81 -0
- package/dist/cli.js +3163 -24
- package/dist/client-CXewzZTj.d.cts +1160 -0
- package/dist/client-CXewzZTj.d.ts +1160 -0
- package/dist/index.cjs +110 -1667
- package/dist/index.d.cts +84 -627
- package/dist/index.d.ts +84 -627
- package/dist/index.js +31 -1
- package/dist/server.cjs +361 -0
- package/dist/server.d.cts +51 -0
- package/dist/server.d.ts +51 -0
- package/dist/server.js +361 -0
- package/package.json +66 -14
- package/swift/helper.swift +748 -255
- package/dist/chunk-FQTRQ3KP.js +0 -1590
- package/dist/cli.cjs +0 -2027
package/swift/helper.swift
CHANGED
|
@@ -1,9 +1,7 @@
|
|
|
1
1
|
//
|
|
2
2
|
// apple-llm helper — the whole on-device path.
|
|
3
3
|
//
|
|
4
|
-
//
|
|
5
|
-
// lived as a TypeScript string literal in `src/llm/apple-helper.ts`. This file
|
|
6
|
-
// is the single source of truth: `scripts/embed-helper.mjs` copies it verbatim
|
|
4
|
+
// This file is the single source of truth: `scripts/embed-helper.mjs` copies it verbatim
|
|
7
5
|
// into both the npm and the pip package, and a test in each asserts the copies
|
|
8
6
|
// still hash equal to this file.
|
|
9
7
|
//
|
|
@@ -11,26 +9,37 @@
|
|
|
11
9
|
// which ships with macOS 27 but is gated behind a machine-wide `sudo fm
|
|
12
10
|
// license`. Do not depend on `fm`.
|
|
13
11
|
//
|
|
14
|
-
// Protocol:
|
|
12
|
+
// Protocol (version 2; every version-1 request still works unchanged):
|
|
15
13
|
// helper --probe -> stdout: one JSON object describing this machine.
|
|
16
|
-
// helper --serve -> one request JSON per stdin line,
|
|
17
|
-
//
|
|
18
|
-
//
|
|
19
|
-
//
|
|
20
|
-
//
|
|
21
|
-
//
|
|
22
|
-
//
|
|
14
|
+
// helper --serve -> one request JSON per stdin line; for each request, zero
|
|
15
|
+
// or more *event* lines ("done":false) and then exactly one
|
|
16
|
+
// *final* line (anything without "done":false), in order.
|
|
17
|
+
// Plain generate emits no events, so a version-1 client
|
|
18
|
+
// that reads one line per request keeps working.
|
|
19
|
+
// The client must not send the next request until the
|
|
20
|
+
// final line arrives. It MAY send control lines (cancel,
|
|
21
|
+
// toolResult) while a request is in flight: stdin is read
|
|
22
|
+
// on its own thread, so those are seen mid-generation.
|
|
23
23
|
// helper -> one request on stdin, one response on stdout
|
|
24
24
|
// ("stream" collapses to a single final response here,
|
|
25
|
-
// since one-shot stdout is not incremental
|
|
25
|
+
// since one-shot stdout is not incremental; function
|
|
26
|
+
// tools need --serve, since stdin is already consumed).
|
|
26
27
|
//
|
|
27
28
|
// request: {"op"?:"generate"|"stream"|"countTokens"|"prewarm"|"history"|"reset",
|
|
28
29
|
// // default generate
|
|
30
|
+
// "id"?:string, // target for "cancel"
|
|
29
31
|
// "instructions":string,"prompt":string,"schema"?:object|null,
|
|
30
32
|
// "temperature"?:number,"maxTokens"?:number,
|
|
31
33
|
// "includeSchemaInPrompt"?:bool,"reuseSession"?:bool,
|
|
32
34
|
// "sessionId"?:string, // multi-turn conversation
|
|
35
|
+
// "history"?:[{"role":"user"|"assistant","content":string,
|
|
36
|
+
// "toolCalls"?:[{id,name,arguments}]}
|
|
37
|
+
// |{"role":"tool","toolCallId":string,"name":string,
|
|
38
|
+
// "content":string}], // stateless multi-turn
|
|
39
|
+
// "trimHistory"?:bool, // drop oldest turns to fit
|
|
33
40
|
// "tools"?:["ocr"|"barcode"|"spotlight"], // built-in FM tools, 27+
|
|
41
|
+
// "functions"?:[{"name":string,"description":string,
|
|
42
|
+
// "parameters":object}], // tools run by the client
|
|
34
43
|
// "images"?:[string|{"path":string,"label"?:string}],
|
|
35
44
|
// // file paths, macOS 27+
|
|
36
45
|
// "useCase"?:"general"|"contentTagging",
|
|
@@ -38,10 +47,19 @@
|
|
|
38
47
|
// "sampling"?:{"mode":"greedy"}
|
|
39
48
|
// | {"mode":"topK","k":int,"seed"?:int}
|
|
40
49
|
// | {"mode":"threshold","p":number,"seed"?:int}}
|
|
41
|
-
//
|
|
42
|
-
// | {"
|
|
43
|
-
//
|
|
44
|
-
//
|
|
50
|
+
// control: {"op":"cancel","id"?:string} // no reply line of its own
|
|
51
|
+
// | {"op":"toolResult","callId":string,"output"?:string,
|
|
52
|
+
// "isError"?:bool,"stop"?:bool} // answers a toolCall event
|
|
53
|
+
// event: {"ok":true,"delta":string,"done":false} // stream, text
|
|
54
|
+
// | {"ok":true,"partial":string,"done":false} // stream, schema:
|
|
55
|
+
// // JSON so far
|
|
56
|
+
// | {"ok":true,"toolCall":{"id","name","arguments"},"done":false}
|
|
57
|
+
// final: {"ok":true,"content":string, // generate / stream
|
|
58
|
+
// "done"?:true,"finishReason"?:"stop"|"length"|"toolCalls",
|
|
59
|
+
// "usage"?:{inputTokens,outputTokens,cachedInputTokens},
|
|
60
|
+
// "toolCalls"?:[{id,name,arguments,output?}],
|
|
61
|
+
// "trimmedTurns"?:int}
|
|
62
|
+
// | {"ok":true,"tokens":int,"contextSize":int} // countTokens
|
|
45
63
|
// | {"ok":true,"prewarmed":true} // prewarm
|
|
46
64
|
// | {"ok":true,"history":[{role,content}], // history
|
|
47
65
|
// "instructions":string,"sessionId":string}
|
|
@@ -49,17 +67,23 @@
|
|
|
49
67
|
// | {"ok":false,"error":string,"kind"?:string}
|
|
50
68
|
//
|
|
51
69
|
// `kind` is one of availability | schema | context | quota | guardrail |
|
|
52
|
-
// timeout | unsupported | generation, so callers can raise
|
|
53
|
-
// instead of matching on strings. A quota error may carry
|
|
70
|
+
// timeout | unsupported | tool | cancelled | busy | generation, so callers can raise
|
|
71
|
+
// typed errors instead of matching on strings. A quota error may carry
|
|
72
|
+
// `resetDate`; a context error may carry `contextSize` and `tokenCount`.
|
|
73
|
+
//
|
|
74
|
+
// A function tool call is answered by the client with a toolResult line. A
|
|
75
|
+
// reply with "stop":true ends generation there with finishReason "toolCalls"
|
|
76
|
+
// and empty content — the client executes the call itself and continues the
|
|
77
|
+
// conversation later through `history`, which is how an OpenAI-style caller
|
|
78
|
+
// (or the Vercel AI SDK) drives tools.
|
|
54
79
|
//
|
|
55
80
|
// --probe reports availability, contextSize, variant, the model's
|
|
56
|
-
// capabilities (vision / guidedGeneration / reasoning / toolCalling) and
|
|
57
|
-
//
|
|
58
|
-
// `com.apple.developer.private-cloud-compute`
|
|
59
|
-
// *inference*, so
|
|
60
|
-
// to an unentitled process. It also reports `features`
|
|
61
|
-
//
|
|
62
|
-
// instead of a stale binary's silent misbehaviour.
|
|
81
|
+
// capabilities (vision / guidedGeneration / reasoning / toolCalling) and,
|
|
82
|
+
// under "cloud", the same for Private Cloud Compute plus its quota. Those are
|
|
83
|
+
// readable without the `com.apple.developer.private-cloud-compute`
|
|
84
|
+
// entitlement that blocks PCC *inference*, so they are the first-party cloud
|
|
85
|
+
// state available to an unentitled process. It also reports `features` so callers can fail
|
|
86
|
+
// fast with a message instead of a stale binary's silent misbehaviour.
|
|
63
87
|
//
|
|
64
88
|
|
|
65
89
|
import Foundation
|
|
@@ -71,36 +95,225 @@ import _Vision_FoundationModels
|
|
|
71
95
|
import _CoreSpotlight_FoundationModels
|
|
72
96
|
#endif
|
|
73
97
|
|
|
98
|
+
/// Serialises writes. Function tools can run concurrently when the model asks
|
|
99
|
+
/// for several at once, and two unsynchronised `print`s can interleave bytes
|
|
100
|
+
/// inside a line, which would corrupt the framing for every later request.
|
|
101
|
+
private let emitLock = NSLock()
|
|
102
|
+
|
|
74
103
|
private func emit(_ object: [String: Any]) {
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
return
|
|
104
|
+
var text = "{\"ok\":false,\"error\":\"helper could not encode its response\"}"
|
|
105
|
+
if let data = try? JSONSerialization.data(withJSONObject: object),
|
|
106
|
+
let encoded = String(data: data, encoding: .utf8) {
|
|
107
|
+
text = encoded
|
|
80
108
|
}
|
|
109
|
+
emitLock.lock()
|
|
81
110
|
// stdout is a pipe here, so it is fully buffered; the parent is waiting on
|
|
82
111
|
// this line and would otherwise see nothing until the process exits.
|
|
83
112
|
print(text)
|
|
84
113
|
fflush(stdout)
|
|
114
|
+
emitLock.unlock()
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
private func jsonValue(_ text: String) -> Any? {
|
|
118
|
+
guard let data = text.data(using: .utf8) else { return nil }
|
|
119
|
+
return try? JSONSerialization.jsonObject(with: data, options: [.fragmentsAllowed])
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
private func jsonText(_ value: Any) -> String {
|
|
123
|
+
guard let data = try? JSONSerialization.data(withJSONObject: value, options: [.fragmentsAllowed]),
|
|
124
|
+
let text = String(data: data, encoding: .utf8) else { return "{}" }
|
|
125
|
+
return text
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/// A reply to a function tool call, sent by the parent as a toolResult line
|
|
129
|
+
/// while the request is still in flight.
|
|
130
|
+
private enum ToolReply {
|
|
131
|
+
case output(String)
|
|
132
|
+
case failure(String)
|
|
133
|
+
case stop
|
|
134
|
+
case cancelled
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/// Routes the control lines that have to be seen *while* a request runs —
|
|
138
|
+
/// cancellation and tool results. Everything else is a request and waits its
|
|
139
|
+
/// turn in the serial queue.
|
|
140
|
+
///
|
|
141
|
+
/// Why a reader thread rather than reading stdin between requests: a tool call
|
|
142
|
+
/// suspends generation until the parent answers, and the answer arrives on
|
|
143
|
+
/// stdin. Reading stdin only between requests would deadlock the first tool
|
|
144
|
+
/// call; it would also make cancellation impossible, since the cancel line
|
|
145
|
+
/// would sit unread until the generation it was meant to stop had finished.
|
|
146
|
+
private final class Control: @unchecked Sendable {
|
|
147
|
+
static let shared = Control()
|
|
148
|
+
|
|
149
|
+
private let lock = NSLock()
|
|
150
|
+
private var currentId: String?
|
|
151
|
+
private var currentTask: Task<Void, Never>?
|
|
152
|
+
/// Cancels that arrived before their request started. Bounded below.
|
|
153
|
+
private var cancelledEarly: Set<String> = []
|
|
154
|
+
private var waiting: [String: CheckedContinuation<ToolReply, Never>] = [:]
|
|
155
|
+
private var closed = false
|
|
156
|
+
|
|
157
|
+
/// True when `line` was a control message and has been handled here.
|
|
158
|
+
func intercept(_ line: String) -> Bool {
|
|
159
|
+
// Cheap pre-check: a prompt can be tens of KB, and most lines are
|
|
160
|
+
// requests, which never carry these two ops.
|
|
161
|
+
guard line.contains("\"cancel\"") || line.contains("\"toolResult\""),
|
|
162
|
+
let object = jsonValue(line) as? [String: Any],
|
|
163
|
+
let op = object["op"] as? String else { return false }
|
|
164
|
+
switch op {
|
|
165
|
+
case "cancel":
|
|
166
|
+
cancel(id: object["id"] as? String)
|
|
167
|
+
return true
|
|
168
|
+
case "toolResult":
|
|
169
|
+
guard let callId = object["callId"] as? String else { return true }
|
|
170
|
+
let reply: ToolReply
|
|
171
|
+
if object["stop"] as? Bool == true {
|
|
172
|
+
reply = .stop
|
|
173
|
+
} else if object["isError"] as? Bool == true {
|
|
174
|
+
reply = .failure(object["output"] as? String ?? "the tool failed")
|
|
175
|
+
} else {
|
|
176
|
+
reply = .output(object["output"] as? String ?? "")
|
|
177
|
+
}
|
|
178
|
+
deliver(callId: callId, reply: reply)
|
|
179
|
+
return true
|
|
180
|
+
default:
|
|
181
|
+
return false
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
private func cancel(id: String?) {
|
|
186
|
+
lock.lock()
|
|
187
|
+
defer { lock.unlock() }
|
|
188
|
+
if id == nil || id == currentId {
|
|
189
|
+
currentTask?.cancel()
|
|
190
|
+
// A tool call parked on the parent would otherwise never notice.
|
|
191
|
+
for continuation in waiting.values { continuation.resume(returning: .cancelled) }
|
|
192
|
+
waiting.removeAll()
|
|
193
|
+
} else if let id {
|
|
194
|
+
// The request line may already be queued but not yet started.
|
|
195
|
+
if cancelledEarly.count > 256 { cancelledEarly.removeAll() }
|
|
196
|
+
cancelledEarly.insert(id)
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
private func deliver(callId: String, reply: ToolReply) {
|
|
201
|
+
lock.lock()
|
|
202
|
+
let continuation = waiting.removeValue(forKey: callId)
|
|
203
|
+
lock.unlock()
|
|
204
|
+
continuation?.resume(returning: reply)
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/// Park a tool call until the parent answers it. `announce` emits the
|
|
208
|
+
/// toolCall event, and runs only once the call is registered, so an answer
|
|
209
|
+
/// cannot arrive before anything is waiting for it.
|
|
210
|
+
func awaitToolResult(callId: String, announce: () -> Void) async -> ToolReply {
|
|
211
|
+
await withCheckedContinuation { (continuation: CheckedContinuation<ToolReply, Never>) in
|
|
212
|
+
lock.lock()
|
|
213
|
+
if closed || currentTask?.isCancelled == true {
|
|
214
|
+
lock.unlock()
|
|
215
|
+
continuation.resume(returning: .cancelled)
|
|
216
|
+
return
|
|
217
|
+
}
|
|
218
|
+
waiting[callId] = continuation
|
|
219
|
+
lock.unlock()
|
|
220
|
+
announce()
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/// Run one request as a cancellable task, recording it as the target of
|
|
225
|
+
/// any cancel line that arrives meanwhile.
|
|
226
|
+
func run(id: String?, _ body: @escaping () async -> Void) async {
|
|
227
|
+
let task: Task<Void, Never>? = lock.withLock {
|
|
228
|
+
if let id, cancelledEarly.remove(id) != nil { return nil }
|
|
229
|
+
let task = Task { await body() }
|
|
230
|
+
currentId = id
|
|
231
|
+
currentTask = task
|
|
232
|
+
return task
|
|
233
|
+
}
|
|
234
|
+
guard let task else {
|
|
235
|
+
emit(["ok": false, "kind": "cancelled", "error": "the request was cancelled"])
|
|
236
|
+
return
|
|
237
|
+
}
|
|
238
|
+
await task.value
|
|
239
|
+
lock.withLock {
|
|
240
|
+
currentId = nil
|
|
241
|
+
currentTask = nil
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
/// The parent closed stdin: nothing will ever answer a parked tool call.
|
|
246
|
+
func stdinClosed() {
|
|
247
|
+
lock.lock()
|
|
248
|
+
closed = true
|
|
249
|
+
currentTask?.cancel()
|
|
250
|
+
for continuation in waiting.values { continuation.resume(returning: .cancelled) }
|
|
251
|
+
waiting.removeAll()
|
|
252
|
+
lock.unlock()
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
/// Why a function tool call did not produce output.
|
|
257
|
+
private enum ClientToolFailure: Error, CustomStringConvertible {
|
|
258
|
+
case failed(name: String, message: String)
|
|
259
|
+
/// The client will run the call itself; end generation here.
|
|
260
|
+
case stop
|
|
261
|
+
|
|
262
|
+
var description: String {
|
|
263
|
+
switch self {
|
|
264
|
+
case .failed(let name, let message): return "tool \"\(name)\" failed: \(message)"
|
|
265
|
+
case .stop: return "stopped for a client tool call"
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
/// A tool defined by the client and executed there. The model's arguments are
|
|
271
|
+
/// sent up as a toolCall event and the call suspends until a toolResult line
|
|
272
|
+
/// answers it. `parameters` is the client's JSON Schema, already rewritten into
|
|
273
|
+
/// Apple's dialect, so the arguments are decoded under the same constrained
|
|
274
|
+
/// decoding guarantee as `json()`.
|
|
275
|
+
@available(macOS 26.0, *)
|
|
276
|
+
private struct ClientTool: Tool {
|
|
277
|
+
typealias Arguments = GeneratedContent
|
|
278
|
+
typealias Output = String
|
|
279
|
+
|
|
280
|
+
let name: String
|
|
281
|
+
let description: String
|
|
282
|
+
let parameters: GenerationSchema
|
|
283
|
+
|
|
284
|
+
func call(arguments: GeneratedContent) async throws -> String {
|
|
285
|
+
let callId = "call_" + UUID().uuidString.replacingOccurrences(of: "-", with: "").lowercased().prefix(24)
|
|
286
|
+
let args = jsonValue(arguments.jsonString) ?? [String: Any]()
|
|
287
|
+
let reply = await Control.shared.awaitToolResult(callId: callId) {
|
|
288
|
+
emit(["ok": true, "done": false,
|
|
289
|
+
"toolCall": ["id": callId, "name": name, "arguments": args]])
|
|
290
|
+
}
|
|
291
|
+
switch reply {
|
|
292
|
+
case .output(let text): return text
|
|
293
|
+
case .failure(let message): throw ClientToolFailure.failed(name: name, message: message)
|
|
294
|
+
case .stop: throw ClientToolFailure.stop
|
|
295
|
+
case .cancelled: throw CancellationError()
|
|
296
|
+
}
|
|
297
|
+
}
|
|
85
298
|
}
|
|
86
299
|
|
|
87
300
|
/// Optional session reuse, off by default.
|
|
88
301
|
///
|
|
89
|
-
///
|
|
90
|
-
/// the finding that allocating a fresh LanguageModelSession per
|
|
91
|
-
/// every other call stall ~16s while assets cycled — a strict
|
|
92
|
-
/// alternation, uncorrelated with prompt size.
|
|
302
|
+
/// An earlier prototype kept one session alive while the instructions were
|
|
303
|
+
/// unchanged, on the finding that allocating a fresh LanguageModelSession per
|
|
304
|
+
/// request made every other call stall ~16s while assets cycled — a strict
|
|
305
|
+
/// 17s/1.5s alternation, uncorrelated with prompt size.
|
|
93
306
|
///
|
|
94
307
|
/// That did not reproduce here. Measured on macOS 27 / M-series inside one
|
|
95
308
|
/// long-lived process, 6 calls per arm: session reused -> median 0.63s; a fresh
|
|
96
309
|
/// session every call -> median 0.64s, max 0.68s, no variance. The alternation
|
|
97
310
|
/// would have hit three times in six calls. The load-bearing part of the fix
|
|
98
|
-
/// appears to be the long-lived *process* (
|
|
99
|
-
///
|
|
311
|
+
/// appears to be the long-lived *process* (~17s a call was measured when a
|
|
312
|
+
/// helper was spawned per request, against ~1.5s once resident), not the
|
|
100
313
|
/// shared session.
|
|
101
314
|
///
|
|
102
315
|
/// So the default here is a fresh session per request, because a library cannot
|
|
103
|
-
/// let two unrelated calls share a transcript —
|
|
316
|
+
/// let two unrelated calls share a transcript — the prototype could, since every
|
|
104
317
|
/// one of its prompts carried its own whole context. The old behaviour is kept
|
|
105
318
|
/// behind `reuseSession` rather than deleted: the original finding was measured
|
|
106
319
|
/// too, possibly on macOS 26, and the doubt is worth preserving. If per-call
|
|
@@ -109,10 +322,10 @@ private func emit(_ object: [String: Any]) {
|
|
|
109
322
|
/// Named sessions (`sessionId`) are the multi-turn counterpart: same process,
|
|
110
323
|
/// but the session is keyed by id so a conversation accumulates a native
|
|
111
324
|
/// transcript across calls. History is also mirrored to
|
|
112
|
-
/// `~/Library/Caches/apple-llm/sessions/<id>.json
|
|
113
|
-
///
|
|
114
|
-
///
|
|
115
|
-
///
|
|
325
|
+
/// `~/Library/Caches/apple-llm/sessions/<id>.json`. When the native session
|
|
326
|
+
/// has to be rebuilt — a helper restart, changed instructions or tools, or a
|
|
327
|
+
/// context overflow — it is rebuilt as a native transcript from that mirror,
|
|
328
|
+
/// oldest turns dropped until it fits, so a conversation survives all three.
|
|
116
329
|
@available(macOS 26.0, *)
|
|
117
330
|
private final class SessionHolder {
|
|
118
331
|
static var session: LanguageModelSession?
|
|
@@ -178,10 +391,22 @@ private final class SessionStore {
|
|
|
178
391
|
}
|
|
179
392
|
}
|
|
180
393
|
|
|
394
|
+
/// Decoded schemas, keyed by their JSON text. A changing GenerationSchema costs
|
|
395
|
+
/// only ~0.15s per call, so per-request schemas are fine; this just makes a
|
|
396
|
+
/// repeated one free. Bounded, since a long-lived server may see many.
|
|
181
397
|
@available(macOS 26.0, *)
|
|
182
|
-
private
|
|
183
|
-
|
|
184
|
-
|
|
398
|
+
private final class SchemaCache {
|
|
399
|
+
static var decoded: [String: GenerationSchema] = [:]
|
|
400
|
+
|
|
401
|
+
static func decode(_ object: Any) throws -> GenerationSchema {
|
|
402
|
+
let data = try JSONSerialization.data(withJSONObject: object, options: [.sortedKeys])
|
|
403
|
+
let key = String(data: data, encoding: .utf8) ?? ""
|
|
404
|
+
if let cached = decoded[key] { return cached }
|
|
405
|
+
let schema = try JSONDecoder().decode(GenerationSchema.self, from: data)
|
|
406
|
+
if decoded.count >= 64 { decoded.removeAll() }
|
|
407
|
+
decoded[key] = schema
|
|
408
|
+
return schema
|
|
409
|
+
}
|
|
185
410
|
}
|
|
186
411
|
|
|
187
412
|
@available(macOS 26.0, *)
|
|
@@ -208,49 +433,145 @@ private func sessionFor(
|
|
|
208
433
|
return session
|
|
209
434
|
}
|
|
210
435
|
|
|
436
|
+
/// Turn history turns into native transcript entries.
|
|
437
|
+
///
|
|
438
|
+
/// A native transcript rather than a "Previous conversation:" preamble: the
|
|
439
|
+
/// model was trained on its own turn structure, and a preamble also costs the
|
|
440
|
+
/// history twice when a tool call echoes the prompt back.
|
|
441
|
+
@available(macOS 26.0, *)
|
|
442
|
+
private func transcriptEntries(from history: [[String: Any]]) -> [Transcript.Entry] {
|
|
443
|
+
var entries: [Transcript.Entry] = []
|
|
444
|
+
for turn in history {
|
|
445
|
+
let role = turn["role"] as? String ?? ""
|
|
446
|
+
let content = turn["content"] as? String ?? ""
|
|
447
|
+
switch role {
|
|
448
|
+
case "user":
|
|
449
|
+
entries.append(.prompt(Transcript.Prompt(
|
|
450
|
+
segments: [.text(Transcript.TextSegment(content: content))])))
|
|
451
|
+
case "assistant":
|
|
452
|
+
if !content.isEmpty {
|
|
453
|
+
entries.append(.response(Transcript.Response(
|
|
454
|
+
assetIDs: [], segments: [.text(Transcript.TextSegment(content: content))])))
|
|
455
|
+
}
|
|
456
|
+
let calls = (turn["toolCalls"] as? [[String: Any]] ?? []).compactMap {
|
|
457
|
+
call -> Transcript.ToolCall? in
|
|
458
|
+
guard let name = call["name"] as? String,
|
|
459
|
+
let arguments = try? GeneratedContent(json: jsonText(call["arguments"] ?? [String: Any]()))
|
|
460
|
+
else { return nil }
|
|
461
|
+
return Transcript.ToolCall(id: call["id"] as? String ?? UUID().uuidString,
|
|
462
|
+
toolName: name, arguments: arguments)
|
|
463
|
+
}
|
|
464
|
+
if !calls.isEmpty { entries.append(.toolCalls(Transcript.ToolCalls(calls))) }
|
|
465
|
+
case "tool":
|
|
466
|
+
entries.append(.toolOutput(Transcript.ToolOutput(
|
|
467
|
+
id: turn["toolCallId"] as? String ?? UUID().uuidString,
|
|
468
|
+
toolName: turn["name"] as? String ?? "tool",
|
|
469
|
+
segments: [.text(Transcript.TextSegment(content: content))])))
|
|
470
|
+
default:
|
|
471
|
+
continue
|
|
472
|
+
}
|
|
473
|
+
}
|
|
474
|
+
return entries
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
/// Drop the oldest turns until instructions, tools, history, prompt and the
|
|
478
|
+
/// response budget all fit in the context window. A turn is dropped whole —
|
|
479
|
+
/// from one user prompt up to the next — so a tool call is never left without
|
|
480
|
+
/// its output. Returns how many user turns were dropped, which the caller
|
|
481
|
+
/// reports: trimming is never silent.
|
|
482
|
+
///
|
|
483
|
+
/// Runs only after a context overflow (or when rebuilding a named session),
|
|
484
|
+
/// never ahead of every call, and binary-searches the cut: log2(turns) token
|
|
485
|
+
/// counts rather than one per dropped turn.
|
|
486
|
+
///
|
|
487
|
+
/// macOS 27 only, since it needs `tokenCount`; on 26 the history is left as is
|
|
488
|
+
/// and an overflow surfaces as a ContextLengthError.
|
|
489
|
+
@available(macOS 26.0, *)
|
|
490
|
+
private func fitted(
|
|
491
|
+
_ entries: [Transcript.Entry], model: SystemLanguageModel, instructions: String,
|
|
492
|
+
tools: [any Tool], prompt: Prompt, reserve: Int
|
|
493
|
+
) async -> (entries: [Transcript.Entry], dropped: Int) {
|
|
494
|
+
guard #available(macOS 27.0, *), !entries.isEmpty else { return (entries, 0) }
|
|
495
|
+
var fixed = (try? await model.tokenCount(for: prompt)) ?? 0
|
|
496
|
+
if !instructions.isEmpty {
|
|
497
|
+
fixed += (try? await model.tokenCount(for: Instructions(instructions))) ?? 0
|
|
498
|
+
}
|
|
499
|
+
if !tools.isEmpty { fixed += (try? await model.tokenCount(for: tools)) ?? 0 }
|
|
500
|
+
let budget = model.contextSize - reserve - fixed
|
|
501
|
+
func fits(_ slice: ArraySlice<Transcript.Entry>) async -> Bool {
|
|
502
|
+
if slice.isEmpty { return true }
|
|
503
|
+
let used = (try? await model.tokenCount(for: Array(slice))) ?? Int.max
|
|
504
|
+
return used <= budget
|
|
505
|
+
}
|
|
506
|
+
if await fits(entries[...]) { return (entries, 0) }
|
|
507
|
+
// Cut points: the start of each user turn, plus "drop everything".
|
|
508
|
+
var cuts = entries.indices.filter { $0 > 0 && isPrompt(entries[$0]) }
|
|
509
|
+
cuts.append(entries.count)
|
|
510
|
+
var low = 0
|
|
511
|
+
var high = cuts.count - 1
|
|
512
|
+
while low < high {
|
|
513
|
+
let mid = (low + high) / 2
|
|
514
|
+
if await fits(entries[cuts[mid]...]) { high = mid } else { low = mid + 1 }
|
|
515
|
+
}
|
|
516
|
+
return (Array(entries[cuts[low]...]), low + 1)
|
|
517
|
+
}
|
|
518
|
+
|
|
519
|
+
@available(macOS 26.0, *)
|
|
520
|
+
private func isPrompt(_ entry: Transcript.Entry) -> Bool {
|
|
521
|
+
if case .prompt = entry { return true }
|
|
522
|
+
return false
|
|
523
|
+
}
|
|
524
|
+
|
|
525
|
+
@available(macOS 26.0, *)
|
|
526
|
+
private func transcriptSession(
|
|
527
|
+
model: SystemLanguageModel, instructions: String, tools: [any Tool],
|
|
528
|
+
entries: [Transcript.Entry]
|
|
529
|
+
) -> LanguageModelSession {
|
|
530
|
+
if entries.isEmpty {
|
|
531
|
+
return tools.isEmpty
|
|
532
|
+
? LanguageModelSession(model: model, instructions: instructions)
|
|
533
|
+
: LanguageModelSession(model: model, tools: tools, instructions: instructions)
|
|
534
|
+
}
|
|
535
|
+
var all: [Transcript.Entry] = []
|
|
536
|
+
if !instructions.isEmpty || !tools.isEmpty {
|
|
537
|
+
let segments: [Transcript.Segment] =
|
|
538
|
+
instructions.isEmpty ? [] : [.text(Transcript.TextSegment(content: instructions))]
|
|
539
|
+
all.append(.instructions(Transcript.Instructions(
|
|
540
|
+
segments: segments,
|
|
541
|
+
toolDefinitions: tools.map { Transcript.ToolDefinition(tool: $0) })))
|
|
542
|
+
}
|
|
543
|
+
all.append(contentsOf: entries)
|
|
544
|
+
return LanguageModelSession(model: model, tools: tools, transcript: Transcript(entries: all))
|
|
545
|
+
}
|
|
546
|
+
|
|
211
547
|
/// Named multi-turn session. Tools are fixed at session construction, so a
|
|
212
|
-
/// changed tool set recreates the native session
|
|
213
|
-
///
|
|
548
|
+
/// changed tool set recreates the native session. Recreation (first use in
|
|
549
|
+
/// this process, changed parameters, or `rebuild` after a context overflow)
|
|
550
|
+
/// replays the mirrored history as a native transcript, trimmed to fit, so
|
|
551
|
+
/// nothing the caller said is silently dropped from `history` and the model
|
|
552
|
+
/// keeps as much of the conversation as the window allows.
|
|
214
553
|
@available(macOS 26.0, *)
|
|
215
554
|
private func namedSessionFor(
|
|
216
|
-
id: String, model: SystemLanguageModel, instructions: String,
|
|
217
|
-
|
|
218
|
-
) -> LanguageModelSession {
|
|
219
|
-
let
|
|
220
|
-
|
|
221
|
-
tools: toolNames, instructions: instructions)
|
|
222
|
-
if let existing = SessionStore.named[id], existing.fingerprint == want {
|
|
223
|
-
return existing.session
|
|
224
|
-
}
|
|
225
|
-
// Recreate (first use, param change, or restart). Restore mirrored history
|
|
226
|
-
// so `history` is continuous even though the native transcript restarts.
|
|
227
|
-
let stored = SessionStore.loadHistory(id)
|
|
228
|
-
let history = SessionStore.named[id]?.history ?? stored.history
|
|
229
|
-
let session: LanguageModelSession
|
|
230
|
-
if tools.isEmpty {
|
|
231
|
-
session = LanguageModelSession(model: model, instructions: instructions)
|
|
232
|
-
} else {
|
|
233
|
-
session = LanguageModelSession(model: model, tools: tools, instructions: instructions)
|
|
555
|
+
id: String, model: SystemLanguageModel, instructions: String, fingerprint want: String,
|
|
556
|
+
tools: [any Tool], prompt: Prompt, reserve: Int, rebuild: Bool
|
|
557
|
+
) async -> (session: LanguageModelSession, dropped: Int) {
|
|
558
|
+
if !rebuild, let existing = SessionStore.named[id], existing.fingerprint == want {
|
|
559
|
+
return (existing.session, 0)
|
|
234
560
|
}
|
|
561
|
+
let history = SessionStore.named[id]?.history ?? SessionStore.loadHistory(id).history
|
|
562
|
+
let replay = transcriptEntries(from: history.suffix(40).map { $0 as [String: Any] })
|
|
563
|
+
let (kept, dropped) = await fitted(replay, model: model, instructions: instructions,
|
|
564
|
+
tools: tools, prompt: prompt, reserve: reserve)
|
|
565
|
+
let session = transcriptSession(model: model, instructions: instructions, tools: tools, entries: kept)
|
|
235
566
|
SessionStore.named[id] = NamedSession(
|
|
236
567
|
session: session, instructions: instructions,
|
|
237
568
|
fingerprint: want, history: history)
|
|
238
569
|
// Persist immediately so a fresh id is visible to `history` even before
|
|
239
570
|
// its first turn completes.
|
|
240
571
|
SessionStore.saveHistory(id: id, instructions: instructions, history: history)
|
|
241
|
-
return session
|
|
572
|
+
return (session, dropped)
|
|
242
573
|
}
|
|
243
574
|
|
|
244
|
-
// The tool names for the in-flight request, stashed so namedSessionFor can
|
|
245
|
-
// fingerprint on them without threading another parameter through sessionFor.
|
|
246
|
-
@available(macOS 26.0, *)
|
|
247
|
-
private final class RequestContext {
|
|
248
|
-
static var toolNames: [String] = []
|
|
249
|
-
}
|
|
250
|
-
|
|
251
|
-
@available(macOS 26.0, *)
|
|
252
|
-
private func currentToolNames() -> [String] { RequestContext.toolNames }
|
|
253
|
-
|
|
254
575
|
/// Classify a generation failure so callers can raise a typed error rather than
|
|
255
576
|
/// matching on a message Apple may reword in any OS release.
|
|
256
577
|
///
|
|
@@ -259,23 +580,41 @@ private func currentToolNames() -> [String] { RequestContext.toolNames }
|
|
|
259
580
|
/// `rateLimited` carries a `resetDate`, which is what makes an on-device quota
|
|
260
581
|
/// error actionable rather than just a failure.
|
|
261
582
|
@available(macOS 26.0, *)
|
|
262
|
-
private func classify(_ error: Error) -> (kind: String,
|
|
583
|
+
private func classify(_ error: Error) -> (kind: String, extra: [String: Any]) {
|
|
584
|
+
if error is CancellationError { return ("cancelled", [:]) }
|
|
585
|
+
// Several processes using the model at once: the model manager turns a
|
|
586
|
+
// request away with ModelManagerError 1042, wrapped in an otherwise
|
|
587
|
+
// uninformative LanguageModelError -1. Nothing was generated, so the
|
|
588
|
+
// client can retry it; naming it lets the client do that safely.
|
|
589
|
+
let described = String(describing: error)
|
|
590
|
+
if described.contains("ModelManagerError"), described.contains("1042") { return ("busy", [:]) }
|
|
591
|
+
if let toolError = error as? LanguageModelSession.ToolCallError {
|
|
592
|
+
if toolError.underlyingError is CancellationError { return ("cancelled", [:]) }
|
|
593
|
+
if let client = toolError.underlyingError as? ClientToolFailure, case .stop = client {
|
|
594
|
+
return ("stopped", [:])
|
|
595
|
+
}
|
|
596
|
+
return ("tool", ["tool": toolError.tool.name])
|
|
597
|
+
}
|
|
263
598
|
if #available(macOS 27.0, *) {
|
|
264
599
|
if let modern = error as? LanguageModelError {
|
|
265
600
|
switch modern {
|
|
266
|
-
case .contextSizeExceeded
|
|
601
|
+
case .contextSizeExceeded(let info):
|
|
602
|
+
return ("context", ["contextSize": info.contextSize, "tokenCount": info.tokenCount])
|
|
267
603
|
case .rateLimited(let info):
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
604
|
+
if let reset = info.resetDate {
|
|
605
|
+
return ("quota", ["resetDate": ISO8601DateFormatter().string(from: reset)])
|
|
606
|
+
}
|
|
607
|
+
return ("quota", [:])
|
|
608
|
+
case .guardrailViolation, .refusal: return ("guardrail", [:])
|
|
609
|
+
case .timeout: return ("timeout", [:])
|
|
271
610
|
case .unsupportedCapability, .unsupportedGenerationGuide,
|
|
272
611
|
.unsupportedLanguageOrLocale, .unsupportedTranscriptContent:
|
|
273
|
-
return ("unsupported",
|
|
274
|
-
@unknown default: return ("generation",
|
|
612
|
+
return ("unsupported", [:])
|
|
613
|
+
@unknown default: return ("generation", [:])
|
|
275
614
|
}
|
|
276
615
|
}
|
|
277
616
|
}
|
|
278
|
-
return (legacyKind(error),
|
|
617
|
+
return (legacyKind(error), [:])
|
|
279
618
|
}
|
|
280
619
|
|
|
281
620
|
/// macOS 26's error type, kept so one helper source serves both OS versions.
|
|
@@ -299,6 +638,15 @@ private func legacyKind(_ error: Error) -> String {
|
|
|
299
638
|
return "generation"
|
|
300
639
|
}
|
|
301
640
|
|
|
641
|
+
@available(macOS 26.0, *)
|
|
642
|
+
private func failure(_ error: Error) -> [String: Any] {
|
|
643
|
+
let (kind, extra) = classify(error)
|
|
644
|
+
var payload: [String: Any] = ["ok": false, "kind": kind, "error": "\(error)"]
|
|
645
|
+
if kind == "cancelled" { payload["error"] = "the request was cancelled" }
|
|
646
|
+
for (key, value) in extra { payload[key] = value }
|
|
647
|
+
return payload
|
|
648
|
+
}
|
|
649
|
+
|
|
302
650
|
/// Models are keyed by use case and guardrails, because both are fixed at
|
|
303
651
|
/// construction. Building one is cheap; keeping them avoids re-resolving assets
|
|
304
652
|
/// when a caller alternates between, say, general and contentTagging.
|
|
@@ -370,9 +718,37 @@ private func toolsFor(_ names: [String]) throws -> [any Tool] {
|
|
|
370
718
|
return out
|
|
371
719
|
}
|
|
372
720
|
|
|
721
|
+
/// Function tools from the request. Their parameter schemas go through Apple's
|
|
722
|
+
/// decoder like any response schema, and a rejected one is a schema error
|
|
723
|
+
/// naming the tool rather than a generic failure.
|
|
724
|
+
@available(macOS 26.0, *)
|
|
725
|
+
private func functionToolsFrom(_ value: Any?) throws -> [any Tool] {
|
|
726
|
+
guard let specs = value as? [[String: Any]] else { return [] }
|
|
727
|
+
var out: [any Tool] = []
|
|
728
|
+
for spec in specs {
|
|
729
|
+
guard let name = spec["name"] as? String, !name.isEmpty else {
|
|
730
|
+
throw ToolError.unknown("every function tool needs a name")
|
|
731
|
+
}
|
|
732
|
+
let parameters = spec["parameters"] ?? [
|
|
733
|
+
"type": "object", "title": "\(name)Arguments", "properties": [String: Any](),
|
|
734
|
+
"x-order": [String](), "required": [String](), "additionalProperties": false,
|
|
735
|
+
]
|
|
736
|
+
let schema: GenerationSchema
|
|
737
|
+
do {
|
|
738
|
+
schema = try SchemaCache.decode(parameters)
|
|
739
|
+
} catch {
|
|
740
|
+
throw ToolError.schema("Apple rejected the parameters schema of tool \"\(name)\": \(error)")
|
|
741
|
+
}
|
|
742
|
+
out.append(ClientTool(name: name, description: spec["description"] as? String ?? "",
|
|
743
|
+
parameters: schema))
|
|
744
|
+
}
|
|
745
|
+
return out
|
|
746
|
+
}
|
|
747
|
+
|
|
373
748
|
private enum ToolError: Error {
|
|
374
749
|
case unknown(String)
|
|
375
750
|
case unsupported(String)
|
|
751
|
+
case schema(String)
|
|
376
752
|
}
|
|
377
753
|
|
|
378
754
|
/// Coerce a JSON number regardless of int/double mismatch.
|
|
@@ -469,22 +845,67 @@ private func stringArray(_ value: Any?) -> [String] {
|
|
|
469
845
|
(value as? [Any] ?? []).compactMap { $0 as? String }
|
|
470
846
|
}
|
|
471
847
|
|
|
848
|
+
/// Every tool call made while producing one response, with its output where
|
|
849
|
+
/// the transcript recorded one — built-in and function tools alike, so a
|
|
850
|
+
/// caller can see *why* an answer says what it says.
|
|
851
|
+
@available(macOS 26.0, *)
|
|
852
|
+
private func toolActivity(_ entries: some Sequence<Transcript.Entry>) -> [[String: Any]] {
|
|
853
|
+
var calls: [[String: Any]] = []
|
|
854
|
+
var outputs: [String: String] = [:]
|
|
855
|
+
for entry in entries {
|
|
856
|
+
switch entry {
|
|
857
|
+
case .toolCalls(let batch):
|
|
858
|
+
for call in batch {
|
|
859
|
+
calls.append(["id": call.id, "name": call.toolName,
|
|
860
|
+
"arguments": jsonValue(call.arguments.jsonString) ?? NSNull()])
|
|
861
|
+
}
|
|
862
|
+
case .toolOutput(let output):
|
|
863
|
+
outputs[output.id] = output.segments.compactMap { segment -> String? in
|
|
864
|
+
if case .text(let text) = segment { return text.content }
|
|
865
|
+
return nil
|
|
866
|
+
}.joined()
|
|
867
|
+
default:
|
|
868
|
+
continue
|
|
869
|
+
}
|
|
870
|
+
}
|
|
871
|
+
return calls.map { call in
|
|
872
|
+
var out = call
|
|
873
|
+
if let id = call["id"] as? String, let text = outputs[id] { out["output"] = text }
|
|
874
|
+
return out
|
|
875
|
+
}
|
|
876
|
+
}
|
|
877
|
+
|
|
878
|
+
/// What one successful generation produced, before it is framed as a line.
|
|
879
|
+
private struct Outcome {
|
|
880
|
+
var content: String
|
|
881
|
+
var usage: [String: Any]?
|
|
882
|
+
var toolCalls: [[String: Any]] = []
|
|
883
|
+
var finishReason = "stop"
|
|
884
|
+
}
|
|
885
|
+
|
|
886
|
+
@available(macOS 27.0, *)
|
|
887
|
+
private func usageOf(_ usage: LanguageModelSession.Usage) -> [String: Any] {
|
|
888
|
+
["inputTokens": usage.input.totalTokenCount,
|
|
889
|
+
"cachedInputTokens": usage.input.cachedTokenCount,
|
|
890
|
+
"outputTokens": usage.output.totalTokenCount]
|
|
891
|
+
}
|
|
892
|
+
|
|
893
|
+
/// A response that used its whole token budget was cut off, not finished.
|
|
894
|
+
/// Reported so a caller can tell a short answer from a truncated one.
|
|
895
|
+
private func finishReason(usage: [String: Any]?, maxTokens: Int?) -> String {
|
|
896
|
+
if let maxTokens, let out = usage?["outputTokens"] as? Int, out >= maxTokens { return "length" }
|
|
897
|
+
return "stop"
|
|
898
|
+
}
|
|
899
|
+
|
|
472
900
|
/// One request/response cycle. Kept separate from the transport so that
|
|
473
901
|
/// one-shot and serve modes cannot drift apart.
|
|
474
902
|
@available(macOS 26.0, *)
|
|
475
|
-
private func handle(
|
|
476
|
-
envelope:
|
|
477
|
-
schemaCache: inout [String: GenerationSchema]
|
|
478
|
-
) async {
|
|
479
|
-
await handleEnvelope(envelope: envelope, schemaCache: &schemaCache, streaming: false)
|
|
903
|
+
private func handle(envelope: [String: Any]) async {
|
|
904
|
+
await handleEnvelope(envelope: envelope, streaming: false)
|
|
480
905
|
}
|
|
481
906
|
|
|
482
907
|
@available(macOS 26.0, *)
|
|
483
|
-
private func handleEnvelope(
|
|
484
|
-
envelope: [String: Any],
|
|
485
|
-
schemaCache: inout [String: GenerationSchema],
|
|
486
|
-
streaming: Bool
|
|
487
|
-
) async {
|
|
908
|
+
private func handleEnvelope(envelope: [String: Any], streaming: Bool) async {
|
|
488
909
|
let useCase = envelope["useCase"] as? String ?? "general"
|
|
489
910
|
let guardrails = envelope["guardrails"] as? String ?? "default"
|
|
490
911
|
let model = modelFor(useCase: useCase, guardrails: guardrails)
|
|
@@ -502,8 +923,8 @@ private func handleEnvelope(
|
|
|
502
923
|
let promptText = envelope["prompt"] as? String ?? ""
|
|
503
924
|
let imageSpecs = imageSpecsFrom(envelope["images"])
|
|
504
925
|
let toolNames = stringArray(envelope["tools"])
|
|
505
|
-
RequestContext.toolNames = toolNames
|
|
506
926
|
let sessionId = envelope["sessionId"] as? String
|
|
927
|
+
let historyTurns = envelope["history"] as? [[String: Any]] ?? []
|
|
507
928
|
|
|
508
929
|
if !imageSpecs.isEmpty {
|
|
509
930
|
if #available(macOS 27.0, *) {
|
|
@@ -526,24 +947,42 @@ private func handleEnvelope(
|
|
|
526
947
|
|
|
527
948
|
let tools: [any Tool]
|
|
528
949
|
do {
|
|
529
|
-
tools = try toolsFor(toolNames)
|
|
950
|
+
tools = try toolsFor(toolNames) + functionToolsFrom(envelope["functions"])
|
|
530
951
|
} catch let err as ToolError {
|
|
531
952
|
switch err {
|
|
532
953
|
case .unknown(let m), .unsupported(let m):
|
|
533
954
|
emit(["ok": false, "kind": "unsupported", "error": m])
|
|
534
|
-
|
|
955
|
+
case .schema(let m):
|
|
956
|
+
emit(["ok": false, "kind": "schema", "error": m])
|
|
535
957
|
}
|
|
958
|
+
return
|
|
536
959
|
} catch {
|
|
537
960
|
emit(["ok": false, "kind": "unsupported", "error": "\(error)"])
|
|
538
961
|
return
|
|
539
962
|
}
|
|
540
963
|
if !tools.isEmpty {
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
964
|
+
if #available(macOS 27.0, *) {
|
|
965
|
+
guard model.capabilities.contains(.toolCalling) else {
|
|
966
|
+
emit(["ok": false, "kind": "unsupported",
|
|
967
|
+
"error": "this model does not support tool calling"])
|
|
968
|
+
return
|
|
969
|
+
}
|
|
970
|
+
}
|
|
971
|
+
}
|
|
972
|
+
|
|
973
|
+
// A response schema, decoded once for every op that needs it.
|
|
974
|
+
var schema: GenerationSchema? = nil
|
|
975
|
+
if let schemaObject = envelope["schema"], !(schemaObject is NSNull) {
|
|
976
|
+
do {
|
|
977
|
+
schema = try SchemaCache.decode(schemaObject)
|
|
978
|
+
} catch {
|
|
979
|
+
emit(["ok": false, "kind": "schema",
|
|
980
|
+
"error": "Apple rejected the response schema: \(error)"])
|
|
544
981
|
return
|
|
545
982
|
}
|
|
546
983
|
}
|
|
984
|
+
let prompt = promptWith(text: promptText, imageSpecs: imageSpecs)
|
|
985
|
+
let history = transcriptEntries(from: historyTurns)
|
|
547
986
|
|
|
548
987
|
// Token counting and prewarming share the model but not the generation path.
|
|
549
988
|
// An empty op defaults to generate; an unknown op is an error rather than
|
|
@@ -561,19 +1000,25 @@ private func handleEnvelope(
|
|
|
561
1000
|
return
|
|
562
1001
|
}
|
|
563
1002
|
do {
|
|
564
|
-
// Instructions
|
|
565
|
-
// a caller budgeting against contextSize
|
|
566
|
-
|
|
567
|
-
|
|
1003
|
+
// Instructions, tools, schema and history all ride in the same
|
|
1004
|
+
// window as the prompt, so a caller budgeting against contextSize
|
|
1005
|
+
// needs every one of them counted.
|
|
1006
|
+
var total = try await model.tokenCount(for: prompt)
|
|
568
1007
|
if !instructions.isEmpty {
|
|
569
1008
|
total += try await model.tokenCount(for: Instructions(instructions))
|
|
570
1009
|
}
|
|
571
1010
|
if !tools.isEmpty {
|
|
572
1011
|
total += (try? await model.tokenCount(for: tools)) ?? 0
|
|
573
1012
|
}
|
|
1013
|
+
if let schema {
|
|
1014
|
+
total += (try? await model.tokenCount(for: schema)) ?? 0
|
|
1015
|
+
}
|
|
1016
|
+
if !history.isEmpty {
|
|
1017
|
+
total += (try? await model.tokenCount(for: history)) ?? 0
|
|
1018
|
+
}
|
|
574
1019
|
emit(["ok": true, "tokens": total, "contextSize": model.contextSize])
|
|
575
1020
|
} catch {
|
|
576
|
-
emit(
|
|
1021
|
+
emit(failure(error))
|
|
577
1022
|
}
|
|
578
1023
|
return
|
|
579
1024
|
|
|
@@ -624,117 +1069,117 @@ private func handleEnvelope(
|
|
|
624
1069
|
return
|
|
625
1070
|
}
|
|
626
1071
|
|
|
627
|
-
// Streaming + guided generation do not mix in v1: the schema stream yields
|
|
628
|
-
// GeneratedContent snapshots whose partials are not plain-text deltas.
|
|
629
|
-
// Reject loudly rather than emitting misleading partial JSON.
|
|
630
1072
|
let wantsStream = (op == "stream") || streaming
|
|
631
|
-
|
|
632
|
-
let schemaObject = envelope["schema"], !(schemaObject is NSNull) {
|
|
633
|
-
emit(["ok": false, "kind": "unsupported",
|
|
634
|
-
"error": "streaming with a schema is not supported; use generate for JSON"])
|
|
635
|
-
return
|
|
636
|
-
}
|
|
637
|
-
|
|
638
|
-
// No schema means a free-text call. api-scribe only ever asked for JSON and
|
|
639
|
-
// so required a schema on every request; `text()` needs the unconstrained
|
|
640
|
-
// `respond(to:options:)` overload instead.
|
|
641
|
-
var schema: GenerationSchema? = nil
|
|
642
|
-
if let schemaObject = envelope["schema"], !(schemaObject is NSNull) {
|
|
643
|
-
guard let schemaData = try? JSONSerialization.data(withJSONObject: schemaObject),
|
|
644
|
-
let schemaKey = String(data: schemaData, encoding: .utf8) else {
|
|
645
|
-
emit(["ok": false, "kind": "schema",
|
|
646
|
-
"error": "helper payload is missing a usable schema"])
|
|
647
|
-
return
|
|
648
|
-
}
|
|
649
|
-
// A changing GenerationSchema costs only ~0.15s per call, so per-request
|
|
650
|
-
// schemas are fine; this cache just makes a repeated one free.
|
|
651
|
-
if let cached = schemaCache[schemaKey] {
|
|
652
|
-
schema = cached
|
|
653
|
-
} else {
|
|
654
|
-
do {
|
|
655
|
-
let decoded = try JSONDecoder().decode(GenerationSchema.self, from: schemaData)
|
|
656
|
-
// Bound the cache: a long-lived server may see many schemas.
|
|
657
|
-
if schemaCache.count >= 64 { schemaCache.removeAll() }
|
|
658
|
-
schemaCache[schemaKey] = decoded
|
|
659
|
-
schema = decoded
|
|
660
|
-
} catch {
|
|
661
|
-
emit(["ok": false, "kind": "schema",
|
|
662
|
-
"error": "Apple rejected the response schema: \(error)"])
|
|
663
|
-
return
|
|
664
|
-
}
|
|
665
|
-
}
|
|
666
|
-
}
|
|
667
|
-
|
|
1073
|
+
let maxTokens = intNum(envelope["maxTokens"])
|
|
668
1074
|
let options = GenerationOptions(
|
|
669
1075
|
samplingMode: samplingFrom(envelope["sampling"]),
|
|
670
1076
|
temperature: num(envelope["temperature"]),
|
|
671
|
-
maximumResponseTokens:
|
|
1077
|
+
maximumResponseTokens: maxTokens
|
|
672
1078
|
)
|
|
673
1079
|
let reuse = envelope["reuseSession"] as? Bool ?? false
|
|
674
|
-
//
|
|
675
|
-
//
|
|
1080
|
+
// A caller that spells the schema out in its own system prompt could pass
|
|
1081
|
+
// false. A library cannot assume that, and a schema's `description`
|
|
676
1082
|
// fields are how the decoder gets its generation guidance, so default true.
|
|
677
1083
|
let includeSchema = envelope["includeSchemaInPrompt"] as? Bool ?? true
|
|
1084
|
+
// Room left for the response when trimming history to fit.
|
|
1085
|
+
let reserve = min(maxTokens ?? 1024, model.contextSize / 2)
|
|
1086
|
+
let fingerprint = [useCase, guardrails, toolNames.sorted().joined(separator: ","),
|
|
1087
|
+
jsonText(envelope["functions"] ?? [Any]()), instructions].joined(separator: "|")
|
|
678
1088
|
|
|
679
|
-
|
|
680
|
-
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
|
|
687
|
-
|
|
688
|
-
|
|
689
|
-
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
}
|
|
703
|
-
let prompt = promptWith(text: effectivePromptText, imageSpecs: imageSpecs)
|
|
704
|
-
|
|
705
|
-
if wantsStream {
|
|
706
|
-
await handleStream(session: session, prompt: prompt, options: options,
|
|
707
|
-
sessionId: sessionId, promptText: promptText)
|
|
708
|
-
return
|
|
1089
|
+
var trimmed = 0
|
|
1090
|
+
func makeSession(rebuild: Bool) async -> LanguageModelSession {
|
|
1091
|
+
if let sid = sessionId, !sid.isEmpty {
|
|
1092
|
+
let (named, dropped) = await namedSessionFor(
|
|
1093
|
+
id: sid, model: model, instructions: instructions, fingerprint: fingerprint,
|
|
1094
|
+
tools: tools, prompt: prompt, reserve: reserve, rebuild: rebuild)
|
|
1095
|
+
trimmed += dropped
|
|
1096
|
+
return named
|
|
1097
|
+
}
|
|
1098
|
+
if !history.isEmpty {
|
|
1099
|
+
// Trimmed only on the retry after an overflow: counting tokens up
|
|
1100
|
+
// front would tax every call to learn, almost always, that it fits.
|
|
1101
|
+
var entries = history
|
|
1102
|
+
if rebuild {
|
|
1103
|
+
let (kept, dropped) = await fitted(history, model: model, instructions: instructions,
|
|
1104
|
+
tools: tools, prompt: prompt, reserve: reserve)
|
|
1105
|
+
entries = kept
|
|
1106
|
+
trimmed += dropped
|
|
1107
|
+
}
|
|
1108
|
+
return transcriptSession(model: model, instructions: instructions, tools: tools, entries: entries)
|
|
1109
|
+
}
|
|
1110
|
+
if tools.isEmpty { return sessionFor(model: model, instructions: instructions, reuse: reuse) }
|
|
1111
|
+
return LanguageModelSession(model: model, tools: tools, instructions: instructions)
|
|
709
1112
|
}
|
|
710
1113
|
|
|
711
|
-
|
|
712
|
-
|
|
1114
|
+
var emittedEvents = false
|
|
1115
|
+
func generate(on session: LanguageModelSession) async throws -> Outcome {
|
|
1116
|
+
if wantsStream {
|
|
1117
|
+
return try await streamed(session: session, prompt: prompt, schema: schema,
|
|
1118
|
+
includeSchema: includeSchema, options: options,
|
|
1119
|
+
maxTokens: maxTokens, onEvent: { emittedEvents = true })
|
|
1120
|
+
}
|
|
713
1121
|
if let schema {
|
|
714
1122
|
let response = try await session.respond(
|
|
715
|
-
to: prompt, schema: schema, includeSchemaInPrompt: includeSchema, options: options
|
|
716
|
-
)
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
1123
|
+
to: prompt, schema: schema, includeSchemaInPrompt: includeSchema, options: options)
|
|
1124
|
+
try Task.checkCancellation()
|
|
1125
|
+
var outcome = Outcome(content: response.content.jsonString)
|
|
1126
|
+
outcome.toolCalls = toolActivity(response.transcriptEntries)
|
|
1127
|
+
if #available(macOS 27.0, *) { outcome.usage = usageOf(response.usage) }
|
|
1128
|
+
outcome.finishReason = finishReason(usage: outcome.usage, maxTokens: maxTokens)
|
|
1129
|
+
return outcome
|
|
1130
|
+
}
|
|
1131
|
+
let response = try await session.respond(to: prompt, options: options)
|
|
1132
|
+
try Task.checkCancellation()
|
|
1133
|
+
var outcome = Outcome(content: response.content)
|
|
1134
|
+
outcome.toolCalls = toolActivity(response.transcriptEntries)
|
|
1135
|
+
if #available(macOS 27.0, *) { outcome.usage = usageOf(response.usage) }
|
|
1136
|
+
outcome.finishReason = finishReason(usage: outcome.usage, maxTokens: maxTokens)
|
|
1137
|
+
return outcome
|
|
1138
|
+
}
|
|
1139
|
+
|
|
1140
|
+
// A named session recovers from overflow on its own: it is stateful, so
|
|
1141
|
+
// without this one long turn would leave it permanently unusable.
|
|
1142
|
+
// Stateless history is trimmed only when the caller opted in.
|
|
1143
|
+
let canTrim = (sessionId?.isEmpty == false)
|
|
1144
|
+
|| (!history.isEmpty && envelope["trimHistory"] as? Bool == true)
|
|
1145
|
+
var outcome: Outcome
|
|
1146
|
+
do {
|
|
1147
|
+
do {
|
|
1148
|
+
outcome = try await generate(on: await makeSession(rebuild: false))
|
|
1149
|
+
} catch let error where classify(error).kind == "context" && canTrim && !emittedEvents {
|
|
1150
|
+
// The conversation outgrew the window. Rebuild it from the newest
|
|
1151
|
+
// turns that still fit and try once more. Only when nothing has
|
|
1152
|
+
// been streamed yet: a retry after deltas would duplicate output.
|
|
1153
|
+
outcome = try await generate(on: await makeSession(rebuild: true))
|
|
1154
|
+
}
|
|
727
1155
|
} catch {
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
|
|
1156
|
+
if classify(error).kind == "stopped" {
|
|
1157
|
+
// The client is running a tool call itself; it saw the call as an
|
|
1158
|
+
// event and will continue through `history`.
|
|
1159
|
+
var payload: [String: Any] = ["ok": true, "content": "", "finishReason": "toolCalls"]
|
|
1160
|
+
if wantsStream { payload["done"] = true }
|
|
1161
|
+
if trimmed > 0 { payload["trimmedTurns"] = trimmed }
|
|
1162
|
+
emit(payload)
|
|
1163
|
+
return
|
|
1164
|
+
}
|
|
1165
|
+
var payload = failure(error)
|
|
1166
|
+
if trimmed > 0 { payload["trimmedTurns"] = trimmed }
|
|
731
1167
|
emit(payload)
|
|
1168
|
+
return
|
|
732
1169
|
}
|
|
1170
|
+
|
|
1171
|
+
recordTurn(sessionId: sessionId, prompt: promptText, content: outcome.content)
|
|
1172
|
+
var payload: [String: Any] = ["ok": true, "content": outcome.content,
|
|
1173
|
+
"finishReason": outcome.finishReason]
|
|
1174
|
+
if wantsStream { payload["done"] = true }
|
|
1175
|
+
if let usage = outcome.usage { payload["usage"] = usage }
|
|
1176
|
+
if !outcome.toolCalls.isEmpty { payload["toolCalls"] = outcome.toolCalls }
|
|
1177
|
+
if trimmed > 0 { payload["trimmedTurns"] = trimmed }
|
|
1178
|
+
emit(payload)
|
|
733
1179
|
}
|
|
734
1180
|
|
|
735
1181
|
@available(macOS 26.0, *)
|
|
736
|
-
private func recordTurn(sessionId: String?, prompt: String, content: String
|
|
737
|
-
instructions: String) {
|
|
1182
|
+
private func recordTurn(sessionId: String?, prompt: String, content: String) {
|
|
738
1183
|
guard let sid = sessionId, !sid.isEmpty else { return }
|
|
739
1184
|
if var entry = SessionStore.named[sid] {
|
|
740
1185
|
entry.history.append(["role": "user", "content": prompt])
|
|
@@ -749,42 +1194,66 @@ private func recordTurn(sessionId: String?, prompt: String, content: String,
|
|
|
749
1194
|
}
|
|
750
1195
|
}
|
|
751
1196
|
|
|
752
|
-
/// Streaming generation.
|
|
753
|
-
///
|
|
754
|
-
///
|
|
755
|
-
///
|
|
1197
|
+
/// Streaming generation. Text snapshots carry the cumulative partial, so deltas
|
|
1198
|
+
/// are computed by stripping the previous prefix; when the model revises
|
|
1199
|
+
/// earlier text (rare for plain prose) the whole new partial is sent so the
|
|
1200
|
+
/// client never silently drops a correction.
|
|
1201
|
+
///
|
|
1202
|
+
/// With a schema, each snapshot is the JSON generated so far, sent whole as
|
|
1203
|
+
/// `partial`: a partial object is a usable thing to render, where a raw text
|
|
1204
|
+
/// delta of JSON is not.
|
|
756
1205
|
@available(macOS 26.0, *)
|
|
757
|
-
private func
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
1206
|
+
private func streamed(
|
|
1207
|
+
session: LanguageModelSession, prompt: Prompt, schema: GenerationSchema?,
|
|
1208
|
+
includeSchema: Bool, options: GenerationOptions, maxTokens: Int?,
|
|
1209
|
+
onEvent: () -> Void
|
|
1210
|
+
) async throws -> Outcome {
|
|
1211
|
+
if let schema {
|
|
1212
|
+
let stream = session.streamResponse(to: prompt, schema: schema,
|
|
1213
|
+
includeSchemaInPrompt: includeSchema, options: options)
|
|
1214
|
+
var last = ""
|
|
1215
|
+
var outcome = Outcome(content: "")
|
|
764
1216
|
for try await snapshot in stream {
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
delta = current
|
|
1217
|
+
try Task.checkCancellation()
|
|
1218
|
+
let json = snapshot.rawContent.jsonString
|
|
1219
|
+
if json != last {
|
|
1220
|
+
emit(["ok": true, "partial": json, "done": false])
|
|
1221
|
+
onEvent()
|
|
1222
|
+
last = json
|
|
772
1223
|
}
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
1224
|
+
if #available(macOS 27.0, *) {
|
|
1225
|
+
outcome.usage = usageOf(snapshot.usage)
|
|
1226
|
+
outcome.toolCalls = toolActivity(snapshot.transcriptEntries)
|
|
776
1227
|
}
|
|
777
1228
|
}
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
1229
|
+
// A cancelled stream can end quietly rather than throw; without this the
|
|
1230
|
+
// truncated text would be reported as a normal, finished answer.
|
|
1231
|
+
try Task.checkCancellation()
|
|
1232
|
+
outcome.content = last
|
|
1233
|
+
outcome.finishReason = finishReason(usage: outcome.usage, maxTokens: maxTokens)
|
|
1234
|
+
return outcome
|
|
1235
|
+
}
|
|
1236
|
+
let stream = session.streamResponse(to: prompt, options: options)
|
|
1237
|
+
var previous = ""
|
|
1238
|
+
var outcome = Outcome(content: "")
|
|
1239
|
+
for try await snapshot in stream {
|
|
1240
|
+
try Task.checkCancellation()
|
|
1241
|
+
let current: String = snapshot.content
|
|
1242
|
+
let delta = current.hasPrefix(previous) ? String(current.dropFirst(previous.count)) : current
|
|
1243
|
+
previous = current
|
|
1244
|
+
if !delta.isEmpty {
|
|
1245
|
+
emit(["ok": true, "delta": delta, "done": false])
|
|
1246
|
+
onEvent()
|
|
1247
|
+
}
|
|
1248
|
+
if #available(macOS 27.0, *) {
|
|
1249
|
+
outcome.usage = usageOf(snapshot.usage)
|
|
1250
|
+
outcome.toolCalls = toolActivity(snapshot.transcriptEntries)
|
|
1251
|
+
}
|
|
787
1252
|
}
|
|
1253
|
+
try Task.checkCancellation()
|
|
1254
|
+
outcome.content = previous
|
|
1255
|
+
outcome.finishReason = finishReason(usage: outcome.usage, maxTokens: maxTokens)
|
|
1256
|
+
return outcome
|
|
788
1257
|
}
|
|
789
1258
|
|
|
790
1259
|
private func describe(_ reason: SystemLanguageModel.Availability.UnavailableReason) -> String {
|
|
@@ -796,15 +1265,18 @@ private func describe(_ reason: SystemLanguageModel.Availability.UnavailableReas
|
|
|
796
1265
|
}
|
|
797
1266
|
}
|
|
798
1267
|
|
|
799
|
-
/// Private Cloud Compute
|
|
1268
|
+
/// Private Cloud Compute, described without calling it.
|
|
800
1269
|
///
|
|
801
1270
|
/// PCC *inference* needs `com.apple.developer.private-cloud-compute`, which is
|
|
802
1271
|
/// AMFI-restricted and unavailable to any installable package — that is why the
|
|
803
|
-
/// cloud tier goes through Shortcuts. But
|
|
804
|
-
/// unentitled process, so a caller can
|
|
805
|
-
///
|
|
1272
|
+
/// cloud tier goes through Shortcuts. But the model's quota, capabilities and
|
|
1273
|
+
/// context size are all readable from an unentitled process, so a caller can
|
|
1274
|
+
/// see what the cloud tier offers, and whether it is worth trying, before
|
|
1275
|
+
/// spending a Shortcuts round trip on it. On macOS 27 Golden Gate this is the
|
|
1276
|
+
/// next-generation server model Siri AI is built on: it reports reasoning,
|
|
1277
|
+
/// vision and tool calling, with a 32,768-token window.
|
|
806
1278
|
@available(macOS 27.0, *)
|
|
807
|
-
private func
|
|
1279
|
+
private func cloudModel() async -> [String: Any] {
|
|
808
1280
|
let pcc = PrivateCloudComputeLanguageModel()
|
|
809
1281
|
var out: [String: Any] = ["isAvailable": pcc.isAvailable]
|
|
810
1282
|
let usage = pcc.quotaUsage
|
|
@@ -820,6 +1292,14 @@ private func cloudQuota() -> [String: Any] {
|
|
|
820
1292
|
if let reset = usage.resetDate {
|
|
821
1293
|
out["resetDate"] = ISO8601DateFormatter().string(from: reset)
|
|
822
1294
|
}
|
|
1295
|
+
let capabilities = pcc.capabilities
|
|
1296
|
+
out["capabilities"] = [
|
|
1297
|
+
"vision": capabilities.contains(.vision),
|
|
1298
|
+
"guidedGeneration": capabilities.contains(.guidedGeneration),
|
|
1299
|
+
"reasoning": capabilities.contains(.reasoning),
|
|
1300
|
+
"toolCalling": capabilities.contains(.toolCalling),
|
|
1301
|
+
]
|
|
1302
|
+
if let contextSize = try? await pcc.contextSize { out["contextSize"] = contextSize }
|
|
823
1303
|
return out
|
|
824
1304
|
}
|
|
825
1305
|
|
|
@@ -850,41 +1330,54 @@ struct AppleLLMHelper {
|
|
|
850
1330
|
"toolCalling": capabilities.contains(.toolCalling),
|
|
851
1331
|
]
|
|
852
1332
|
payload["useCases"] = ["general", "contentTagging"]
|
|
853
|
-
payload["cloud"] =
|
|
854
|
-
payload["features"] = [
|
|
855
|
-
"streaming": true,
|
|
856
|
-
"sessions": true,
|
|
857
|
-
"history": true,
|
|
858
|
-
"labelledAttachments": true,
|
|
859
|
-
"builtInTools": ["ocr", "barcode", "spotlight"],
|
|
860
|
-
]
|
|
1333
|
+
payload["cloud"] = await cloudModel()
|
|
861
1334
|
}
|
|
1335
|
+
payload["features"] = [
|
|
1336
|
+
"protocol": 2,
|
|
1337
|
+
"streaming": true,
|
|
1338
|
+
"structuredStreaming": true,
|
|
1339
|
+
"sessions": true,
|
|
1340
|
+
"history": true,
|
|
1341
|
+
"transcripts": true,
|
|
1342
|
+
"labelledAttachments": true,
|
|
1343
|
+
"functionTools": true,
|
|
1344
|
+
"cancellation": true,
|
|
1345
|
+
"builtInTools": ["ocr", "barcode", "spotlight"],
|
|
1346
|
+
]
|
|
862
1347
|
emit(payload)
|
|
863
1348
|
return
|
|
864
1349
|
}
|
|
865
1350
|
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
//
|
|
869
|
-
//
|
|
870
|
-
// request instead makes the model reload between calls, which measured
|
|
871
|
-
// at ~17s per request against ~1.5s once it is resident.
|
|
1351
|
+
// Serve mode: one request per stdin line, for as long as the parent
|
|
1352
|
+
// keeps the pipe open. Spawning a process per request instead makes the
|
|
1353
|
+
// model reload between calls, which measured at ~17s per request
|
|
1354
|
+
// against ~1.5s once it is resident.
|
|
872
1355
|
//
|
|
873
|
-
//
|
|
874
|
-
//
|
|
875
|
-
//
|
|
876
|
-
// sending the next request; the next stdin line is not consumed until
|
|
877
|
-
// the stream completes, so replies cannot interleave.
|
|
1356
|
+
// stdin is read on its own thread so control lines (cancel, toolResult)
|
|
1357
|
+
// reach a request that is still running; see Control. Requests are
|
|
1358
|
+
// still handled strictly one at a time, in order.
|
|
878
1359
|
if CommandLine.arguments.contains("--serve") {
|
|
879
|
-
|
|
880
|
-
|
|
1360
|
+
let requests = AsyncStream<String> { continuation in
|
|
1361
|
+
let reader = Thread {
|
|
1362
|
+
while let line = readLine(strippingNewline: true) {
|
|
1363
|
+
if line.isEmpty || Control.shared.intercept(line) { continue }
|
|
1364
|
+
continuation.yield(line)
|
|
1365
|
+
}
|
|
1366
|
+
Control.shared.stdinClosed()
|
|
1367
|
+
continuation.finish()
|
|
1368
|
+
}
|
|
1369
|
+
reader.start()
|
|
1370
|
+
}
|
|
1371
|
+
for await line in requests {
|
|
881
1372
|
guard let data = line.data(using: .utf8),
|
|
882
1373
|
let envelope = (try? JSONSerialization.jsonObject(with: data)) as? [String: Any]
|
|
883
1374
|
else {
|
|
884
1375
|
emit(["ok": false, "error": "helper could not parse a request line as JSON"])
|
|
885
1376
|
continue
|
|
886
1377
|
}
|
|
887
|
-
await
|
|
1378
|
+
await Control.shared.run(id: envelope["id"] as? String) {
|
|
1379
|
+
await handle(envelope: envelope)
|
|
1380
|
+
}
|
|
888
1381
|
}
|
|
889
1382
|
return
|
|
890
1383
|
}
|
|
@@ -894,7 +1387,7 @@ struct AppleLLMHelper {
|
|
|
894
1387
|
emit(["ok": false, "error": "helper could not parse its stdin payload as JSON"])
|
|
895
1388
|
return
|
|
896
1389
|
}
|
|
897
|
-
|
|
898
|
-
await handle(envelope: envelope
|
|
1390
|
+
Control.shared.stdinClosed()
|
|
1391
|
+
await handle(envelope: envelope)
|
|
899
1392
|
}
|
|
900
1393
|
}
|