apple-llm 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,9 +1,7 @@
1
1
  //
2
2
  // apple-llm helper — the whole on-device path.
3
3
  //
4
- // Extracted from api-scribe (MIT, https://github.com/…/api-scribe), where it
5
- // lived as a TypeScript string literal in `src/llm/apple-helper.ts`. This file
6
- // is the single source of truth: `scripts/embed-helper.mjs` copies it verbatim
4
+ // This file is the single source of truth: `scripts/embed-helper.mjs` copies it verbatim
7
5
  // into both the npm and the pip package, and a test in each asserts the copies
8
6
  // still hash equal to this file.
9
7
  //
@@ -11,26 +9,37 @@
11
9
  // which ships with macOS 27 but is gated behind a machine-wide `sudo fm
12
10
  // license`. Do not depend on `fm`.
13
11
  //
14
- // Protocol:
12
+ // Protocol (version 2; every version-1 request still works unchanged):
15
13
  // helper --probe -> stdout: one JSON object describing this machine.
16
- // helper --serve -> one request JSON per stdin line,
17
- // one response JSON per stdout line, in order.
18
- // EXCEPT op "stream": N delta lines + one final line,
19
- // all belonging to that single request. The next request
20
- // is not read until the stream completes, so ordering is
21
- // preserved: the client must keep reading lines until
22
- // "done":true before sending the next request.
14
+ // helper --serve -> one request JSON per stdin line; for each request, zero
15
+ // or more *event* lines ("done":false) and then exactly one
16
+ // *final* line (anything without "done":false), in order.
17
+ // Plain generate emits no events, so a version-1 client
18
+ // that reads one line per request keeps working.
19
+ // The client must not send the next request until the
20
+ // final line arrives. It MAY send control lines (cancel,
21
+ // toolResult) while a request is in flight: stdin is read
22
+ // on its own thread, so those are seen mid-generation.
23
23
  // helper -> one request on stdin, one response on stdout
24
24
  // ("stream" collapses to a single final response here,
25
- // since one-shot stdout is not incremental).
25
+ // since one-shot stdout is not incremental; function
26
+ // tools need --serve, since stdin is already consumed).
26
27
  //
27
28
  // request: {"op"?:"generate"|"stream"|"countTokens"|"prewarm"|"history"|"reset",
28
29
  // // default generate
30
+ // "id"?:string, // target for "cancel"
29
31
  // "instructions":string,"prompt":string,"schema"?:object|null,
30
32
  // "temperature"?:number,"maxTokens"?:number,
31
33
  // "includeSchemaInPrompt"?:bool,"reuseSession"?:bool,
32
34
  // "sessionId"?:string, // multi-turn conversation
35
+ // "history"?:[{"role":"user"|"assistant","content":string,
36
+ // "toolCalls"?:[{id,name,arguments}]}
37
+ // |{"role":"tool","toolCallId":string,"name":string,
38
+ // "content":string}], // stateless multi-turn
39
+ // "trimHistory"?:bool, // drop oldest turns to fit
33
40
  // "tools"?:["ocr"|"barcode"|"spotlight"], // built-in FM tools, 27+
41
+ // "functions"?:[{"name":string,"description":string,
42
+ // "parameters":object}], // tools run by the client
34
43
  // "images"?:[string|{"path":string,"label"?:string}],
35
44
  // // file paths, macOS 27+
36
45
  // "useCase"?:"general"|"contentTagging",
@@ -38,10 +47,19 @@
38
47
  // "sampling"?:{"mode":"greedy"}
39
48
  // | {"mode":"topK","k":int,"seed"?:int}
40
49
  // | {"mode":"threshold","p":number,"seed"?:int}}
41
- // response: {"ok":true,"content":string} // generate
42
- // | {"ok":true,"delta":string,"done":false} // stream partial
43
- // | {"ok":true,"content":string,"done":true} // stream final
44
- // | {"ok":true,"tokens":int} // countTokens
50
+ // control: {"op":"cancel","id"?:string} // no reply line of its own
51
+ // | {"op":"toolResult","callId":string,"output"?:string,
52
+ // "isError"?:bool,"stop"?:bool} // answers a toolCall event
53
+ // event: {"ok":true,"delta":string,"done":false} // stream, text
54
+ // | {"ok":true,"partial":string,"done":false} // stream, schema:
55
+ // // JSON so far
56
+ // | {"ok":true,"toolCall":{"id","name","arguments"},"done":false}
57
+ // final: {"ok":true,"content":string, // generate / stream
58
+ // "done"?:true,"finishReason"?:"stop"|"length"|"toolCalls",
59
+ // "usage"?:{inputTokens,outputTokens,cachedInputTokens},
60
+ // "toolCalls"?:[{id,name,arguments,output?}],
61
+ // "trimmedTurns"?:int}
62
+ // | {"ok":true,"tokens":int,"contextSize":int} // countTokens
45
63
  // | {"ok":true,"prewarmed":true} // prewarm
46
64
  // | {"ok":true,"history":[{role,content}], // history
47
65
  // "instructions":string,"sessionId":string}
@@ -49,17 +67,23 @@
49
67
  // | {"ok":false,"error":string,"kind"?:string}
50
68
  //
51
69
  // `kind` is one of availability | schema | context | quota | guardrail |
52
- // timeout | unsupported | generation, so callers can raise typed errors
53
- // instead of matching on strings. A quota error may carry `resetDate`.
70
+ // timeout | unsupported | tool | cancelled | busy | generation, so callers can raise
71
+ // typed errors instead of matching on strings. A quota error may carry
72
+ // `resetDate`; a context error may carry `contextSize` and `tokenCount`.
73
+ //
74
+ // A function tool call is answered by the client with a toolResult line. A
75
+ // reply with "stop":true ends generation there with finishReason "toolCalls"
76
+ // and empty content — the client executes the call itself and continues the
77
+ // conversation later through `history`, which is how an OpenAI-style caller
78
+ // (or the Vercel AI SDK) drives tools.
54
79
  //
55
80
  // --probe reports availability, contextSize, variant, the model's
56
- // capabilities (vision / guidedGeneration / reasoning / toolCalling) and the
57
- // Private Cloud Compute quota status. That last one is readable without the
58
- // `com.apple.developer.private-cloud-compute` entitlement that blocks PCC
59
- // *inference*, so it is the one piece of first-party cloud state available
60
- // to an unentitled process. It also reports `features` (streaming, sessions,
61
- // labelledAttachments, builtInTools) so callers can fail fast with a message
62
- // instead of a stale binary's silent misbehaviour.
81
+ // capabilities (vision / guidedGeneration / reasoning / toolCalling) and,
82
+ // under "cloud", the same for Private Cloud Compute plus its quota. Those are
83
+ // readable without the `com.apple.developer.private-cloud-compute`
84
+ // entitlement that blocks PCC *inference*, so they are the first-party cloud
85
+ // state available to an unentitled process. It also reports `features` so callers can fail
86
+ // fast with a message instead of a stale binary's silent misbehaviour.
63
87
  //
64
88
 
65
89
  import Foundation
@@ -71,36 +95,225 @@ import _Vision_FoundationModels
71
95
  import _CoreSpotlight_FoundationModels
72
96
  #endif
73
97
 
98
+ /// Serialises writes. Function tools can run concurrently when the model asks
99
+ /// for several at once, and two unsynchronised `print`s can interleave bytes
100
+ /// inside a line, which would corrupt the framing for every later request.
101
+ private let emitLock = NSLock()
102
+
74
103
  private func emit(_ object: [String: Any]) {
75
- guard let data = try? JSONSerialization.data(withJSONObject: object),
76
- let text = String(data: data, encoding: .utf8) else {
77
- print("{\"ok\":false,\"error\":\"helper could not encode its response\"}")
78
- fflush(stdout)
79
- return
104
+ var text = "{\"ok\":false,\"error\":\"helper could not encode its response\"}"
105
+ if let data = try? JSONSerialization.data(withJSONObject: object),
106
+ let encoded = String(data: data, encoding: .utf8) {
107
+ text = encoded
80
108
  }
109
+ emitLock.lock()
81
110
  // stdout is a pipe here, so it is fully buffered; the parent is waiting on
82
111
  // this line and would otherwise see nothing until the process exits.
83
112
  print(text)
84
113
  fflush(stdout)
114
+ emitLock.unlock()
115
+ }
116
+
117
+ private func jsonValue(_ text: String) -> Any? {
118
+ guard let data = text.data(using: .utf8) else { return nil }
119
+ return try? JSONSerialization.jsonObject(with: data, options: [.fragmentsAllowed])
120
+ }
121
+
122
+ private func jsonText(_ value: Any) -> String {
123
+ guard let data = try? JSONSerialization.data(withJSONObject: value, options: [.fragmentsAllowed]),
124
+ let text = String(data: data, encoding: .utf8) else { return "{}" }
125
+ return text
126
+ }
127
+
128
+ /// A reply to a function tool call, sent by the parent as a toolResult line
129
+ /// while the request is still in flight.
130
+ private enum ToolReply {
131
+ case output(String)
132
+ case failure(String)
133
+ case stop
134
+ case cancelled
135
+ }
136
+
137
+ /// Routes the control lines that have to be seen *while* a request runs —
138
+ /// cancellation and tool results. Everything else is a request and waits its
139
+ /// turn in the serial queue.
140
+ ///
141
+ /// Why a reader thread rather than reading stdin between requests: a tool call
142
+ /// suspends generation until the parent answers, and the answer arrives on
143
+ /// stdin. Reading stdin only between requests would deadlock the first tool
144
+ /// call; it would also make cancellation impossible, since the cancel line
145
+ /// would sit unread until the generation it was meant to stop had finished.
146
+ private final class Control: @unchecked Sendable {
147
+ static let shared = Control()
148
+
149
+ private let lock = NSLock()
150
+ private var currentId: String?
151
+ private var currentTask: Task<Void, Never>?
152
+ /// Cancels that arrived before their request started. Bounded below.
153
+ private var cancelledEarly: Set<String> = []
154
+ private var waiting: [String: CheckedContinuation<ToolReply, Never>] = [:]
155
+ private var closed = false
156
+
157
+ /// True when `line` was a control message and has been handled here.
158
+ func intercept(_ line: String) -> Bool {
159
+ // Cheap pre-check: a prompt can be tens of KB, and most lines are
160
+ // requests, which never carry these two ops.
161
+ guard line.contains("\"cancel\"") || line.contains("\"toolResult\""),
162
+ let object = jsonValue(line) as? [String: Any],
163
+ let op = object["op"] as? String else { return false }
164
+ switch op {
165
+ case "cancel":
166
+ cancel(id: object["id"] as? String)
167
+ return true
168
+ case "toolResult":
169
+ guard let callId = object["callId"] as? String else { return true }
170
+ let reply: ToolReply
171
+ if object["stop"] as? Bool == true {
172
+ reply = .stop
173
+ } else if object["isError"] as? Bool == true {
174
+ reply = .failure(object["output"] as? String ?? "the tool failed")
175
+ } else {
176
+ reply = .output(object["output"] as? String ?? "")
177
+ }
178
+ deliver(callId: callId, reply: reply)
179
+ return true
180
+ default:
181
+ return false
182
+ }
183
+ }
184
+
185
+ private func cancel(id: String?) {
186
+ lock.lock()
187
+ defer { lock.unlock() }
188
+ if id == nil || id == currentId {
189
+ currentTask?.cancel()
190
+ // A tool call parked on the parent would otherwise never notice.
191
+ for continuation in waiting.values { continuation.resume(returning: .cancelled) }
192
+ waiting.removeAll()
193
+ } else if let id {
194
+ // The request line may already be queued but not yet started.
195
+ if cancelledEarly.count > 256 { cancelledEarly.removeAll() }
196
+ cancelledEarly.insert(id)
197
+ }
198
+ }
199
+
200
+ private func deliver(callId: String, reply: ToolReply) {
201
+ lock.lock()
202
+ let continuation = waiting.removeValue(forKey: callId)
203
+ lock.unlock()
204
+ continuation?.resume(returning: reply)
205
+ }
206
+
207
+ /// Park a tool call until the parent answers it. `announce` emits the
208
+ /// toolCall event, and runs only once the call is registered, so an answer
209
+ /// cannot arrive before anything is waiting for it.
210
+ func awaitToolResult(callId: String, announce: () -> Void) async -> ToolReply {
211
+ await withCheckedContinuation { (continuation: CheckedContinuation<ToolReply, Never>) in
212
+ lock.lock()
213
+ if closed || currentTask?.isCancelled == true {
214
+ lock.unlock()
215
+ continuation.resume(returning: .cancelled)
216
+ return
217
+ }
218
+ waiting[callId] = continuation
219
+ lock.unlock()
220
+ announce()
221
+ }
222
+ }
223
+
224
+ /// Run one request as a cancellable task, recording it as the target of
225
+ /// any cancel line that arrives meanwhile.
226
+ func run(id: String?, _ body: @escaping () async -> Void) async {
227
+ let task: Task<Void, Never>? = lock.withLock {
228
+ if let id, cancelledEarly.remove(id) != nil { return nil }
229
+ let task = Task { await body() }
230
+ currentId = id
231
+ currentTask = task
232
+ return task
233
+ }
234
+ guard let task else {
235
+ emit(["ok": false, "kind": "cancelled", "error": "the request was cancelled"])
236
+ return
237
+ }
238
+ await task.value
239
+ lock.withLock {
240
+ currentId = nil
241
+ currentTask = nil
242
+ }
243
+ }
244
+
245
+ /// The parent closed stdin: nothing will ever answer a parked tool call.
246
+ func stdinClosed() {
247
+ lock.lock()
248
+ closed = true
249
+ currentTask?.cancel()
250
+ for continuation in waiting.values { continuation.resume(returning: .cancelled) }
251
+ waiting.removeAll()
252
+ lock.unlock()
253
+ }
254
+ }
255
+
256
+ /// Why a function tool call did not produce output.
257
+ private enum ClientToolFailure: Error, CustomStringConvertible {
258
+ case failed(name: String, message: String)
259
+ /// The client will run the call itself; end generation here.
260
+ case stop
261
+
262
+ var description: String {
263
+ switch self {
264
+ case .failed(let name, let message): return "tool \"\(name)\" failed: \(message)"
265
+ case .stop: return "stopped for a client tool call"
266
+ }
267
+ }
268
+ }
269
+
270
+ /// A tool defined by the client and executed there. The model's arguments are
271
+ /// sent up as a toolCall event and the call suspends until a toolResult line
272
+ /// answers it. `parameters` is the client's JSON Schema, already rewritten into
273
+ /// Apple's dialect, so the arguments are decoded under the same constrained
274
+ /// decoding guarantee as `json()`.
275
+ @available(macOS 26.0, *)
276
+ private struct ClientTool: Tool {
277
+ typealias Arguments = GeneratedContent
278
+ typealias Output = String
279
+
280
+ let name: String
281
+ let description: String
282
+ let parameters: GenerationSchema
283
+
284
+ func call(arguments: GeneratedContent) async throws -> String {
285
+ let callId = "call_" + UUID().uuidString.replacingOccurrences(of: "-", with: "").lowercased().prefix(24)
286
+ let args = jsonValue(arguments.jsonString) ?? [String: Any]()
287
+ let reply = await Control.shared.awaitToolResult(callId: callId) {
288
+ emit(["ok": true, "done": false,
289
+ "toolCall": ["id": callId, "name": name, "arguments": args]])
290
+ }
291
+ switch reply {
292
+ case .output(let text): return text
293
+ case .failure(let message): throw ClientToolFailure.failed(name: name, message: message)
294
+ case .stop: throw ClientToolFailure.stop
295
+ case .cancelled: throw CancellationError()
296
+ }
297
+ }
85
298
  }
86
299
 
87
300
  /// Optional session reuse, off by default.
88
301
  ///
89
- /// api-scribe kept one session alive while the instructions were unchanged, on
90
- /// the finding that allocating a fresh LanguageModelSession per request made
91
- /// every other call stall ~16s while assets cycled — a strict 17s/1.5s
92
- /// alternation, uncorrelated with prompt size.
302
+ /// An earlier prototype kept one session alive while the instructions were
303
+ /// unchanged, on the finding that allocating a fresh LanguageModelSession per
304
+ /// request made every other call stall ~16s while assets cycled — a strict
305
+ /// 17s/1.5s alternation, uncorrelated with prompt size.
93
306
  ///
94
307
  /// That did not reproduce here. Measured on macOS 27 / M-series inside one
95
308
  /// long-lived process, 6 calls per arm: session reused -> median 0.63s; a fresh
96
309
  /// session every call -> median 0.64s, max 0.68s, no variance. The alternation
97
310
  /// would have hit three times in six calls. The load-bearing part of the fix
98
- /// appears to be the long-lived *process* (api-scribe measured ~17s a call when
99
- /// it spawned a helper per request, against ~1.5s once resident), not the
311
+ /// appears to be the long-lived *process* (~17s a call was measured when a
312
+ /// helper was spawned per request, against ~1.5s once resident), not the
100
313
  /// shared session.
101
314
  ///
102
315
  /// So the default here is a fresh session per request, because a library cannot
103
- /// let two unrelated calls share a transcript — api-scribe could, since every
316
+ /// let two unrelated calls share a transcript — the prototype could, since every
104
317
  /// one of its prompts carried its own whole context. The old behaviour is kept
105
318
  /// behind `reuseSession` rather than deleted: the original finding was measured
106
319
  /// too, possibly on macOS 26, and the doubt is worth preserving. If per-call
@@ -109,10 +322,10 @@ private func emit(_ object: [String: Any]) {
109
322
  /// Named sessions (`sessionId`) are the multi-turn counterpart: same process,
110
323
  /// but the session is keyed by id so a conversation accumulates a native
111
324
  /// transcript across calls. History is also mirrored to
112
- /// `~/Library/Caches/apple-llm/sessions/<id>.json` so `history` survives a
113
- /// helper restart; the native transcript does not — after a restart the next
114
- /// call recreates the native session and continues from the stored history
115
- /// length, which callers should treat as a context break, not a loss.
325
+ /// `~/Library/Caches/apple-llm/sessions/<id>.json`. When the native session
326
+ /// has to be rebuilt — a helper restart, changed instructions or tools, or a
327
+ /// context overflow — it is rebuilt as a native transcript from that mirror,
328
+ /// oldest turns dropped until it fits, so a conversation survives all three.
116
329
  @available(macOS 26.0, *)
117
330
  private final class SessionHolder {
118
331
  static var session: LanguageModelSession?
@@ -178,10 +391,22 @@ private final class SessionStore {
178
391
  }
179
392
  }
180
393
 
394
+ /// Decoded schemas, keyed by their JSON text. A changing GenerationSchema costs
395
+ /// only ~0.15s per call, so per-request schemas are fine; this just makes a
396
+ /// repeated one free. Bounded, since a long-lived server may see many.
181
397
  @available(macOS 26.0, *)
182
- private func sessionFingerprint(useCase: String, guardrails: String, tools: [String],
183
- instructions: String) -> String {
184
- "\(useCase)|\(guardrails)|\(tools.sorted().joined(separator: ","))|\(instructions)"
398
+ private final class SchemaCache {
399
+ static var decoded: [String: GenerationSchema] = [:]
400
+
401
+ static func decode(_ object: Any) throws -> GenerationSchema {
402
+ let data = try JSONSerialization.data(withJSONObject: object, options: [.sortedKeys])
403
+ let key = String(data: data, encoding: .utf8) ?? ""
404
+ if let cached = decoded[key] { return cached }
405
+ let schema = try JSONDecoder().decode(GenerationSchema.self, from: data)
406
+ if decoded.count >= 64 { decoded.removeAll() }
407
+ decoded[key] = schema
408
+ return schema
409
+ }
185
410
  }
186
411
 
187
412
  @available(macOS 26.0, *)
@@ -208,49 +433,145 @@ private func sessionFor(
208
433
  return session
209
434
  }
210
435
 
436
+ /// Turn history turns into native transcript entries.
437
+ ///
438
+ /// A native transcript rather than a "Previous conversation:" preamble: the
439
+ /// model was trained on its own turn structure, and a preamble also costs the
440
+ /// history twice when a tool call echoes the prompt back.
441
+ @available(macOS 26.0, *)
442
+ private func transcriptEntries(from history: [[String: Any]]) -> [Transcript.Entry] {
443
+ var entries: [Transcript.Entry] = []
444
+ for turn in history {
445
+ let role = turn["role"] as? String ?? ""
446
+ let content = turn["content"] as? String ?? ""
447
+ switch role {
448
+ case "user":
449
+ entries.append(.prompt(Transcript.Prompt(
450
+ segments: [.text(Transcript.TextSegment(content: content))])))
451
+ case "assistant":
452
+ if !content.isEmpty {
453
+ entries.append(.response(Transcript.Response(
454
+ assetIDs: [], segments: [.text(Transcript.TextSegment(content: content))])))
455
+ }
456
+ let calls = (turn["toolCalls"] as? [[String: Any]] ?? []).compactMap {
457
+ call -> Transcript.ToolCall? in
458
+ guard let name = call["name"] as? String,
459
+ let arguments = try? GeneratedContent(json: jsonText(call["arguments"] ?? [String: Any]()))
460
+ else { return nil }
461
+ return Transcript.ToolCall(id: call["id"] as? String ?? UUID().uuidString,
462
+ toolName: name, arguments: arguments)
463
+ }
464
+ if !calls.isEmpty { entries.append(.toolCalls(Transcript.ToolCalls(calls))) }
465
+ case "tool":
466
+ entries.append(.toolOutput(Transcript.ToolOutput(
467
+ id: turn["toolCallId"] as? String ?? UUID().uuidString,
468
+ toolName: turn["name"] as? String ?? "tool",
469
+ segments: [.text(Transcript.TextSegment(content: content))])))
470
+ default:
471
+ continue
472
+ }
473
+ }
474
+ return entries
475
+ }
476
+
477
+ /// Drop the oldest turns until instructions, tools, history, prompt and the
478
+ /// response budget all fit in the context window. A turn is dropped whole —
479
+ /// from one user prompt up to the next — so a tool call is never left without
480
+ /// its output. Returns how many user turns were dropped, which the caller
481
+ /// reports: trimming is never silent.
482
+ ///
483
+ /// Runs only after a context overflow (or when rebuilding a named session),
484
+ /// never ahead of every call, and binary-searches the cut: log2(turns) token
485
+ /// counts rather than one per dropped turn.
486
+ ///
487
+ /// macOS 27 only, since it needs `tokenCount`; on 26 the history is left as is
488
+ /// and an overflow surfaces as a ContextLengthError.
489
+ @available(macOS 26.0, *)
490
+ private func fitted(
491
+ _ entries: [Transcript.Entry], model: SystemLanguageModel, instructions: String,
492
+ tools: [any Tool], prompt: Prompt, reserve: Int
493
+ ) async -> (entries: [Transcript.Entry], dropped: Int) {
494
+ guard #available(macOS 27.0, *), !entries.isEmpty else { return (entries, 0) }
495
+ var fixed = (try? await model.tokenCount(for: prompt)) ?? 0
496
+ if !instructions.isEmpty {
497
+ fixed += (try? await model.tokenCount(for: Instructions(instructions))) ?? 0
498
+ }
499
+ if !tools.isEmpty { fixed += (try? await model.tokenCount(for: tools)) ?? 0 }
500
+ let budget = model.contextSize - reserve - fixed
501
+ func fits(_ slice: ArraySlice<Transcript.Entry>) async -> Bool {
502
+ if slice.isEmpty { return true }
503
+ let used = (try? await model.tokenCount(for: Array(slice))) ?? Int.max
504
+ return used <= budget
505
+ }
506
+ if await fits(entries[...]) { return (entries, 0) }
507
+ // Cut points: the start of each user turn, plus "drop everything".
508
+ var cuts = entries.indices.filter { $0 > 0 && isPrompt(entries[$0]) }
509
+ cuts.append(entries.count)
510
+ var low = 0
511
+ var high = cuts.count - 1
512
+ while low < high {
513
+ let mid = (low + high) / 2
514
+ if await fits(entries[cuts[mid]...]) { high = mid } else { low = mid + 1 }
515
+ }
516
+ return (Array(entries[cuts[low]...]), low + 1)
517
+ }
518
+
519
+ @available(macOS 26.0, *)
520
+ private func isPrompt(_ entry: Transcript.Entry) -> Bool {
521
+ if case .prompt = entry { return true }
522
+ return false
523
+ }
524
+
525
+ @available(macOS 26.0, *)
526
+ private func transcriptSession(
527
+ model: SystemLanguageModel, instructions: String, tools: [any Tool],
528
+ entries: [Transcript.Entry]
529
+ ) -> LanguageModelSession {
530
+ if entries.isEmpty {
531
+ return tools.isEmpty
532
+ ? LanguageModelSession(model: model, instructions: instructions)
533
+ : LanguageModelSession(model: model, tools: tools, instructions: instructions)
534
+ }
535
+ var all: [Transcript.Entry] = []
536
+ if !instructions.isEmpty || !tools.isEmpty {
537
+ let segments: [Transcript.Segment] =
538
+ instructions.isEmpty ? [] : [.text(Transcript.TextSegment(content: instructions))]
539
+ all.append(.instructions(Transcript.Instructions(
540
+ segments: segments,
541
+ toolDefinitions: tools.map { Transcript.ToolDefinition(tool: $0) })))
542
+ }
543
+ all.append(contentsOf: entries)
544
+ return LanguageModelSession(model: model, tools: tools, transcript: Transcript(entries: all))
545
+ }
546
+
211
547
  /// Named multi-turn session. Tools are fixed at session construction, so a
212
- /// changed tool set recreates the native session; the mirrored history is kept
213
- /// so nothing the caller said is silently dropped from `history`.
548
+ /// changed tool set recreates the native session. Recreation (first use in
549
+ /// this process, changed parameters, or `rebuild` after a context overflow)
550
+ /// replays the mirrored history as a native transcript, trimmed to fit, so
551
+ /// nothing the caller said is silently dropped from `history` and the model
552
+ /// keeps as much of the conversation as the window allows.
214
553
  @available(macOS 26.0, *)
215
554
  private func namedSessionFor(
216
- id: String, model: SystemLanguageModel, instructions: String,
217
- useCase: String, guardrails: String, tools: [any Tool]
218
- ) -> LanguageModelSession {
219
- let toolNames = currentToolNames()
220
- let want = sessionFingerprint(useCase: useCase, guardrails: guardrails,
221
- tools: toolNames, instructions: instructions)
222
- if let existing = SessionStore.named[id], existing.fingerprint == want {
223
- return existing.session
224
- }
225
- // Recreate (first use, param change, or restart). Restore mirrored history
226
- // so `history` is continuous even though the native transcript restarts.
227
- let stored = SessionStore.loadHistory(id)
228
- let history = SessionStore.named[id]?.history ?? stored.history
229
- let session: LanguageModelSession
230
- if tools.isEmpty {
231
- session = LanguageModelSession(model: model, instructions: instructions)
232
- } else {
233
- session = LanguageModelSession(model: model, tools: tools, instructions: instructions)
555
+ id: String, model: SystemLanguageModel, instructions: String, fingerprint want: String,
556
+ tools: [any Tool], prompt: Prompt, reserve: Int, rebuild: Bool
557
+ ) async -> (session: LanguageModelSession, dropped: Int) {
558
+ if !rebuild, let existing = SessionStore.named[id], existing.fingerprint == want {
559
+ return (existing.session, 0)
234
560
  }
561
+ let history = SessionStore.named[id]?.history ?? SessionStore.loadHistory(id).history
562
+ let replay = transcriptEntries(from: history.suffix(40).map { $0 as [String: Any] })
563
+ let (kept, dropped) = await fitted(replay, model: model, instructions: instructions,
564
+ tools: tools, prompt: prompt, reserve: reserve)
565
+ let session = transcriptSession(model: model, instructions: instructions, tools: tools, entries: kept)
235
566
  SessionStore.named[id] = NamedSession(
236
567
  session: session, instructions: instructions,
237
568
  fingerprint: want, history: history)
238
569
  // Persist immediately so a fresh id is visible to `history` even before
239
570
  // its first turn completes.
240
571
  SessionStore.saveHistory(id: id, instructions: instructions, history: history)
241
- return session
572
+ return (session, dropped)
242
573
  }
243
574
 
244
- // The tool names for the in-flight request, stashed so namedSessionFor can
245
- // fingerprint on them without threading another parameter through sessionFor.
246
- @available(macOS 26.0, *)
247
- private final class RequestContext {
248
- static var toolNames: [String] = []
249
- }
250
-
251
- @available(macOS 26.0, *)
252
- private func currentToolNames() -> [String] { RequestContext.toolNames }
253
-
254
575
  /// Classify a generation failure so callers can raise a typed error rather than
255
576
  /// matching on a message Apple may reword in any OS release.
256
577
  ///
@@ -259,23 +580,41 @@ private func currentToolNames() -> [String] { RequestContext.toolNames }
259
580
  /// `rateLimited` carries a `resetDate`, which is what makes an on-device quota
260
581
  /// error actionable rather than just a failure.
261
582
  @available(macOS 26.0, *)
262
- private func classify(_ error: Error) -> (kind: String, resetDate: String?) {
583
+ private func classify(_ error: Error) -> (kind: String, extra: [String: Any]) {
584
+ if error is CancellationError { return ("cancelled", [:]) }
585
+ // Several processes using the model at once: the model manager turns a
586
+ // request away with ModelManagerError 1042, wrapped in an otherwise
587
+ // uninformative LanguageModelError -1. Nothing was generated, so the
588
+ // client can retry it; naming it lets the client do that safely.
589
+ let described = String(describing: error)
590
+ if described.contains("ModelManagerError"), described.contains("1042") { return ("busy", [:]) }
591
+ if let toolError = error as? LanguageModelSession.ToolCallError {
592
+ if toolError.underlyingError is CancellationError { return ("cancelled", [:]) }
593
+ if let client = toolError.underlyingError as? ClientToolFailure, case .stop = client {
594
+ return ("stopped", [:])
595
+ }
596
+ return ("tool", ["tool": toolError.tool.name])
597
+ }
263
598
  if #available(macOS 27.0, *) {
264
599
  if let modern = error as? LanguageModelError {
265
600
  switch modern {
266
- case .contextSizeExceeded: return ("context", nil)
601
+ case .contextSizeExceeded(let info):
602
+ return ("context", ["contextSize": info.contextSize, "tokenCount": info.tokenCount])
267
603
  case .rateLimited(let info):
268
- return ("quota", info.resetDate.map(ISO8601DateFormatter().string(from:)))
269
- case .guardrailViolation, .refusal: return ("guardrail", nil)
270
- case .timeout: return ("timeout", nil)
604
+ if let reset = info.resetDate {
605
+ return ("quota", ["resetDate": ISO8601DateFormatter().string(from: reset)])
606
+ }
607
+ return ("quota", [:])
608
+ case .guardrailViolation, .refusal: return ("guardrail", [:])
609
+ case .timeout: return ("timeout", [:])
271
610
  case .unsupportedCapability, .unsupportedGenerationGuide,
272
611
  .unsupportedLanguageOrLocale, .unsupportedTranscriptContent:
273
- return ("unsupported", nil)
274
- @unknown default: return ("generation", nil)
612
+ return ("unsupported", [:])
613
+ @unknown default: return ("generation", [:])
275
614
  }
276
615
  }
277
616
  }
278
- return (legacyKind(error), nil)
617
+ return (legacyKind(error), [:])
279
618
  }
280
619
 
281
620
  /// macOS 26's error type, kept so one helper source serves both OS versions.
@@ -299,6 +638,15 @@ private func legacyKind(_ error: Error) -> String {
299
638
  return "generation"
300
639
  }
301
640
 
641
+ @available(macOS 26.0, *)
642
+ private func failure(_ error: Error) -> [String: Any] {
643
+ let (kind, extra) = classify(error)
644
+ var payload: [String: Any] = ["ok": false, "kind": kind, "error": "\(error)"]
645
+ if kind == "cancelled" { payload["error"] = "the request was cancelled" }
646
+ for (key, value) in extra { payload[key] = value }
647
+ return payload
648
+ }
649
+
302
650
  /// Models are keyed by use case and guardrails, because both are fixed at
303
651
  /// construction. Building one is cheap; keeping them avoids re-resolving assets
304
652
  /// when a caller alternates between, say, general and contentTagging.
@@ -370,9 +718,37 @@ private func toolsFor(_ names: [String]) throws -> [any Tool] {
370
718
  return out
371
719
  }
372
720
 
721
+ /// Function tools from the request. Their parameter schemas go through Apple's
722
+ /// decoder like any response schema, and a rejected one is a schema error
723
+ /// naming the tool rather than a generic failure.
724
+ @available(macOS 26.0, *)
725
+ private func functionToolsFrom(_ value: Any?) throws -> [any Tool] {
726
+ guard let specs = value as? [[String: Any]] else { return [] }
727
+ var out: [any Tool] = []
728
+ for spec in specs {
729
+ guard let name = spec["name"] as? String, !name.isEmpty else {
730
+ throw ToolError.unknown("every function tool needs a name")
731
+ }
732
+ let parameters = spec["parameters"] ?? [
733
+ "type": "object", "title": "\(name)Arguments", "properties": [String: Any](),
734
+ "x-order": [String](), "required": [String](), "additionalProperties": false,
735
+ ]
736
+ let schema: GenerationSchema
737
+ do {
738
+ schema = try SchemaCache.decode(parameters)
739
+ } catch {
740
+ throw ToolError.schema("Apple rejected the parameters schema of tool \"\(name)\": \(error)")
741
+ }
742
+ out.append(ClientTool(name: name, description: spec["description"] as? String ?? "",
743
+ parameters: schema))
744
+ }
745
+ return out
746
+ }
747
+
373
748
  private enum ToolError: Error {
374
749
  case unknown(String)
375
750
  case unsupported(String)
751
+ case schema(String)
376
752
  }
377
753
 
378
754
  /// Coerce a JSON number regardless of int/double mismatch.
@@ -469,22 +845,67 @@ private func stringArray(_ value: Any?) -> [String] {
469
845
  (value as? [Any] ?? []).compactMap { $0 as? String }
470
846
  }
471
847
 
848
+ /// Every tool call made while producing one response, with its output where
849
+ /// the transcript recorded one — built-in and function tools alike, so a
850
+ /// caller can see *why* an answer says what it says.
851
+ @available(macOS 26.0, *)
852
+ private func toolActivity(_ entries: some Sequence<Transcript.Entry>) -> [[String: Any]] {
853
+ var calls: [[String: Any]] = []
854
+ var outputs: [String: String] = [:]
855
+ for entry in entries {
856
+ switch entry {
857
+ case .toolCalls(let batch):
858
+ for call in batch {
859
+ calls.append(["id": call.id, "name": call.toolName,
860
+ "arguments": jsonValue(call.arguments.jsonString) ?? NSNull()])
861
+ }
862
+ case .toolOutput(let output):
863
+ outputs[output.id] = output.segments.compactMap { segment -> String? in
864
+ if case .text(let text) = segment { return text.content }
865
+ return nil
866
+ }.joined()
867
+ default:
868
+ continue
869
+ }
870
+ }
871
+ return calls.map { call in
872
+ var out = call
873
+ if let id = call["id"] as? String, let text = outputs[id] { out["output"] = text }
874
+ return out
875
+ }
876
+ }
877
+
878
+ /// What one successful generation produced, before it is framed as a line.
879
+ private struct Outcome {
880
+ var content: String
881
+ var usage: [String: Any]?
882
+ var toolCalls: [[String: Any]] = []
883
+ var finishReason = "stop"
884
+ }
885
+
886
+ @available(macOS 27.0, *)
887
+ private func usageOf(_ usage: LanguageModelSession.Usage) -> [String: Any] {
888
+ ["inputTokens": usage.input.totalTokenCount,
889
+ "cachedInputTokens": usage.input.cachedTokenCount,
890
+ "outputTokens": usage.output.totalTokenCount]
891
+ }
892
+
893
+ /// A response that used its whole token budget was cut off, not finished.
894
+ /// Reported so a caller can tell a short answer from a truncated one.
895
+ private func finishReason(usage: [String: Any]?, maxTokens: Int?) -> String {
896
+ if let maxTokens, let out = usage?["outputTokens"] as? Int, out >= maxTokens { return "length" }
897
+ return "stop"
898
+ }
899
+
472
900
  /// One request/response cycle. Kept separate from the transport so that
473
901
  /// one-shot and serve modes cannot drift apart.
474
902
  @available(macOS 26.0, *)
475
- private func handle(
476
- envelope: [String: Any],
477
- schemaCache: inout [String: GenerationSchema]
478
- ) async {
479
- await handleEnvelope(envelope: envelope, schemaCache: &schemaCache, streaming: false)
903
+ private func handle(envelope: [String: Any]) async {
904
+ await handleEnvelope(envelope: envelope, streaming: false)
480
905
  }
481
906
 
482
907
  @available(macOS 26.0, *)
483
- private func handleEnvelope(
484
- envelope: [String: Any],
485
- schemaCache: inout [String: GenerationSchema],
486
- streaming: Bool
487
- ) async {
908
+ private func handleEnvelope(envelope: [String: Any], streaming: Bool) async {
488
909
  let useCase = envelope["useCase"] as? String ?? "general"
489
910
  let guardrails = envelope["guardrails"] as? String ?? "default"
490
911
  let model = modelFor(useCase: useCase, guardrails: guardrails)
@@ -502,8 +923,8 @@ private func handleEnvelope(
502
923
  let promptText = envelope["prompt"] as? String ?? ""
503
924
  let imageSpecs = imageSpecsFrom(envelope["images"])
504
925
  let toolNames = stringArray(envelope["tools"])
505
- RequestContext.toolNames = toolNames
506
926
  let sessionId = envelope["sessionId"] as? String
927
+ let historyTurns = envelope["history"] as? [[String: Any]] ?? []
507
928
 
508
929
  if !imageSpecs.isEmpty {
509
930
  if #available(macOS 27.0, *) {
@@ -526,24 +947,42 @@ private func handleEnvelope(
526
947
 
527
948
  let tools: [any Tool]
528
949
  do {
529
- tools = try toolsFor(toolNames)
950
+ tools = try toolsFor(toolNames) + functionToolsFrom(envelope["functions"])
530
951
  } catch let err as ToolError {
531
952
  switch err {
532
953
  case .unknown(let m), .unsupported(let m):
533
954
  emit(["ok": false, "kind": "unsupported", "error": m])
534
- return
955
+ case .schema(let m):
956
+ emit(["ok": false, "kind": "schema", "error": m])
535
957
  }
958
+ return
536
959
  } catch {
537
960
  emit(["ok": false, "kind": "unsupported", "error": "\(error)"])
538
961
  return
539
962
  }
540
963
  if !tools.isEmpty {
541
- guard model.capabilities.contains(.toolCalling) else {
542
- emit(["ok": false, "kind": "unsupported",
543
- "error": "this model does not support tool calling"])
964
+ if #available(macOS 27.0, *) {
965
+ guard model.capabilities.contains(.toolCalling) else {
966
+ emit(["ok": false, "kind": "unsupported",
967
+ "error": "this model does not support tool calling"])
968
+ return
969
+ }
970
+ }
971
+ }
972
+
973
+ // A response schema, decoded once for every op that needs it.
974
+ var schema: GenerationSchema? = nil
975
+ if let schemaObject = envelope["schema"], !(schemaObject is NSNull) {
976
+ do {
977
+ schema = try SchemaCache.decode(schemaObject)
978
+ } catch {
979
+ emit(["ok": false, "kind": "schema",
980
+ "error": "Apple rejected the response schema: \(error)"])
544
981
  return
545
982
  }
546
983
  }
984
+ let prompt = promptWith(text: promptText, imageSpecs: imageSpecs)
985
+ let history = transcriptEntries(from: historyTurns)
547
986
 
548
987
  // Token counting and prewarming share the model but not the generation path.
549
988
  // An empty op defaults to generate; an unknown op is an error rather than
@@ -561,19 +1000,25 @@ private func handleEnvelope(
561
1000
  return
562
1001
  }
563
1002
  do {
564
- // Instructions and schema ride in the same window as the prompt, so
565
- // a caller budgeting against contextSize needs all three counted.
566
- var total = try await model.tokenCount(for: promptWith(text: promptText,
567
- imageSpecs: imageSpecs))
1003
+ // Instructions, tools, schema and history all ride in the same
1004
+ // window as the prompt, so a caller budgeting against contextSize
1005
+ // needs every one of them counted.
1006
+ var total = try await model.tokenCount(for: prompt)
568
1007
  if !instructions.isEmpty {
569
1008
  total += try await model.tokenCount(for: Instructions(instructions))
570
1009
  }
571
1010
  if !tools.isEmpty {
572
1011
  total += (try? await model.tokenCount(for: tools)) ?? 0
573
1012
  }
1013
+ if let schema {
1014
+ total += (try? await model.tokenCount(for: schema)) ?? 0
1015
+ }
1016
+ if !history.isEmpty {
1017
+ total += (try? await model.tokenCount(for: history)) ?? 0
1018
+ }
574
1019
  emit(["ok": true, "tokens": total, "contextSize": model.contextSize])
575
1020
  } catch {
576
- emit(["ok": false, "kind": classify(error).kind, "error": "\(error)"])
1021
+ emit(failure(error))
577
1022
  }
578
1023
  return
579
1024
 
@@ -624,117 +1069,117 @@ private func handleEnvelope(
624
1069
  return
625
1070
  }
626
1071
 
627
- // Streaming + guided generation do not mix in v1: the schema stream yields
628
- // GeneratedContent snapshots whose partials are not plain-text deltas.
629
- // Reject loudly rather than emitting misleading partial JSON.
630
1072
  let wantsStream = (op == "stream") || streaming
631
- if wantsStream,
632
- let schemaObject = envelope["schema"], !(schemaObject is NSNull) {
633
- emit(["ok": false, "kind": "unsupported",
634
- "error": "streaming with a schema is not supported; use generate for JSON"])
635
- return
636
- }
637
-
638
- // No schema means a free-text call. api-scribe only ever asked for JSON and
639
- // so required a schema on every request; `text()` needs the unconstrained
640
- // `respond(to:options:)` overload instead.
641
- var schema: GenerationSchema? = nil
642
- if let schemaObject = envelope["schema"], !(schemaObject is NSNull) {
643
- guard let schemaData = try? JSONSerialization.data(withJSONObject: schemaObject),
644
- let schemaKey = String(data: schemaData, encoding: .utf8) else {
645
- emit(["ok": false, "kind": "schema",
646
- "error": "helper payload is missing a usable schema"])
647
- return
648
- }
649
- // A changing GenerationSchema costs only ~0.15s per call, so per-request
650
- // schemas are fine; this cache just makes a repeated one free.
651
- if let cached = schemaCache[schemaKey] {
652
- schema = cached
653
- } else {
654
- do {
655
- let decoded = try JSONDecoder().decode(GenerationSchema.self, from: schemaData)
656
- // Bound the cache: a long-lived server may see many schemas.
657
- if schemaCache.count >= 64 { schemaCache.removeAll() }
658
- schemaCache[schemaKey] = decoded
659
- schema = decoded
660
- } catch {
661
- emit(["ok": false, "kind": "schema",
662
- "error": "Apple rejected the response schema: \(error)"])
663
- return
664
- }
665
- }
666
- }
667
-
1073
+ let maxTokens = intNum(envelope["maxTokens"])
668
1074
  let options = GenerationOptions(
669
1075
  samplingMode: samplingFrom(envelope["sampling"]),
670
1076
  temperature: num(envelope["temperature"]),
671
- maximumResponseTokens: intNum(envelope["maxTokens"])
1077
+ maximumResponseTokens: maxTokens
672
1078
  )
673
1079
  let reuse = envelope["reuseSession"] as? Bool ?? false
674
- // api-scribe passed false because it spelled the schema out in its own
675
- // system prompt. A library cannot assume that, and a schema's `description`
1080
+ // A caller that spells the schema out in its own system prompt could pass
1081
+ // false. A library cannot assume that, and a schema's `description`
676
1082
  // fields are how the decoder gets its generation guidance, so default true.
677
1083
  let includeSchema = envelope["includeSchemaInPrompt"] as? Bool ?? true
1084
+ // Room left for the response when trimming history to fit.
1085
+ let reserve = min(maxTokens ?? 1024, model.contextSize / 2)
1086
+ let fingerprint = [useCase, guardrails, toolNames.sorted().joined(separator: ","),
1087
+ jsonText(envelope["functions"] ?? [Any]()), instructions].joined(separator: "|")
678
1088
 
679
- let session: LanguageModelSession
680
- // Cross-process continuity: a fresh native session (new process, or param
681
- // change) has no transcript, but the mirrored history survived on disk.
682
- // Reprise the recent turns as prompt context so `run --session` remembers
683
- // across CLI invocations too. Bounded to the last 10 turns; in-process
684
- // calls skip this and use the native transcript alone.
685
- var effectivePromptText = promptText
686
- if let sid = sessionId, !sid.isEmpty {
687
- let hadMemory = SessionStore.named[sid] != nil
688
- session = namedSessionFor(id: sid, model: model, instructions: instructions,
689
- useCase: useCase, guardrails: guardrails, tools: tools)
690
- if !hadMemory, let entry = SessionStore.named[sid], !entry.history.isEmpty {
691
- let recent = entry.history.suffix(10).map { turn -> String in
692
- let who = (turn["role"] == "user") ? "User" : "Assistant"
693
- return "\(who): \(turn["content"] ?? "")"
694
- }.joined(separator: "\n")
695
- effectivePromptText =
696
- "Previous conversation:\n\(recent)\n\nCurrent request:\n\(promptText)"
697
- }
698
- } else if tools.isEmpty {
699
- session = sessionFor(model: model, instructions: instructions, reuse: reuse)
700
- } else {
701
- session = LanguageModelSession(model: model, tools: tools, instructions: instructions)
702
- }
703
- let prompt = promptWith(text: effectivePromptText, imageSpecs: imageSpecs)
704
-
705
- if wantsStream {
706
- await handleStream(session: session, prompt: prompt, options: options,
707
- sessionId: sessionId, promptText: promptText)
708
- return
1089
+ var trimmed = 0
1090
+ func makeSession(rebuild: Bool) async -> LanguageModelSession {
1091
+ if let sid = sessionId, !sid.isEmpty {
1092
+ let (named, dropped) = await namedSessionFor(
1093
+ id: sid, model: model, instructions: instructions, fingerprint: fingerprint,
1094
+ tools: tools, prompt: prompt, reserve: reserve, rebuild: rebuild)
1095
+ trimmed += dropped
1096
+ return named
1097
+ }
1098
+ if !history.isEmpty {
1099
+ // Trimmed only on the retry after an overflow: counting tokens up
1100
+ // front would tax every call to learn, almost always, that it fits.
1101
+ var entries = history
1102
+ if rebuild {
1103
+ let (kept, dropped) = await fitted(history, model: model, instructions: instructions,
1104
+ tools: tools, prompt: prompt, reserve: reserve)
1105
+ entries = kept
1106
+ trimmed += dropped
1107
+ }
1108
+ return transcriptSession(model: model, instructions: instructions, tools: tools, entries: entries)
1109
+ }
1110
+ if tools.isEmpty { return sessionFor(model: model, instructions: instructions, reuse: reuse) }
1111
+ return LanguageModelSession(model: model, tools: tools, instructions: instructions)
709
1112
  }
710
1113
 
711
- do {
712
- let content: String
1114
+ var emittedEvents = false
1115
+ func generate(on session: LanguageModelSession) async throws -> Outcome {
1116
+ if wantsStream {
1117
+ return try await streamed(session: session, prompt: prompt, schema: schema,
1118
+ includeSchema: includeSchema, options: options,
1119
+ maxTokens: maxTokens, onEvent: { emittedEvents = true })
1120
+ }
713
1121
  if let schema {
714
1122
  let response = try await session.respond(
715
- to: prompt, schema: schema, includeSchemaInPrompt: includeSchema, options: options
716
- )
717
- content = response.content.jsonString
718
- } else {
719
- // Non-streaming. `session.streamResponse` is the streaming path;
720
- // see handleStream below.
721
- let response = try await session.respond(to: prompt, options: options)
722
- content = response.content
723
- }
724
- recordTurn(sessionId: sessionId, prompt: promptText, content: content,
725
- instructions: instructions)
726
- emit(["ok": true, "content": content])
1123
+ to: prompt, schema: schema, includeSchemaInPrompt: includeSchema, options: options)
1124
+ try Task.checkCancellation()
1125
+ var outcome = Outcome(content: response.content.jsonString)
1126
+ outcome.toolCalls = toolActivity(response.transcriptEntries)
1127
+ if #available(macOS 27.0, *) { outcome.usage = usageOf(response.usage) }
1128
+ outcome.finishReason = finishReason(usage: outcome.usage, maxTokens: maxTokens)
1129
+ return outcome
1130
+ }
1131
+ let response = try await session.respond(to: prompt, options: options)
1132
+ try Task.checkCancellation()
1133
+ var outcome = Outcome(content: response.content)
1134
+ outcome.toolCalls = toolActivity(response.transcriptEntries)
1135
+ if #available(macOS 27.0, *) { outcome.usage = usageOf(response.usage) }
1136
+ outcome.finishReason = finishReason(usage: outcome.usage, maxTokens: maxTokens)
1137
+ return outcome
1138
+ }
1139
+
1140
+ // A named session recovers from overflow on its own: it is stateful, so
1141
+ // without this one long turn would leave it permanently unusable.
1142
+ // Stateless history is trimmed only when the caller opted in.
1143
+ let canTrim = (sessionId?.isEmpty == false)
1144
+ || (!history.isEmpty && envelope["trimHistory"] as? Bool == true)
1145
+ var outcome: Outcome
1146
+ do {
1147
+ do {
1148
+ outcome = try await generate(on: await makeSession(rebuild: false))
1149
+ } catch let error where classify(error).kind == "context" && canTrim && !emittedEvents {
1150
+ // The conversation outgrew the window. Rebuild it from the newest
1151
+ // turns that still fit and try once more. Only when nothing has
1152
+ // been streamed yet: a retry after deltas would duplicate output.
1153
+ outcome = try await generate(on: await makeSession(rebuild: true))
1154
+ }
727
1155
  } catch {
728
- let (kind, resetDate) = classify(error)
729
- var payload: [String: Any] = ["ok": false, "kind": kind, "error": "\(error)"]
730
- if let resetDate { payload["resetDate"] = resetDate }
1156
+ if classify(error).kind == "stopped" {
1157
+ // The client is running a tool call itself; it saw the call as an
1158
+ // event and will continue through `history`.
1159
+ var payload: [String: Any] = ["ok": true, "content": "", "finishReason": "toolCalls"]
1160
+ if wantsStream { payload["done"] = true }
1161
+ if trimmed > 0 { payload["trimmedTurns"] = trimmed }
1162
+ emit(payload)
1163
+ return
1164
+ }
1165
+ var payload = failure(error)
1166
+ if trimmed > 0 { payload["trimmedTurns"] = trimmed }
731
1167
  emit(payload)
1168
+ return
732
1169
  }
1170
+
1171
+ recordTurn(sessionId: sessionId, prompt: promptText, content: outcome.content)
1172
+ var payload: [String: Any] = ["ok": true, "content": outcome.content,
1173
+ "finishReason": outcome.finishReason]
1174
+ if wantsStream { payload["done"] = true }
1175
+ if let usage = outcome.usage { payload["usage"] = usage }
1176
+ if !outcome.toolCalls.isEmpty { payload["toolCalls"] = outcome.toolCalls }
1177
+ if trimmed > 0 { payload["trimmedTurns"] = trimmed }
1178
+ emit(payload)
733
1179
  }
734
1180
 
735
1181
  @available(macOS 26.0, *)
736
- private func recordTurn(sessionId: String?, prompt: String, content: String,
737
- instructions: String) {
1182
+ private func recordTurn(sessionId: String?, prompt: String, content: String) {
738
1183
  guard let sid = sessionId, !sid.isEmpty else { return }
739
1184
  if var entry = SessionStore.named[sid] {
740
1185
  entry.history.append(["role": "user", "content": prompt])
@@ -749,42 +1194,66 @@ private func recordTurn(sessionId: String?, prompt: String, content: String,
749
1194
  }
750
1195
  }
751
1196
 
752
- /// Streaming generation. Each snapshot's `content` is the cumulative partial,
753
- /// so deltas are computed by stripping the previous prefix; when the model
754
- /// revises earlier text (rare for plain prose) the whole new partial is sent
755
- /// so the client never silently drops a correction.
1197
+ /// Streaming generation. Text snapshots carry the cumulative partial, so deltas
1198
+ /// are computed by stripping the previous prefix; when the model revises
1199
+ /// earlier text (rare for plain prose) the whole new partial is sent so the
1200
+ /// client never silently drops a correction.
1201
+ ///
1202
+ /// With a schema, each snapshot is the JSON generated so far, sent whole as
1203
+ /// `partial`: a partial object is a usable thing to render, where a raw text
1204
+ /// delta of JSON is not.
756
1205
  @available(macOS 26.0, *)
757
- private func handleStream(session: LanguageModelSession, prompt: Prompt,
758
- options: GenerationOptions, sessionId: String?,
759
- promptText: String) async {
760
- do {
761
- let stream = session.streamResponse(to: prompt, options: options)
762
- var previous = ""
763
- var full = ""
1206
+ private func streamed(
1207
+ session: LanguageModelSession, prompt: Prompt, schema: GenerationSchema?,
1208
+ includeSchema: Bool, options: GenerationOptions, maxTokens: Int?,
1209
+ onEvent: () -> Void
1210
+ ) async throws -> Outcome {
1211
+ if let schema {
1212
+ let stream = session.streamResponse(to: prompt, schema: schema,
1213
+ includeSchemaInPrompt: includeSchema, options: options)
1214
+ var last = ""
1215
+ var outcome = Outcome(content: "")
764
1216
  for try await snapshot in stream {
765
- let current: String = snapshot.content
766
- full = current
767
- let delta: String
768
- if current.hasPrefix(previous) {
769
- delta = String(current.dropFirst(previous.count))
770
- } else {
771
- delta = current
1217
+ try Task.checkCancellation()
1218
+ let json = snapshot.rawContent.jsonString
1219
+ if json != last {
1220
+ emit(["ok": true, "partial": json, "done": false])
1221
+ onEvent()
1222
+ last = json
772
1223
  }
773
- previous = current
774
- if !delta.isEmpty {
775
- emit(["ok": true, "delta": delta, "done": false])
1224
+ if #available(macOS 27.0, *) {
1225
+ outcome.usage = usageOf(snapshot.usage)
1226
+ outcome.toolCalls = toolActivity(snapshot.transcriptEntries)
776
1227
  }
777
1228
  }
778
- recordTurn(sessionId: sessionId, prompt: promptText, content: full,
779
- instructions: "")
780
- // Refresh persisted instructions for named sessions without clobbering.
781
- emit(["ok": true, "content": full, "done": true])
782
- } catch {
783
- let (kind, resetDate) = classify(error)
784
- var payload: [String: Any] = ["ok": false, "kind": kind, "error": "\(error)"]
785
- if let resetDate { payload["resetDate"] = resetDate }
786
- emit(payload)
1229
+ // A cancelled stream can end quietly rather than throw; without this the
1230
+ // truncated text would be reported as a normal, finished answer.
1231
+ try Task.checkCancellation()
1232
+ outcome.content = last
1233
+ outcome.finishReason = finishReason(usage: outcome.usage, maxTokens: maxTokens)
1234
+ return outcome
1235
+ }
1236
+ let stream = session.streamResponse(to: prompt, options: options)
1237
+ var previous = ""
1238
+ var outcome = Outcome(content: "")
1239
+ for try await snapshot in stream {
1240
+ try Task.checkCancellation()
1241
+ let current: String = snapshot.content
1242
+ let delta = current.hasPrefix(previous) ? String(current.dropFirst(previous.count)) : current
1243
+ previous = current
1244
+ if !delta.isEmpty {
1245
+ emit(["ok": true, "delta": delta, "done": false])
1246
+ onEvent()
1247
+ }
1248
+ if #available(macOS 27.0, *) {
1249
+ outcome.usage = usageOf(snapshot.usage)
1250
+ outcome.toolCalls = toolActivity(snapshot.transcriptEntries)
1251
+ }
787
1252
  }
1253
+ try Task.checkCancellation()
1254
+ outcome.content = previous
1255
+ outcome.finishReason = finishReason(usage: outcome.usage, maxTokens: maxTokens)
1256
+ return outcome
788
1257
  }
789
1258
 
790
1259
  private func describe(_ reason: SystemLanguageModel.Availability.UnavailableReason) -> String {
@@ -796,15 +1265,18 @@ private func describe(_ reason: SystemLanguageModel.Availability.UnavailableReas
796
1265
  }
797
1266
  }
798
1267
 
799
- /// Private Cloud Compute quota, read without calling it.
1268
+ /// Private Cloud Compute, described without calling it.
800
1269
  ///
801
1270
  /// PCC *inference* needs `com.apple.developer.private-cloud-compute`, which is
802
1271
  /// AMFI-restricted and unavailable to any installable package — that is why the
803
- /// cloud tier goes through Shortcuts. But `quotaUsage` is readable from an
804
- /// unentitled process, so a caller can see whether the cloud tier is worth
805
- /// trying before spending a Shortcuts round trip on it.
1272
+ /// cloud tier goes through Shortcuts. But the model's quota, capabilities and
1273
+ /// context size are all readable from an unentitled process, so a caller can
1274
+ /// see what the cloud tier offers, and whether it is worth trying, before
1275
+ /// spending a Shortcuts round trip on it. On macOS 27 Golden Gate this is the
1276
+ /// next-generation server model Siri AI is built on: it reports reasoning,
1277
+ /// vision and tool calling, with a 32,768-token window.
806
1278
  @available(macOS 27.0, *)
807
- private func cloudQuota() -> [String: Any] {
1279
+ private func cloudModel() async -> [String: Any] {
808
1280
  let pcc = PrivateCloudComputeLanguageModel()
809
1281
  var out: [String: Any] = ["isAvailable": pcc.isAvailable]
810
1282
  let usage = pcc.quotaUsage
@@ -820,6 +1292,14 @@ private func cloudQuota() -> [String: Any] {
820
1292
  if let reset = usage.resetDate {
821
1293
  out["resetDate"] = ISO8601DateFormatter().string(from: reset)
822
1294
  }
1295
+ let capabilities = pcc.capabilities
1296
+ out["capabilities"] = [
1297
+ "vision": capabilities.contains(.vision),
1298
+ "guidedGeneration": capabilities.contains(.guidedGeneration),
1299
+ "reasoning": capabilities.contains(.reasoning),
1300
+ "toolCalling": capabilities.contains(.toolCalling),
1301
+ ]
1302
+ if let contextSize = try? await pcc.contextSize { out["contextSize"] = contextSize }
823
1303
  return out
824
1304
  }
825
1305
 
@@ -850,41 +1330,54 @@ struct AppleLLMHelper {
850
1330
  "toolCalling": capabilities.contains(.toolCalling),
851
1331
  ]
852
1332
  payload["useCases"] = ["general", "contentTagging"]
853
- payload["cloud"] = cloudQuota()
854
- payload["features"] = [
855
- "streaming": true,
856
- "sessions": true,
857
- "history": true,
858
- "labelledAttachments": true,
859
- "builtInTools": ["ocr", "barcode", "spotlight"],
860
- ]
1333
+ payload["cloud"] = await cloudModel()
861
1334
  }
1335
+ payload["features"] = [
1336
+ "protocol": 2,
1337
+ "streaming": true,
1338
+ "structuredStreaming": true,
1339
+ "sessions": true,
1340
+ "history": true,
1341
+ "transcripts": true,
1342
+ "labelledAttachments": true,
1343
+ "functionTools": true,
1344
+ "cancellation": true,
1345
+ "builtInTools": ["ocr", "barcode", "spotlight"],
1346
+ ]
862
1347
  emit(payload)
863
1348
  return
864
1349
  }
865
1350
 
866
- var schemaCache: [String: GenerationSchema] = [:]
867
-
868
- // Serve mode: one request per stdin line, one response per stdout line,
869
- // for as long as the parent keeps the pipe open. Spawning a process per
870
- // request instead makes the model reload between calls, which measured
871
- // at ~17s per request against ~1.5s once it is resident.
1351
+ // Serve mode: one request per stdin line, for as long as the parent
1352
+ // keeps the pipe open. Spawning a process per request instead makes the
1353
+ // model reload between calls, which measured at ~17s per request
1354
+ // against ~1.5s once it is resident.
872
1355
  //
873
- // Streaming is the one exception to one-line-per-request: op "stream"
874
- // emits N {"delta","done":false} lines plus a final {"content",
875
- // "done":true}. The parent must keep reading until done:true before
876
- // sending the next request; the next stdin line is not consumed until
877
- // the stream completes, so replies cannot interleave.
1356
+ // stdin is read on its own thread so control lines (cancel, toolResult)
1357
+ // reach a request that is still running; see Control. Requests are
1358
+ // still handled strictly one at a time, in order.
878
1359
  if CommandLine.arguments.contains("--serve") {
879
- while let line = readLine(strippingNewline: true) {
880
- if line.isEmpty { continue }
1360
+ let requests = AsyncStream<String> { continuation in
1361
+ let reader = Thread {
1362
+ while let line = readLine(strippingNewline: true) {
1363
+ if line.isEmpty || Control.shared.intercept(line) { continue }
1364
+ continuation.yield(line)
1365
+ }
1366
+ Control.shared.stdinClosed()
1367
+ continuation.finish()
1368
+ }
1369
+ reader.start()
1370
+ }
1371
+ for await line in requests {
881
1372
  guard let data = line.data(using: .utf8),
882
1373
  let envelope = (try? JSONSerialization.jsonObject(with: data)) as? [String: Any]
883
1374
  else {
884
1375
  emit(["ok": false, "error": "helper could not parse a request line as JSON"])
885
1376
  continue
886
1377
  }
887
- await handle(envelope: envelope, schemaCache: &schemaCache)
1378
+ await Control.shared.run(id: envelope["id"] as? String) {
1379
+ await handle(envelope: envelope)
1380
+ }
888
1381
  }
889
1382
  return
890
1383
  }
@@ -894,7 +1387,7 @@ struct AppleLLMHelper {
894
1387
  emit(["ok": false, "error": "helper could not parse its stdin payload as JSON"])
895
1388
  return
896
1389
  }
897
-
898
- await handle(envelope: envelope, schemaCache: &schemaCache)
1390
+ Control.shared.stdinClosed()
1391
+ await handle(envelope: envelope)
899
1392
  }
900
1393
  }