apple-llm 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,900 @@
1
+ //
2
+ // apple-llm helper — the whole on-device path.
3
+ //
4
+ // Extracted from api-scribe (MIT, https://github.com/…/api-scribe), where it
5
+ // lived as a TypeScript string literal in `src/llm/apple-helper.ts`. This file
6
+ // is the single source of truth: `scripts/embed-helper.mjs` copies it verbatim
7
+ // into both the npm and the pip package, and a test in each asserts the copies
8
+ // still hash equal to this file.
9
+ //
10
+ // It needs no entitlements and no license agreement — unlike `/usr/bin/fm`,
11
+ // which ships with macOS 27 but is gated behind a machine-wide `sudo fm
12
+ // license`. Do not depend on `fm`.
13
+ //
14
+ // Protocol:
15
+ // helper --probe -> stdout: one JSON object describing this machine.
16
+ // helper --serve -> one request JSON per stdin line,
17
+ // one response JSON per stdout line, in order.
18
+ // EXCEPT op "stream": N delta lines + one final line,
19
+ // all belonging to that single request. The next request
20
+ // is not read until the stream completes, so ordering is
21
+ // preserved: the client must keep reading lines until
22
+ // "done":true before sending the next request.
23
+ // helper -> one request on stdin, one response on stdout
24
+ // ("stream" collapses to a single final response here,
25
+ // since one-shot stdout is not incremental).
26
+ //
27
+ // request: {"op"?:"generate"|"stream"|"countTokens"|"prewarm"|"history"|"reset",
28
+ // // default generate
29
+ // "instructions":string,"prompt":string,"schema"?:object|null,
30
+ // "temperature"?:number,"maxTokens"?:number,
31
+ // "includeSchemaInPrompt"?:bool,"reuseSession"?:bool,
32
+ // "sessionId"?:string, // multi-turn conversation
33
+ // "tools"?:["ocr"|"barcode"|"spotlight"], // built-in FM tools, 27+
34
+ // "images"?:[string|{"path":string,"label"?:string}],
35
+ // // file paths, macOS 27+
36
+ // "useCase"?:"general"|"contentTagging",
37
+ // "guardrails"?:"default"|"permissive",
38
+ // "sampling"?:{"mode":"greedy"}
39
+ // | {"mode":"topK","k":int,"seed"?:int}
40
+ // | {"mode":"threshold","p":number,"seed"?:int}}
41
+ // response: {"ok":true,"content":string} // generate
42
+ // | {"ok":true,"delta":string,"done":false} // stream partial
43
+ // | {"ok":true,"content":string,"done":true} // stream final
44
+ // | {"ok":true,"tokens":int} // countTokens
45
+ // | {"ok":true,"prewarmed":true} // prewarm
46
+ // | {"ok":true,"history":[{role,content}], // history
47
+ // "instructions":string,"sessionId":string}
48
+ // | {"ok":true,"reset":true} // reset
49
+ // | {"ok":false,"error":string,"kind"?:string}
50
+ //
51
+ // `kind` is one of availability | schema | context | quota | guardrail |
52
+ // timeout | unsupported | generation, so callers can raise typed errors
53
+ // instead of matching on strings. A quota error may carry `resetDate`.
54
+ //
55
+ // --probe reports availability, contextSize, variant, the model's
56
+ // capabilities (vision / guidedGeneration / reasoning / toolCalling) and the
57
+ // Private Cloud Compute quota status. That last one is readable without the
58
+ // `com.apple.developer.private-cloud-compute` entitlement that blocks PCC
59
+ // *inference*, so it is the one piece of first-party cloud state available
60
+ // to an unentitled process. It also reports `features` (streaming, sessions,
61
+ // labelledAttachments, builtInTools) so callers can fail fast with a message
62
+ // instead of a stale binary's silent misbehaviour.
63
+ //
64
+
65
+ import Foundation
66
+ import FoundationModels
67
+ #if canImport(_Vision_FoundationModels)
68
+ import _Vision_FoundationModels
69
+ #endif
70
+ #if canImport(_CoreSpotlight_FoundationModels)
71
+ import _CoreSpotlight_FoundationModels
72
+ #endif
73
+
74
+ private func emit(_ object: [String: Any]) {
75
+ guard let data = try? JSONSerialization.data(withJSONObject: object),
76
+ let text = String(data: data, encoding: .utf8) else {
77
+ print("{\"ok\":false,\"error\":\"helper could not encode its response\"}")
78
+ fflush(stdout)
79
+ return
80
+ }
81
+ // stdout is a pipe here, so it is fully buffered; the parent is waiting on
82
+ // this line and would otherwise see nothing until the process exits.
83
+ print(text)
84
+ fflush(stdout)
85
+ }
86
+
87
+ /// Optional session reuse, off by default.
88
+ ///
89
+ /// api-scribe kept one session alive while the instructions were unchanged, on
90
+ /// the finding that allocating a fresh LanguageModelSession per request made
91
+ /// every other call stall ~16s while assets cycled — a strict 17s/1.5s
92
+ /// alternation, uncorrelated with prompt size.
93
+ ///
94
+ /// That did not reproduce here. Measured on macOS 27 / M-series inside one
95
+ /// long-lived process, 6 calls per arm: session reused -> median 0.63s; a fresh
96
+ /// session every call -> median 0.64s, max 0.68s, no variance. The alternation
97
+ /// would have hit three times in six calls. The load-bearing part of the fix
98
+ /// appears to be the long-lived *process* (api-scribe measured ~17s a call when
99
+ /// it spawned a helper per request, against ~1.5s once resident), not the
100
+ /// shared session.
101
+ ///
102
+ /// So the default here is a fresh session per request, because a library cannot
103
+ /// let two unrelated calls share a transcript — api-scribe could, since every
104
+ /// one of its prompts carried its own whole context. The old behaviour is kept
105
+ /// behind `reuseSession` rather than deleted: the original finding was measured
106
+ /// too, possibly on macOS 26, and the doubt is worth preserving. If per-call
107
+ /// latency ever regresses to ~17s, try `reuseSession: true` first.
108
+ ///
109
+ /// Named sessions (`sessionId`) are the multi-turn counterpart: same process,
110
+ /// but the session is keyed by id so a conversation accumulates a native
111
+ /// transcript across calls. History is also mirrored to
112
+ /// `~/Library/Caches/apple-llm/sessions/<id>.json` so `history` survives a
113
+ /// helper restart; the native transcript does not — after a restart the next
114
+ /// call recreates the native session and continues from the stored history
115
+ /// length, which callers should treat as a context break, not a loss.
116
+ @available(macOS 26.0, *)
117
+ private final class SessionHolder {
118
+ static var session: LanguageModelSession?
119
+ static var instructions: String?
120
+ static var turns = 0
121
+ }
122
+
123
+ @available(macOS 26.0, *)
124
+ private struct NamedSession {
125
+ var session: LanguageModelSession
126
+ var instructions: String
127
+ var fingerprint: String
128
+ var history: [[String: String]]
129
+ }
130
+
131
+ @available(macOS 26.0, *)
132
+ private final class SessionStore {
133
+ static var named: [String: NamedSession] = [:]
134
+
135
+ static func sessionsDir() -> URL {
136
+ let home = FileManager.default.homeDirectoryForCurrentUser
137
+ return home
138
+ .appendingPathComponent("Library/Caches/apple-llm/sessions", isDirectory: true)
139
+ }
140
+
141
+ /// Filename-safe: Siri conversation titles can contain anything.
142
+ static func safeId(_ id: String) -> String {
143
+ let allowed = CharacterSet.alphanumerics.union(CharacterSet(charactersIn: "-_"))
144
+ let mapped = id.unicodeScalars.map { allowed.contains($0) ? String($0) : "_" }.joined()
145
+ let trimmed = String(mapped.prefix(64))
146
+ return trimmed.isEmpty ? "session" : trimmed
147
+ }
148
+
149
+ static func fileFor(_ id: String) -> URL {
150
+ sessionsDir().appendingPathComponent("\(safeId(id)).json")
151
+ }
152
+
153
+ static func loadHistory(_ id: String) -> (instructions: String?, history: [[String: String]]) {
154
+ let url = fileFor(id)
155
+ guard let data = try? Data(contentsOf: url),
156
+ let obj = (try? JSONSerialization.jsonObject(with: data)) as? [String: Any]
157
+ else { return (nil, []) }
158
+ let history = obj["history"] as? [[String: String]] ?? []
159
+ let instructions = obj["instructions"] as? String
160
+ return (instructions, history)
161
+ }
162
+
163
+ static func saveHistory(id: String, instructions: String, history: [[String: String]]) {
164
+ let url = fileFor(id)
165
+ try? FileManager.default.createDirectory(
166
+ at: url.deletingLastPathComponent(), withIntermediateDirectories: true)
167
+ let obj: [String: Any] = [
168
+ "version": 1, "sessionId": id,
169
+ "instructions": instructions, "history": history,
170
+ ]
171
+ if let data = try? JSONSerialization.data(withJSONObject: obj) {
172
+ try? data.write(to: url, options: .atomic)
173
+ }
174
+ }
175
+
176
+ static func removeFile(_ id: String) {
177
+ try? FileManager.default.removeItem(at: fileFor(id))
178
+ }
179
+ }
180
+
181
+ @available(macOS 26.0, *)
182
+ private func sessionFingerprint(useCase: String, guardrails: String, tools: [String],
183
+ instructions: String) -> String {
184
+ "\(useCase)|\(guardrails)|\(tools.sorted().joined(separator: ","))|\(instructions)"
185
+ }
186
+
187
+ @available(macOS 26.0, *)
188
+ private func sessionFor(
189
+ model: SystemLanguageModel, instructions: String, reuse: Bool
190
+ ) -> LanguageModelSession {
191
+ // Note a prewarmed session is deliberately *not* reused here. Doing so was
192
+ // tried and broke seeded reproducibility: the first call after a prewarm ran
193
+ // against the warmed session while later calls got fresh ones, so identical
194
+ // seeds produced different text. What prewarming actually buys is loading
195
+ // the model assets, and that is process-wide rather than session-bound, so
196
+ // discarding the session costs nothing and keeps every request identical.
197
+ guard reuse else { return LanguageModelSession(model: model, instructions: instructions) }
198
+ if let existing = SessionHolder.session,
199
+ SessionHolder.instructions == instructions,
200
+ SessionHolder.turns < 16 {
201
+ SessionHolder.turns += 1
202
+ return existing
203
+ }
204
+ let session = LanguageModelSession(model: model, instructions: instructions)
205
+ SessionHolder.session = session
206
+ SessionHolder.instructions = instructions
207
+ SessionHolder.turns = 1
208
+ return session
209
+ }
210
+
211
+ /// Named multi-turn session. Tools are fixed at session construction, so a
212
+ /// changed tool set recreates the native session; the mirrored history is kept
213
+ /// so nothing the caller said is silently dropped from `history`.
214
+ @available(macOS 26.0, *)
215
+ private func namedSessionFor(
216
+ id: String, model: SystemLanguageModel, instructions: String,
217
+ useCase: String, guardrails: String, tools: [any Tool]
218
+ ) -> LanguageModelSession {
219
+ let toolNames = currentToolNames()
220
+ let want = sessionFingerprint(useCase: useCase, guardrails: guardrails,
221
+ tools: toolNames, instructions: instructions)
222
+ if let existing = SessionStore.named[id], existing.fingerprint == want {
223
+ return existing.session
224
+ }
225
+ // Recreate (first use, param change, or restart). Restore mirrored history
226
+ // so `history` is continuous even though the native transcript restarts.
227
+ let stored = SessionStore.loadHistory(id)
228
+ let history = SessionStore.named[id]?.history ?? stored.history
229
+ let session: LanguageModelSession
230
+ if tools.isEmpty {
231
+ session = LanguageModelSession(model: model, instructions: instructions)
232
+ } else {
233
+ session = LanguageModelSession(model: model, tools: tools, instructions: instructions)
234
+ }
235
+ SessionStore.named[id] = NamedSession(
236
+ session: session, instructions: instructions,
237
+ fingerprint: want, history: history)
238
+ // Persist immediately so a fresh id is visible to `history` even before
239
+ // its first turn completes.
240
+ SessionStore.saveHistory(id: id, instructions: instructions, history: history)
241
+ return session
242
+ }
243
+
244
+ // The tool names for the in-flight request, stashed so namedSessionFor can
245
+ // fingerprint on them without threading another parameter through sessionFor.
246
+ @available(macOS 26.0, *)
247
+ private final class RequestContext {
248
+ static var toolNames: [String] = []
249
+ }
250
+
251
+ @available(macOS 26.0, *)
252
+ private func currentToolNames() -> [String] { RequestContext.toolNames }
253
+
254
+ /// Classify a generation failure so callers can raise a typed error rather than
255
+ /// matching on a message Apple may reword in any OS release.
256
+ ///
257
+ /// macOS 27 replaced `LanguageModelSession.GenerationError` with
258
+ /// `LanguageModelError`; both are consulted so one helper serves 26 and 27.
259
+ /// `rateLimited` carries a `resetDate`, which is what makes an on-device quota
260
+ /// error actionable rather than just a failure.
261
+ @available(macOS 26.0, *)
262
+ private func classify(_ error: Error) -> (kind: String, resetDate: String?) {
263
+ if #available(macOS 27.0, *) {
264
+ if let modern = error as? LanguageModelError {
265
+ switch modern {
266
+ case .contextSizeExceeded: return ("context", nil)
267
+ case .rateLimited(let info):
268
+ return ("quota", info.resetDate.map(ISO8601DateFormatter().string(from:)))
269
+ case .guardrailViolation, .refusal: return ("guardrail", nil)
270
+ case .timeout: return ("timeout", nil)
271
+ case .unsupportedCapability, .unsupportedGenerationGuide,
272
+ .unsupportedLanguageOrLocale, .unsupportedTranscriptContent:
273
+ return ("unsupported", nil)
274
+ @unknown default: return ("generation", nil)
275
+ }
276
+ }
277
+ }
278
+ return (legacyKind(error), nil)
279
+ }
280
+
281
+ /// macOS 26's error type, kept so one helper source serves both OS versions.
282
+ ///
283
+ /// Building on macOS 27 emits one deprecation warning at the call site below.
284
+ /// That is benign and self-correcting: the deployment target is derived from the
285
+ /// host SDK, so on a macOS 26 machine — the only place this branch is reachable —
286
+ /// the target is macos26.0 and nothing is deprecated yet. On macOS 27 the
287
+ /// warning points at code `#available` has already made unreachable.
288
+ @available(macOS 26.0, *)
289
+ @available(macOS, deprecated: 27.0)
290
+ private func legacyKind(_ error: Error) -> String {
291
+ if let legacy = error as? LanguageModelSession.GenerationError {
292
+ switch legacy {
293
+ case .exceededContextWindowSize: return "context"
294
+ case .guardrailViolation, .refusal: return "guardrail"
295
+ case .rateLimited: return "quota"
296
+ default: return "generation"
297
+ }
298
+ }
299
+ return "generation"
300
+ }
301
+
302
+ /// Models are keyed by use case and guardrails, because both are fixed at
303
+ /// construction. Building one is cheap; keeping them avoids re-resolving assets
304
+ /// when a caller alternates between, say, general and contentTagging.
305
+ @available(macOS 26.0, *)
306
+ private final class ModelCache {
307
+ static var models: [String: SystemLanguageModel] = [:]
308
+ }
309
+
310
+ /// `contentTagging` is a use case Apple ships specifically for tagging and
311
+ /// topic extraction; `permissive` relaxes the guardrails for content
312
+ /// *transformation* tasks (rewriting, summarising text that the default
313
+ /// guardrails would refuse). Both are macOS 26+.
314
+ @available(macOS 26.0, *)
315
+ private func modelFor(useCase: String, guardrails: String) -> SystemLanguageModel {
316
+ let key = "\(useCase)|\(guardrails)"
317
+ if let cached = ModelCache.models[key] { return cached }
318
+ let resolvedUseCase: SystemLanguageModel.UseCase =
319
+ useCase == "contentTagging" ? .contentTagging : .general
320
+ let resolvedGuardrails: SystemLanguageModel.Guardrails =
321
+ guardrails == "permissive" ? .permissiveContentTransformations : .default
322
+ let model = SystemLanguageModel(useCase: resolvedUseCase, guardrails: resolvedGuardrails)
323
+ ModelCache.models[key] = model
324
+ return model
325
+ }
326
+
327
+ /// Apple's built-in tools: on-device OCR and barcode reading (Vision) and the
328
+ /// Spotlight semantic index (local RAG). All macOS 27+. Unknown names are
329
+ /// rejected loudly — a silently ignored tool would mislead the caller into
330
+ /// thinking the model had a capability it did not.
331
+ @available(macOS 26.0, *)
332
+ private func toolsFor(_ names: [String]) throws -> [any Tool] {
333
+ var out: [any Tool] = []
334
+ for name in names {
335
+ switch name {
336
+ case "ocr":
337
+ if #available(macOS 27.0, *) {
338
+ #if canImport(_Vision_FoundationModels)
339
+ out.append(OCRTool())
340
+ #else
341
+ throw ToolError.unsupported("ocr needs macOS 27 with Vision tools")
342
+ #endif
343
+ } else {
344
+ throw ToolError.unsupported("ocr needs macOS 27 or later")
345
+ }
346
+ case "barcode":
347
+ if #available(macOS 27.0, *) {
348
+ #if canImport(_Vision_FoundationModels)
349
+ out.append(BarcodeReaderTool())
350
+ #else
351
+ throw ToolError.unsupported("barcode needs macOS 27 with Vision tools")
352
+ #endif
353
+ } else {
354
+ throw ToolError.unsupported("barcode needs macOS 27 or later")
355
+ }
356
+ case "spotlight":
357
+ if #available(macOS 27.0, *) {
358
+ #if canImport(_CoreSpotlight_FoundationModels)
359
+ out.append(SpotlightSearchTool())
360
+ #else
361
+ throw ToolError.unsupported("spotlight needs macOS 27 with Spotlight tools")
362
+ #endif
363
+ } else {
364
+ throw ToolError.unsupported("spotlight needs macOS 27 or later")
365
+ }
366
+ default:
367
+ throw ToolError.unknown("unknown tool \"\(name)\" (want ocr, barcode, spotlight)")
368
+ }
369
+ }
370
+ return out
371
+ }
372
+
373
+ private enum ToolError: Error {
374
+ case unknown(String)
375
+ case unsupported(String)
376
+ }
377
+
378
+ /// Coerce a JSON number regardless of int/double mismatch.
379
+ ///
380
+ /// `JSONSerialization` produces `NSNumber`, but `as? Double` fails for an
381
+ /// integer-valued `NSNumber` (and vice versa), so a caller passing
382
+ /// `temperature: 1` or `maxTokens: 100.0` would silently get `nil`.
383
+ private func num(_ v: Any?) -> Double? { (v as? NSNumber)?.doubleValue }
384
+
385
+ /// Coerce a JSON number to `Int`, returning `nil` for missing or out-of-range
386
+ /// values rather than trapping or wrapping.
387
+ private func intNum(_ v: Any?) -> Int? {
388
+ guard let n = v as? NSNumber else { return nil }
389
+ let d = n.doubleValue
390
+ guard d.isFinite, d >= Double(Int.min), d <= Double(Int.max) else { return nil }
391
+ return n.intValue
392
+ }
393
+
394
+ /// Coerce a JSON number to `UInt64` for seeds, which may exceed `Int.max`.
395
+ private func u64Num(_ v: Any?) -> UInt64? { (v as? NSNumber)?.uint64Value }
396
+
397
+ /// Sampling mode.
398
+ ///
399
+ /// `greedy` is deterministic but degenerates under guided generation — it padded
400
+ /// an unbounded array forever, then ran away inside a single string. The seeded
401
+ /// modes give the same determinism *without* that failure: `topK` with a seed
402
+ /// returned byte-identical output across three fresh sessions here. Note the
403
+ /// determinism depends on a fresh session per request, which is what this helper
404
+ /// does by default; reusing a session changes the transcript and with it the
405
+ /// output.
406
+ @available(macOS 26.0, *)
407
+ private func samplingFrom(_ value: Any?) -> GenerationOptions.SamplingMode? {
408
+ guard let spec = value as? [String: Any], let mode = spec["mode"] as? String else { return nil }
409
+ let seed = u64Num(spec["seed"])
410
+ switch mode {
411
+ case "greedy":
412
+ return .greedy
413
+ case "topK":
414
+ return .random(top: intNum(spec["k"]) ?? 50, seed: seed)
415
+ case "threshold":
416
+ return .random(probabilityThreshold: num(spec["p"]) ?? 0.9, seed: seed)
417
+ default:
418
+ return nil
419
+ }
420
+ }
421
+
422
+ @available(macOS 26.0, *)
423
+ private struct ImageSpec {
424
+ var path: String
425
+ var label: String
426
+ }
427
+
428
+ /// `images` accepts a plain path or `{"path":..,"label":..}`. Labels mirror
429
+ /// `fm --label`: they let the caller name attachments ("receipt", "chart")
430
+ /// so follow-up turns can refer to them. A missing file is an error, never a
431
+ /// silent drop.
432
+ @available(macOS 26.0, *)
433
+ private func imageSpecsFrom(_ value: Any?) -> [ImageSpec] {
434
+ guard let raw = value as? [Any] else { return [] }
435
+ var out: [ImageSpec] = []
436
+ for (index, item) in raw.enumerated() {
437
+ if let path = item as? String {
438
+ out.append(ImageSpec(path: path, label: "image\(index + 1)"))
439
+ } else if let dict = item as? [String: Any],
440
+ let path = dict["path"] as? String {
441
+ let label = (dict["label"] as? String)?.isEmpty == false
442
+ ? dict["label"] as! String : "image\(index + 1)"
443
+ out.append(ImageSpec(path: path, label: label))
444
+ }
445
+ }
446
+ return out
447
+ }
448
+
449
+ /// Build the prompt, attaching any images. Vision is macOS 27+; on 26 the paths
450
+ /// are reported as unsupported rather than silently dropped, because a caller
451
+ /// who asked about an image and got an answer that ignored it has been misled.
452
+ @available(macOS 26.0, *)
453
+ private func promptWith(text: String, imageSpecs: [ImageSpec]) -> Prompt {
454
+ guard !imageSpecs.isEmpty else { return Prompt(text) }
455
+ if #available(macOS 27.0, *) {
456
+ let attachments = imageSpecs.map { spec in
457
+ Attachment(imageURL: URL(fileURLWithPath: spec.path)).label(spec.label)
458
+ }
459
+ return Prompt {
460
+ for attachment in attachments { attachment }
461
+ text
462
+ }
463
+ }
464
+ return Prompt(text)
465
+ }
466
+
467
+ @available(macOS 26.0, *)
468
+ private func stringArray(_ value: Any?) -> [String] {
469
+ (value as? [Any] ?? []).compactMap { $0 as? String }
470
+ }
471
+
472
+ /// One request/response cycle. Kept separate from the transport so that
473
+ /// one-shot and serve modes cannot drift apart.
474
+ @available(macOS 26.0, *)
475
+ private func handle(
476
+ envelope: [String: Any],
477
+ schemaCache: inout [String: GenerationSchema]
478
+ ) async {
479
+ await handleEnvelope(envelope: envelope, schemaCache: &schemaCache, streaming: false)
480
+ }
481
+
482
+ @available(macOS 26.0, *)
483
+ private func handleEnvelope(
484
+ envelope: [String: Any],
485
+ schemaCache: inout [String: GenerationSchema],
486
+ streaming: Bool
487
+ ) async {
488
+ let useCase = envelope["useCase"] as? String ?? "general"
489
+ let guardrails = envelope["guardrails"] as? String ?? "default"
490
+ let model = modelFor(useCase: useCase, guardrails: guardrails)
491
+
492
+ switch model.availability {
493
+ case .available:
494
+ break
495
+ case .unavailable(let reason):
496
+ emit(["ok": false, "kind": "availability",
497
+ "error": "the on-device model is unavailable: \(describe(reason))"])
498
+ return
499
+ }
500
+
501
+ let instructions = envelope["instructions"] as? String ?? ""
502
+ let promptText = envelope["prompt"] as? String ?? ""
503
+ let imageSpecs = imageSpecsFrom(envelope["images"])
504
+ let toolNames = stringArray(envelope["tools"])
505
+ RequestContext.toolNames = toolNames
506
+ let sessionId = envelope["sessionId"] as? String
507
+
508
+ if !imageSpecs.isEmpty {
509
+ if #available(macOS 27.0, *) {
510
+ guard model.capabilities.contains(.vision) else {
511
+ emit(["ok": false, "kind": "unsupported",
512
+ "error": "this model does not support vision"])
513
+ return
514
+ }
515
+ for spec in imageSpecs where !FileManager.default.fileExists(atPath: spec.path) {
516
+ emit(["ok": false, "kind": "unsupported",
517
+ "error": "image not found: \(spec.path)"])
518
+ return
519
+ }
520
+ } else {
521
+ emit(["ok": false, "kind": "unsupported",
522
+ "error": "images need macOS 27 or later"])
523
+ return
524
+ }
525
+ }
526
+
527
+ let tools: [any Tool]
528
+ do {
529
+ tools = try toolsFor(toolNames)
530
+ } catch let err as ToolError {
531
+ switch err {
532
+ case .unknown(let m), .unsupported(let m):
533
+ emit(["ok": false, "kind": "unsupported", "error": m])
534
+ return
535
+ }
536
+ } catch {
537
+ emit(["ok": false, "kind": "unsupported", "error": "\(error)"])
538
+ return
539
+ }
540
+ if !tools.isEmpty {
541
+ guard model.capabilities.contains(.toolCalling) else {
542
+ emit(["ok": false, "kind": "unsupported",
543
+ "error": "this model does not support tool calling"])
544
+ return
545
+ }
546
+ }
547
+
548
+ // Token counting and prewarming share the model but not the generation path.
549
+ // An empty op defaults to generate; an unknown op is an error rather than
550
+ // silently running generate (which would hide a caller typo like
551
+ // "countToken").
552
+ let rawOp = envelope["op"] as? String
553
+ let op = (rawOp == nil || rawOp == "") ? "generate" : rawOp!
554
+ switch op {
555
+ case "generate", "stream":
556
+ break
557
+ case "countTokens":
558
+ guard #available(macOS 27.0, *) else {
559
+ emit(["ok": false, "kind": "unsupported",
560
+ "error": "counting tokens needs macOS 27 or later"])
561
+ return
562
+ }
563
+ do {
564
+ // Instructions and schema ride in the same window as the prompt, so
565
+ // a caller budgeting against contextSize needs all three counted.
566
+ var total = try await model.tokenCount(for: promptWith(text: promptText,
567
+ imageSpecs: imageSpecs))
568
+ if !instructions.isEmpty {
569
+ total += try await model.tokenCount(for: Instructions(instructions))
570
+ }
571
+ if !tools.isEmpty {
572
+ total += (try? await model.tokenCount(for: tools)) ?? 0
573
+ }
574
+ emit(["ok": true, "tokens": total, "contextSize": model.contextSize])
575
+ } catch {
576
+ emit(["ok": false, "kind": classify(error).kind, "error": "\(error)"])
577
+ }
578
+ return
579
+
580
+ case "prewarm":
581
+ // Loads model assets now so the first real call does not pay for it.
582
+ // Measured benefit is small once the assets are resident system-wide
583
+ // (0.31s vs 0.36s for an unprewarmed first call here); the win is on a
584
+ // genuinely cold system, where the first framework call took 7.8s.
585
+ // The session is discarded on purpose; see sessionFor above.
586
+ if tools.isEmpty {
587
+ LanguageModelSession(model: model, instructions: instructions).prewarm()
588
+ } else {
589
+ LanguageModelSession(model: model, tools: tools, instructions: instructions).prewarm()
590
+ }
591
+ emit(["ok": true, "prewarmed": true])
592
+ return
593
+
594
+ case "history":
595
+ guard let sid = sessionId, !sid.isEmpty else {
596
+ emit(["ok": false, "kind": "unsupported",
597
+ "error": "history needs a sessionId"])
598
+ return
599
+ }
600
+ if let entry = SessionStore.named[sid] {
601
+ emit(["ok": true, "sessionId": sid, "instructions": entry.instructions,
602
+ "history": entry.history])
603
+ } else {
604
+ let stored = SessionStore.loadHistory(sid)
605
+ emit(["ok": true, "sessionId": sid,
606
+ "instructions": stored.instructions ?? instructions,
607
+ "history": stored.history])
608
+ }
609
+ return
610
+
611
+ case "reset":
612
+ if let sid = sessionId, !sid.isEmpty {
613
+ SessionStore.named.removeValue(forKey: sid)
614
+ SessionStore.removeFile(sid)
615
+ } else {
616
+ SessionStore.named.removeAll()
617
+ }
618
+ emit(["ok": true, "reset": true])
619
+ return
620
+
621
+ default:
622
+ emit(["ok": false, "kind": "unsupported",
623
+ "error": "unsupported op \"\(op)\""])
624
+ return
625
+ }
626
+
627
+ // Streaming + guided generation do not mix in v1: the schema stream yields
628
+ // GeneratedContent snapshots whose partials are not plain-text deltas.
629
+ // Reject loudly rather than emitting misleading partial JSON.
630
+ let wantsStream = (op == "stream") || streaming
631
+ if wantsStream,
632
+ let schemaObject = envelope["schema"], !(schemaObject is NSNull) {
633
+ emit(["ok": false, "kind": "unsupported",
634
+ "error": "streaming with a schema is not supported; use generate for JSON"])
635
+ return
636
+ }
637
+
638
+ // No schema means a free-text call. api-scribe only ever asked for JSON and
639
+ // so required a schema on every request; `text()` needs the unconstrained
640
+ // `respond(to:options:)` overload instead.
641
+ var schema: GenerationSchema? = nil
642
+ if let schemaObject = envelope["schema"], !(schemaObject is NSNull) {
643
+ guard let schemaData = try? JSONSerialization.data(withJSONObject: schemaObject),
644
+ let schemaKey = String(data: schemaData, encoding: .utf8) else {
645
+ emit(["ok": false, "kind": "schema",
646
+ "error": "helper payload is missing a usable schema"])
647
+ return
648
+ }
649
+ // A changing GenerationSchema costs only ~0.15s per call, so per-request
650
+ // schemas are fine; this cache just makes a repeated one free.
651
+ if let cached = schemaCache[schemaKey] {
652
+ schema = cached
653
+ } else {
654
+ do {
655
+ let decoded = try JSONDecoder().decode(GenerationSchema.self, from: schemaData)
656
+ // Bound the cache: a long-lived server may see many schemas.
657
+ if schemaCache.count >= 64 { schemaCache.removeAll() }
658
+ schemaCache[schemaKey] = decoded
659
+ schema = decoded
660
+ } catch {
661
+ emit(["ok": false, "kind": "schema",
662
+ "error": "Apple rejected the response schema: \(error)"])
663
+ return
664
+ }
665
+ }
666
+ }
667
+
668
+ let options = GenerationOptions(
669
+ samplingMode: samplingFrom(envelope["sampling"]),
670
+ temperature: num(envelope["temperature"]),
671
+ maximumResponseTokens: intNum(envelope["maxTokens"])
672
+ )
673
+ let reuse = envelope["reuseSession"] as? Bool ?? false
674
+ // api-scribe passed false because it spelled the schema out in its own
675
+ // system prompt. A library cannot assume that, and a schema's `description`
676
+ // fields are how the decoder gets its generation guidance, so default true.
677
+ let includeSchema = envelope["includeSchemaInPrompt"] as? Bool ?? true
678
+
679
+ let session: LanguageModelSession
680
+ // Cross-process continuity: a fresh native session (new process, or param
681
+ // change) has no transcript, but the mirrored history survived on disk.
682
+ // Reprise the recent turns as prompt context so `run --session` remembers
683
+ // across CLI invocations too. Bounded to the last 10 turns; in-process
684
+ // calls skip this and use the native transcript alone.
685
+ var effectivePromptText = promptText
686
+ if let sid = sessionId, !sid.isEmpty {
687
+ let hadMemory = SessionStore.named[sid] != nil
688
+ session = namedSessionFor(id: sid, model: model, instructions: instructions,
689
+ useCase: useCase, guardrails: guardrails, tools: tools)
690
+ if !hadMemory, let entry = SessionStore.named[sid], !entry.history.isEmpty {
691
+ let recent = entry.history.suffix(10).map { turn -> String in
692
+ let who = (turn["role"] == "user") ? "User" : "Assistant"
693
+ return "\(who): \(turn["content"] ?? "")"
694
+ }.joined(separator: "\n")
695
+ effectivePromptText =
696
+ "Previous conversation:\n\(recent)\n\nCurrent request:\n\(promptText)"
697
+ }
698
+ } else if tools.isEmpty {
699
+ session = sessionFor(model: model, instructions: instructions, reuse: reuse)
700
+ } else {
701
+ session = LanguageModelSession(model: model, tools: tools, instructions: instructions)
702
+ }
703
+ let prompt = promptWith(text: effectivePromptText, imageSpecs: imageSpecs)
704
+
705
+ if wantsStream {
706
+ await handleStream(session: session, prompt: prompt, options: options,
707
+ sessionId: sessionId, promptText: promptText)
708
+ return
709
+ }
710
+
711
+ do {
712
+ let content: String
713
+ if let schema {
714
+ let response = try await session.respond(
715
+ to: prompt, schema: schema, includeSchemaInPrompt: includeSchema, options: options
716
+ )
717
+ content = response.content.jsonString
718
+ } else {
719
+ // Non-streaming. `session.streamResponse` is the streaming path;
720
+ // see handleStream below.
721
+ let response = try await session.respond(to: prompt, options: options)
722
+ content = response.content
723
+ }
724
+ recordTurn(sessionId: sessionId, prompt: promptText, content: content,
725
+ instructions: instructions)
726
+ emit(["ok": true, "content": content])
727
+ } catch {
728
+ let (kind, resetDate) = classify(error)
729
+ var payload: [String: Any] = ["ok": false, "kind": kind, "error": "\(error)"]
730
+ if let resetDate { payload["resetDate"] = resetDate }
731
+ emit(payload)
732
+ }
733
+ }
734
+
735
+ @available(macOS 26.0, *)
736
+ private func recordTurn(sessionId: String?, prompt: String, content: String,
737
+ instructions: String) {
738
+ guard let sid = sessionId, !sid.isEmpty else { return }
739
+ if var entry = SessionStore.named[sid] {
740
+ entry.history.append(["role": "user", "content": prompt])
741
+ entry.history.append(["role": "assistant", "content": content])
742
+ // Bound mirrored history: native transcript still holds the full
743
+ // context for this process lifetime; the mirror is for `history` and
744
+ // restart continuity, not inference.
745
+ if entry.history.count > 200 { entry.history.removeFirst(entry.history.count - 200) }
746
+ SessionStore.named[sid] = entry
747
+ SessionStore.saveHistory(id: sid, instructions: entry.instructions,
748
+ history: entry.history)
749
+ }
750
+ }
751
+
752
+ /// Streaming generation. Each snapshot's `content` is the cumulative partial,
753
+ /// so deltas are computed by stripping the previous prefix; when the model
754
+ /// revises earlier text (rare for plain prose) the whole new partial is sent
755
+ /// so the client never silently drops a correction.
756
+ @available(macOS 26.0, *)
757
+ private func handleStream(session: LanguageModelSession, prompt: Prompt,
758
+ options: GenerationOptions, sessionId: String?,
759
+ promptText: String) async {
760
+ do {
761
+ let stream = session.streamResponse(to: prompt, options: options)
762
+ var previous = ""
763
+ var full = ""
764
+ for try await snapshot in stream {
765
+ let current: String = snapshot.content
766
+ full = current
767
+ let delta: String
768
+ if current.hasPrefix(previous) {
769
+ delta = String(current.dropFirst(previous.count))
770
+ } else {
771
+ delta = current
772
+ }
773
+ previous = current
774
+ if !delta.isEmpty {
775
+ emit(["ok": true, "delta": delta, "done": false])
776
+ }
777
+ }
778
+ recordTurn(sessionId: sessionId, prompt: promptText, content: full,
779
+ instructions: "")
780
+ // Refresh persisted instructions for named sessions without clobbering.
781
+ emit(["ok": true, "content": full, "done": true])
782
+ } catch {
783
+ let (kind, resetDate) = classify(error)
784
+ var payload: [String: Any] = ["ok": false, "kind": kind, "error": "\(error)"]
785
+ if let resetDate { payload["resetDate"] = resetDate }
786
+ emit(payload)
787
+ }
788
+ }
789
+
790
+ private func describe(_ reason: SystemLanguageModel.Availability.UnavailableReason) -> String {
791
+ switch reason {
792
+ case .deviceNotEligible: return "deviceNotEligible"
793
+ case .appleIntelligenceNotEnabled: return "appleIntelligenceNotEnabled"
794
+ case .modelNotReady: return "modelNotReady"
795
+ @unknown default: return "unknown"
796
+ }
797
+ }
798
+
799
+ /// Private Cloud Compute quota, read without calling it.
800
+ ///
801
+ /// PCC *inference* needs `com.apple.developer.private-cloud-compute`, which is
802
+ /// AMFI-restricted and unavailable to any installable package — that is why the
803
+ /// cloud tier goes through Shortcuts. But `quotaUsage` is readable from an
804
+ /// unentitled process, so a caller can see whether the cloud tier is worth
805
+ /// trying before spending a Shortcuts round trip on it.
806
+ @available(macOS 27.0, *)
807
+ private func cloudQuota() -> [String: Any] {
808
+ let pcc = PrivateCloudComputeLanguageModel()
809
+ var out: [String: Any] = ["isAvailable": pcc.isAvailable]
810
+ let usage = pcc.quotaUsage
811
+ switch usage.status {
812
+ case .belowLimit(let below):
813
+ out["status"] = "belowLimit"
814
+ out["approachingLimit"] = below.isApproachingLimit
815
+ case .limitReached:
816
+ out["status"] = "limitReached"
817
+ @unknown default:
818
+ out["status"] = "unknown"
819
+ }
820
+ if let reset = usage.resetDate {
821
+ out["resetDate"] = ISO8601DateFormatter().string(from: reset)
822
+ }
823
+ return out
824
+ }
825
+
826
+ @main
827
+ struct AppleLLMHelper {
828
+ static func main() async {
829
+ let model = SystemLanguageModel.default
830
+
831
+ if CommandLine.arguments.contains("--probe") {
832
+ var payload: [String: Any] = ["contextSize": model.contextSize]
833
+ switch model.availability {
834
+ case .available:
835
+ payload["available"] = true
836
+ case .unavailable(let reason):
837
+ payload["available"] = false
838
+ payload["reason"] = describe(reason)
839
+ }
840
+ if #available(macOS 27.0, *) {
841
+ payload["variant"] = model.variant.displayName
842
+ // What the model can actually do, rather than what the docs
843
+ // imply. On this machine the on-device model reports vision and
844
+ // tool calling but *not* reasoning, while PCC reports all four.
845
+ let capabilities = model.capabilities
846
+ payload["capabilities"] = [
847
+ "vision": capabilities.contains(.vision),
848
+ "guidedGeneration": capabilities.contains(.guidedGeneration),
849
+ "reasoning": capabilities.contains(.reasoning),
850
+ "toolCalling": capabilities.contains(.toolCalling),
851
+ ]
852
+ payload["useCases"] = ["general", "contentTagging"]
853
+ payload["cloud"] = cloudQuota()
854
+ payload["features"] = [
855
+ "streaming": true,
856
+ "sessions": true,
857
+ "history": true,
858
+ "labelledAttachments": true,
859
+ "builtInTools": ["ocr", "barcode", "spotlight"],
860
+ ]
861
+ }
862
+ emit(payload)
863
+ return
864
+ }
865
+
866
+ var schemaCache: [String: GenerationSchema] = [:]
867
+
868
+ // Serve mode: one request per stdin line, one response per stdout line,
869
+ // for as long as the parent keeps the pipe open. Spawning a process per
870
+ // request instead makes the model reload between calls, which measured
871
+ // at ~17s per request against ~1.5s once it is resident.
872
+ //
873
+ // Streaming is the one exception to one-line-per-request: op "stream"
874
+ // emits N {"delta","done":false} lines plus a final {"content",
875
+ // "done":true}. The parent must keep reading until done:true before
876
+ // sending the next request; the next stdin line is not consumed until
877
+ // the stream completes, so replies cannot interleave.
878
+ if CommandLine.arguments.contains("--serve") {
879
+ while let line = readLine(strippingNewline: true) {
880
+ if line.isEmpty { continue }
881
+ guard let data = line.data(using: .utf8),
882
+ let envelope = (try? JSONSerialization.jsonObject(with: data)) as? [String: Any]
883
+ else {
884
+ emit(["ok": false, "error": "helper could not parse a request line as JSON"])
885
+ continue
886
+ }
887
+ await handle(envelope: envelope, schemaCache: &schemaCache)
888
+ }
889
+ return
890
+ }
891
+
892
+ let input = FileHandle.standardInput.readDataToEndOfFile()
893
+ guard let envelope = (try? JSONSerialization.jsonObject(with: input)) as? [String: Any] else {
894
+ emit(["ok": false, "error": "helper could not parse its stdin payload as JSON"])
895
+ return
896
+ }
897
+
898
+ await handle(envelope: envelope, schemaCache: &schemaCache)
899
+ }
900
+ }