apple-llm 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +176 -0
- package/dist/chunk-FQTRQ3KP.js +1590 -0
- package/dist/cli.cjs +2027 -0
- package/dist/cli.js +456 -0
- package/dist/index.cjs +1667 -0
- package/dist/index.d.cts +690 -0
- package/dist/index.d.ts +690 -0
- package/dist/index.js +80 -0
- package/package.json +61 -0
- package/swift/helper.swift +900 -0
|
@@ -0,0 +1,900 @@
|
|
|
1
|
+
//
|
|
2
|
+
// apple-llm helper — the whole on-device path.
|
|
3
|
+
//
|
|
4
|
+
// Extracted from api-scribe (MIT, https://github.com/…/api-scribe), where it
|
|
5
|
+
// lived as a TypeScript string literal in `src/llm/apple-helper.ts`. This file
|
|
6
|
+
// is the single source of truth: `scripts/embed-helper.mjs` copies it verbatim
|
|
7
|
+
// into both the npm and the pip package, and a test in each asserts the copies
|
|
8
|
+
// still hash equal to this file.
|
|
9
|
+
//
|
|
10
|
+
// It needs no entitlements and no license agreement — unlike `/usr/bin/fm`,
|
|
11
|
+
// which ships with macOS 27 but is gated behind a machine-wide `sudo fm
|
|
12
|
+
// license`. Do not depend on `fm`.
|
|
13
|
+
//
|
|
14
|
+
// Protocol:
|
|
15
|
+
// helper --probe -> stdout: one JSON object describing this machine.
|
|
16
|
+
// helper --serve -> one request JSON per stdin line,
|
|
17
|
+
// one response JSON per stdout line, in order.
|
|
18
|
+
// EXCEPT op "stream": N delta lines + one final line,
|
|
19
|
+
// all belonging to that single request. The next request
|
|
20
|
+
// is not read until the stream completes, so ordering is
|
|
21
|
+
// preserved: the client must keep reading lines until
|
|
22
|
+
// "done":true before sending the next request.
|
|
23
|
+
// helper -> one request on stdin, one response on stdout
|
|
24
|
+
// ("stream" collapses to a single final response here,
|
|
25
|
+
// since one-shot stdout is not incremental).
|
|
26
|
+
//
|
|
27
|
+
// request: {"op"?:"generate"|"stream"|"countTokens"|"prewarm"|"history"|"reset",
|
|
28
|
+
// // default generate
|
|
29
|
+
// "instructions":string,"prompt":string,"schema"?:object|null,
|
|
30
|
+
// "temperature"?:number,"maxTokens"?:number,
|
|
31
|
+
// "includeSchemaInPrompt"?:bool,"reuseSession"?:bool,
|
|
32
|
+
// "sessionId"?:string, // multi-turn conversation
|
|
33
|
+
// "tools"?:["ocr"|"barcode"|"spotlight"], // built-in FM tools, 27+
|
|
34
|
+
// "images"?:[string|{"path":string,"label"?:string}],
|
|
35
|
+
// // file paths, macOS 27+
|
|
36
|
+
// "useCase"?:"general"|"contentTagging",
|
|
37
|
+
// "guardrails"?:"default"|"permissive",
|
|
38
|
+
// "sampling"?:{"mode":"greedy"}
|
|
39
|
+
// | {"mode":"topK","k":int,"seed"?:int}
|
|
40
|
+
// | {"mode":"threshold","p":number,"seed"?:int}}
|
|
41
|
+
// response: {"ok":true,"content":string} // generate
|
|
42
|
+
// | {"ok":true,"delta":string,"done":false} // stream partial
|
|
43
|
+
// | {"ok":true,"content":string,"done":true} // stream final
|
|
44
|
+
// | {"ok":true,"tokens":int} // countTokens
|
|
45
|
+
// | {"ok":true,"prewarmed":true} // prewarm
|
|
46
|
+
// | {"ok":true,"history":[{role,content}], // history
|
|
47
|
+
// "instructions":string,"sessionId":string}
|
|
48
|
+
// | {"ok":true,"reset":true} // reset
|
|
49
|
+
// | {"ok":false,"error":string,"kind"?:string}
|
|
50
|
+
//
|
|
51
|
+
// `kind` is one of availability | schema | context | quota | guardrail |
|
|
52
|
+
// timeout | unsupported | generation, so callers can raise typed errors
|
|
53
|
+
// instead of matching on strings. A quota error may carry `resetDate`.
|
|
54
|
+
//
|
|
55
|
+
// --probe reports availability, contextSize, variant, the model's
|
|
56
|
+
// capabilities (vision / guidedGeneration / reasoning / toolCalling) and the
|
|
57
|
+
// Private Cloud Compute quota status. That last one is readable without the
|
|
58
|
+
// `com.apple.developer.private-cloud-compute` entitlement that blocks PCC
|
|
59
|
+
// *inference*, so it is the one piece of first-party cloud state available
|
|
60
|
+
// to an unentitled process. It also reports `features` (streaming, sessions,
|
|
61
|
+
// labelledAttachments, builtInTools) so callers can fail fast with a message
|
|
62
|
+
// instead of a stale binary's silent misbehaviour.
|
|
63
|
+
//
|
|
64
|
+
|
|
65
|
+
import Foundation
|
|
66
|
+
import FoundationModels
|
|
67
|
+
#if canImport(_Vision_FoundationModels)
|
|
68
|
+
import _Vision_FoundationModels
|
|
69
|
+
#endif
|
|
70
|
+
#if canImport(_CoreSpotlight_FoundationModels)
|
|
71
|
+
import _CoreSpotlight_FoundationModels
|
|
72
|
+
#endif
|
|
73
|
+
|
|
74
|
+
private func emit(_ object: [String: Any]) {
|
|
75
|
+
guard let data = try? JSONSerialization.data(withJSONObject: object),
|
|
76
|
+
let text = String(data: data, encoding: .utf8) else {
|
|
77
|
+
print("{\"ok\":false,\"error\":\"helper could not encode its response\"}")
|
|
78
|
+
fflush(stdout)
|
|
79
|
+
return
|
|
80
|
+
}
|
|
81
|
+
// stdout is a pipe here, so it is fully buffered; the parent is waiting on
|
|
82
|
+
// this line and would otherwise see nothing until the process exits.
|
|
83
|
+
print(text)
|
|
84
|
+
fflush(stdout)
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/// Optional session reuse, off by default.
|
|
88
|
+
///
|
|
89
|
+
/// api-scribe kept one session alive while the instructions were unchanged, on
|
|
90
|
+
/// the finding that allocating a fresh LanguageModelSession per request made
|
|
91
|
+
/// every other call stall ~16s while assets cycled — a strict 17s/1.5s
|
|
92
|
+
/// alternation, uncorrelated with prompt size.
|
|
93
|
+
///
|
|
94
|
+
/// That did not reproduce here. Measured on macOS 27 / M-series inside one
|
|
95
|
+
/// long-lived process, 6 calls per arm: session reused -> median 0.63s; a fresh
|
|
96
|
+
/// session every call -> median 0.64s, max 0.68s, no variance. The alternation
|
|
97
|
+
/// would have hit three times in six calls. The load-bearing part of the fix
|
|
98
|
+
/// appears to be the long-lived *process* (api-scribe measured ~17s a call when
|
|
99
|
+
/// it spawned a helper per request, against ~1.5s once resident), not the
|
|
100
|
+
/// shared session.
|
|
101
|
+
///
|
|
102
|
+
/// So the default here is a fresh session per request, because a library cannot
|
|
103
|
+
/// let two unrelated calls share a transcript — api-scribe could, since every
|
|
104
|
+
/// one of its prompts carried its own whole context. The old behaviour is kept
|
|
105
|
+
/// behind `reuseSession` rather than deleted: the original finding was measured
|
|
106
|
+
/// too, possibly on macOS 26, and the doubt is worth preserving. If per-call
|
|
107
|
+
/// latency ever regresses to ~17s, try `reuseSession: true` first.
|
|
108
|
+
///
|
|
109
|
+
/// Named sessions (`sessionId`) are the multi-turn counterpart: same process,
|
|
110
|
+
/// but the session is keyed by id so a conversation accumulates a native
|
|
111
|
+
/// transcript across calls. History is also mirrored to
|
|
112
|
+
/// `~/Library/Caches/apple-llm/sessions/<id>.json` so `history` survives a
|
|
113
|
+
/// helper restart; the native transcript does not — after a restart the next
|
|
114
|
+
/// call recreates the native session and continues from the stored history
|
|
115
|
+
/// length, which callers should treat as a context break, not a loss.
|
|
116
|
+
@available(macOS 26.0, *)
|
|
117
|
+
private final class SessionHolder {
|
|
118
|
+
static var session: LanguageModelSession?
|
|
119
|
+
static var instructions: String?
|
|
120
|
+
static var turns = 0
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
@available(macOS 26.0, *)
|
|
124
|
+
private struct NamedSession {
|
|
125
|
+
var session: LanguageModelSession
|
|
126
|
+
var instructions: String
|
|
127
|
+
var fingerprint: String
|
|
128
|
+
var history: [[String: String]]
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
@available(macOS 26.0, *)
|
|
132
|
+
private final class SessionStore {
|
|
133
|
+
static var named: [String: NamedSession] = [:]
|
|
134
|
+
|
|
135
|
+
static func sessionsDir() -> URL {
|
|
136
|
+
let home = FileManager.default.homeDirectoryForCurrentUser
|
|
137
|
+
return home
|
|
138
|
+
.appendingPathComponent("Library/Caches/apple-llm/sessions", isDirectory: true)
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/// Filename-safe: Siri conversation titles can contain anything.
|
|
142
|
+
static func safeId(_ id: String) -> String {
|
|
143
|
+
let allowed = CharacterSet.alphanumerics.union(CharacterSet(charactersIn: "-_"))
|
|
144
|
+
let mapped = id.unicodeScalars.map { allowed.contains($0) ? String($0) : "_" }.joined()
|
|
145
|
+
let trimmed = String(mapped.prefix(64))
|
|
146
|
+
return trimmed.isEmpty ? "session" : trimmed
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
static func fileFor(_ id: String) -> URL {
|
|
150
|
+
sessionsDir().appendingPathComponent("\(safeId(id)).json")
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
static func loadHistory(_ id: String) -> (instructions: String?, history: [[String: String]]) {
|
|
154
|
+
let url = fileFor(id)
|
|
155
|
+
guard let data = try? Data(contentsOf: url),
|
|
156
|
+
let obj = (try? JSONSerialization.jsonObject(with: data)) as? [String: Any]
|
|
157
|
+
else { return (nil, []) }
|
|
158
|
+
let history = obj["history"] as? [[String: String]] ?? []
|
|
159
|
+
let instructions = obj["instructions"] as? String
|
|
160
|
+
return (instructions, history)
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
static func saveHistory(id: String, instructions: String, history: [[String: String]]) {
|
|
164
|
+
let url = fileFor(id)
|
|
165
|
+
try? FileManager.default.createDirectory(
|
|
166
|
+
at: url.deletingLastPathComponent(), withIntermediateDirectories: true)
|
|
167
|
+
let obj: [String: Any] = [
|
|
168
|
+
"version": 1, "sessionId": id,
|
|
169
|
+
"instructions": instructions, "history": history,
|
|
170
|
+
]
|
|
171
|
+
if let data = try? JSONSerialization.data(withJSONObject: obj) {
|
|
172
|
+
try? data.write(to: url, options: .atomic)
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
static func removeFile(_ id: String) {
|
|
177
|
+
try? FileManager.default.removeItem(at: fileFor(id))
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
@available(macOS 26.0, *)
|
|
182
|
+
private func sessionFingerprint(useCase: String, guardrails: String, tools: [String],
|
|
183
|
+
instructions: String) -> String {
|
|
184
|
+
"\(useCase)|\(guardrails)|\(tools.sorted().joined(separator: ","))|\(instructions)"
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
@available(macOS 26.0, *)
|
|
188
|
+
private func sessionFor(
|
|
189
|
+
model: SystemLanguageModel, instructions: String, reuse: Bool
|
|
190
|
+
) -> LanguageModelSession {
|
|
191
|
+
// Note a prewarmed session is deliberately *not* reused here. Doing so was
|
|
192
|
+
// tried and broke seeded reproducibility: the first call after a prewarm ran
|
|
193
|
+
// against the warmed session while later calls got fresh ones, so identical
|
|
194
|
+
// seeds produced different text. What prewarming actually buys is loading
|
|
195
|
+
// the model assets, and that is process-wide rather than session-bound, so
|
|
196
|
+
// discarding the session costs nothing and keeps every request identical.
|
|
197
|
+
guard reuse else { return LanguageModelSession(model: model, instructions: instructions) }
|
|
198
|
+
if let existing = SessionHolder.session,
|
|
199
|
+
SessionHolder.instructions == instructions,
|
|
200
|
+
SessionHolder.turns < 16 {
|
|
201
|
+
SessionHolder.turns += 1
|
|
202
|
+
return existing
|
|
203
|
+
}
|
|
204
|
+
let session = LanguageModelSession(model: model, instructions: instructions)
|
|
205
|
+
SessionHolder.session = session
|
|
206
|
+
SessionHolder.instructions = instructions
|
|
207
|
+
SessionHolder.turns = 1
|
|
208
|
+
return session
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
/// Named multi-turn session. Tools are fixed at session construction, so a
|
|
212
|
+
/// changed tool set recreates the native session; the mirrored history is kept
|
|
213
|
+
/// so nothing the caller said is silently dropped from `history`.
|
|
214
|
+
@available(macOS 26.0, *)
|
|
215
|
+
private func namedSessionFor(
|
|
216
|
+
id: String, model: SystemLanguageModel, instructions: String,
|
|
217
|
+
useCase: String, guardrails: String, tools: [any Tool]
|
|
218
|
+
) -> LanguageModelSession {
|
|
219
|
+
let toolNames = currentToolNames()
|
|
220
|
+
let want = sessionFingerprint(useCase: useCase, guardrails: guardrails,
|
|
221
|
+
tools: toolNames, instructions: instructions)
|
|
222
|
+
if let existing = SessionStore.named[id], existing.fingerprint == want {
|
|
223
|
+
return existing.session
|
|
224
|
+
}
|
|
225
|
+
// Recreate (first use, param change, or restart). Restore mirrored history
|
|
226
|
+
// so `history` is continuous even though the native transcript restarts.
|
|
227
|
+
let stored = SessionStore.loadHistory(id)
|
|
228
|
+
let history = SessionStore.named[id]?.history ?? stored.history
|
|
229
|
+
let session: LanguageModelSession
|
|
230
|
+
if tools.isEmpty {
|
|
231
|
+
session = LanguageModelSession(model: model, instructions: instructions)
|
|
232
|
+
} else {
|
|
233
|
+
session = LanguageModelSession(model: model, tools: tools, instructions: instructions)
|
|
234
|
+
}
|
|
235
|
+
SessionStore.named[id] = NamedSession(
|
|
236
|
+
session: session, instructions: instructions,
|
|
237
|
+
fingerprint: want, history: history)
|
|
238
|
+
// Persist immediately so a fresh id is visible to `history` even before
|
|
239
|
+
// its first turn completes.
|
|
240
|
+
SessionStore.saveHistory(id: id, instructions: instructions, history: history)
|
|
241
|
+
return session
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
// The tool names for the in-flight request, stashed so namedSessionFor can
|
|
245
|
+
// fingerprint on them without threading another parameter through sessionFor.
|
|
246
|
+
@available(macOS 26.0, *)
|
|
247
|
+
private final class RequestContext {
|
|
248
|
+
static var toolNames: [String] = []
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
@available(macOS 26.0, *)
|
|
252
|
+
private func currentToolNames() -> [String] { RequestContext.toolNames }
|
|
253
|
+
|
|
254
|
+
/// Classify a generation failure so callers can raise a typed error rather than
|
|
255
|
+
/// matching on a message Apple may reword in any OS release.
|
|
256
|
+
///
|
|
257
|
+
/// macOS 27 replaced `LanguageModelSession.GenerationError` with
|
|
258
|
+
/// `LanguageModelError`; both are consulted so one helper serves 26 and 27.
|
|
259
|
+
/// `rateLimited` carries a `resetDate`, which is what makes an on-device quota
|
|
260
|
+
/// error actionable rather than just a failure.
|
|
261
|
+
@available(macOS 26.0, *)
|
|
262
|
+
private func classify(_ error: Error) -> (kind: String, resetDate: String?) {
|
|
263
|
+
if #available(macOS 27.0, *) {
|
|
264
|
+
if let modern = error as? LanguageModelError {
|
|
265
|
+
switch modern {
|
|
266
|
+
case .contextSizeExceeded: return ("context", nil)
|
|
267
|
+
case .rateLimited(let info):
|
|
268
|
+
return ("quota", info.resetDate.map(ISO8601DateFormatter().string(from:)))
|
|
269
|
+
case .guardrailViolation, .refusal: return ("guardrail", nil)
|
|
270
|
+
case .timeout: return ("timeout", nil)
|
|
271
|
+
case .unsupportedCapability, .unsupportedGenerationGuide,
|
|
272
|
+
.unsupportedLanguageOrLocale, .unsupportedTranscriptContent:
|
|
273
|
+
return ("unsupported", nil)
|
|
274
|
+
@unknown default: return ("generation", nil)
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
return (legacyKind(error), nil)
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
/// macOS 26's error type, kept so one helper source serves both OS versions.
|
|
282
|
+
///
|
|
283
|
+
/// Building on macOS 27 emits one deprecation warning at the call site below.
|
|
284
|
+
/// That is benign and self-correcting: the deployment target is derived from the
|
|
285
|
+
/// host SDK, so on a macOS 26 machine — the only place this branch is reachable —
|
|
286
|
+
/// the target is macos26.0 and nothing is deprecated yet. On macOS 27 the
|
|
287
|
+
/// warning points at code `#available` has already made unreachable.
|
|
288
|
+
@available(macOS 26.0, *)
|
|
289
|
+
@available(macOS, deprecated: 27.0)
|
|
290
|
+
private func legacyKind(_ error: Error) -> String {
|
|
291
|
+
if let legacy = error as? LanguageModelSession.GenerationError {
|
|
292
|
+
switch legacy {
|
|
293
|
+
case .exceededContextWindowSize: return "context"
|
|
294
|
+
case .guardrailViolation, .refusal: return "guardrail"
|
|
295
|
+
case .rateLimited: return "quota"
|
|
296
|
+
default: return "generation"
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
return "generation"
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
/// Models are keyed by use case and guardrails, because both are fixed at
|
|
303
|
+
/// construction. Building one is cheap; keeping them avoids re-resolving assets
|
|
304
|
+
/// when a caller alternates between, say, general and contentTagging.
|
|
305
|
+
@available(macOS 26.0, *)
|
|
306
|
+
private final class ModelCache {
|
|
307
|
+
static var models: [String: SystemLanguageModel] = [:]
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
/// `contentTagging` is a use case Apple ships specifically for tagging and
|
|
311
|
+
/// topic extraction; `permissive` relaxes the guardrails for content
|
|
312
|
+
/// *transformation* tasks (rewriting, summarising text that the default
|
|
313
|
+
/// guardrails would refuse). Both are macOS 26+.
|
|
314
|
+
@available(macOS 26.0, *)
|
|
315
|
+
private func modelFor(useCase: String, guardrails: String) -> SystemLanguageModel {
|
|
316
|
+
let key = "\(useCase)|\(guardrails)"
|
|
317
|
+
if let cached = ModelCache.models[key] { return cached }
|
|
318
|
+
let resolvedUseCase: SystemLanguageModel.UseCase =
|
|
319
|
+
useCase == "contentTagging" ? .contentTagging : .general
|
|
320
|
+
let resolvedGuardrails: SystemLanguageModel.Guardrails =
|
|
321
|
+
guardrails == "permissive" ? .permissiveContentTransformations : .default
|
|
322
|
+
let model = SystemLanguageModel(useCase: resolvedUseCase, guardrails: resolvedGuardrails)
|
|
323
|
+
ModelCache.models[key] = model
|
|
324
|
+
return model
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
/// Apple's built-in tools: on-device OCR and barcode reading (Vision) and the
|
|
328
|
+
/// Spotlight semantic index (local RAG). All macOS 27+. Unknown names are
|
|
329
|
+
/// rejected loudly — a silently ignored tool would mislead the caller into
|
|
330
|
+
/// thinking the model had a capability it did not.
|
|
331
|
+
@available(macOS 26.0, *)
|
|
332
|
+
private func toolsFor(_ names: [String]) throws -> [any Tool] {
|
|
333
|
+
var out: [any Tool] = []
|
|
334
|
+
for name in names {
|
|
335
|
+
switch name {
|
|
336
|
+
case "ocr":
|
|
337
|
+
if #available(macOS 27.0, *) {
|
|
338
|
+
#if canImport(_Vision_FoundationModels)
|
|
339
|
+
out.append(OCRTool())
|
|
340
|
+
#else
|
|
341
|
+
throw ToolError.unsupported("ocr needs macOS 27 with Vision tools")
|
|
342
|
+
#endif
|
|
343
|
+
} else {
|
|
344
|
+
throw ToolError.unsupported("ocr needs macOS 27 or later")
|
|
345
|
+
}
|
|
346
|
+
case "barcode":
|
|
347
|
+
if #available(macOS 27.0, *) {
|
|
348
|
+
#if canImport(_Vision_FoundationModels)
|
|
349
|
+
out.append(BarcodeReaderTool())
|
|
350
|
+
#else
|
|
351
|
+
throw ToolError.unsupported("barcode needs macOS 27 with Vision tools")
|
|
352
|
+
#endif
|
|
353
|
+
} else {
|
|
354
|
+
throw ToolError.unsupported("barcode needs macOS 27 or later")
|
|
355
|
+
}
|
|
356
|
+
case "spotlight":
|
|
357
|
+
if #available(macOS 27.0, *) {
|
|
358
|
+
#if canImport(_CoreSpotlight_FoundationModels)
|
|
359
|
+
out.append(SpotlightSearchTool())
|
|
360
|
+
#else
|
|
361
|
+
throw ToolError.unsupported("spotlight needs macOS 27 with Spotlight tools")
|
|
362
|
+
#endif
|
|
363
|
+
} else {
|
|
364
|
+
throw ToolError.unsupported("spotlight needs macOS 27 or later")
|
|
365
|
+
}
|
|
366
|
+
default:
|
|
367
|
+
throw ToolError.unknown("unknown tool \"\(name)\" (want ocr, barcode, spotlight)")
|
|
368
|
+
}
|
|
369
|
+
}
|
|
370
|
+
return out
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
private enum ToolError: Error {
|
|
374
|
+
case unknown(String)
|
|
375
|
+
case unsupported(String)
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
/// Coerce a JSON number regardless of int/double mismatch.
|
|
379
|
+
///
|
|
380
|
+
/// `JSONSerialization` produces `NSNumber`, but `as? Double` fails for an
|
|
381
|
+
/// integer-valued `NSNumber` (and vice versa), so a caller passing
|
|
382
|
+
/// `temperature: 1` or `maxTokens: 100.0` would silently get `nil`.
|
|
383
|
+
private func num(_ v: Any?) -> Double? { (v as? NSNumber)?.doubleValue }
|
|
384
|
+
|
|
385
|
+
/// Coerce a JSON number to `Int`, returning `nil` for missing or out-of-range
|
|
386
|
+
/// values rather than trapping or wrapping.
|
|
387
|
+
private func intNum(_ v: Any?) -> Int? {
|
|
388
|
+
guard let n = v as? NSNumber else { return nil }
|
|
389
|
+
let d = n.doubleValue
|
|
390
|
+
guard d.isFinite, d >= Double(Int.min), d <= Double(Int.max) else { return nil }
|
|
391
|
+
return n.intValue
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
/// Coerce a JSON number to `UInt64` for seeds, which may exceed `Int.max`.
|
|
395
|
+
private func u64Num(_ v: Any?) -> UInt64? { (v as? NSNumber)?.uint64Value }
|
|
396
|
+
|
|
397
|
+
/// Sampling mode.
|
|
398
|
+
///
|
|
399
|
+
/// `greedy` is deterministic but degenerates under guided generation — it padded
|
|
400
|
+
/// an unbounded array forever, then ran away inside a single string. The seeded
|
|
401
|
+
/// modes give the same determinism *without* that failure: `topK` with a seed
|
|
402
|
+
/// returned byte-identical output across three fresh sessions here. Note the
|
|
403
|
+
/// determinism depends on a fresh session per request, which is what this helper
|
|
404
|
+
/// does by default; reusing a session changes the transcript and with it the
|
|
405
|
+
/// output.
|
|
406
|
+
@available(macOS 26.0, *)
|
|
407
|
+
private func samplingFrom(_ value: Any?) -> GenerationOptions.SamplingMode? {
|
|
408
|
+
guard let spec = value as? [String: Any], let mode = spec["mode"] as? String else { return nil }
|
|
409
|
+
let seed = u64Num(spec["seed"])
|
|
410
|
+
switch mode {
|
|
411
|
+
case "greedy":
|
|
412
|
+
return .greedy
|
|
413
|
+
case "topK":
|
|
414
|
+
return .random(top: intNum(spec["k"]) ?? 50, seed: seed)
|
|
415
|
+
case "threshold":
|
|
416
|
+
return .random(probabilityThreshold: num(spec["p"]) ?? 0.9, seed: seed)
|
|
417
|
+
default:
|
|
418
|
+
return nil
|
|
419
|
+
}
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
@available(macOS 26.0, *)
|
|
423
|
+
private struct ImageSpec {
|
|
424
|
+
var path: String
|
|
425
|
+
var label: String
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
/// `images` accepts a plain path or `{"path":..,"label":..}`. Labels mirror
|
|
429
|
+
/// `fm --label`: they let the caller name attachments ("receipt", "chart")
|
|
430
|
+
/// so follow-up turns can refer to them. A missing file is an error, never a
|
|
431
|
+
/// silent drop.
|
|
432
|
+
@available(macOS 26.0, *)
|
|
433
|
+
private func imageSpecsFrom(_ value: Any?) -> [ImageSpec] {
|
|
434
|
+
guard let raw = value as? [Any] else { return [] }
|
|
435
|
+
var out: [ImageSpec] = []
|
|
436
|
+
for (index, item) in raw.enumerated() {
|
|
437
|
+
if let path = item as? String {
|
|
438
|
+
out.append(ImageSpec(path: path, label: "image\(index + 1)"))
|
|
439
|
+
} else if let dict = item as? [String: Any],
|
|
440
|
+
let path = dict["path"] as? String {
|
|
441
|
+
let label = (dict["label"] as? String)?.isEmpty == false
|
|
442
|
+
? dict["label"] as! String : "image\(index + 1)"
|
|
443
|
+
out.append(ImageSpec(path: path, label: label))
|
|
444
|
+
}
|
|
445
|
+
}
|
|
446
|
+
return out
|
|
447
|
+
}
|
|
448
|
+
|
|
449
|
+
/// Build the prompt, attaching any images. Vision is macOS 27+; on 26 the paths
|
|
450
|
+
/// are reported as unsupported rather than silently dropped, because a caller
|
|
451
|
+
/// who asked about an image and got an answer that ignored it has been misled.
|
|
452
|
+
@available(macOS 26.0, *)
|
|
453
|
+
private func promptWith(text: String, imageSpecs: [ImageSpec]) -> Prompt {
|
|
454
|
+
guard !imageSpecs.isEmpty else { return Prompt(text) }
|
|
455
|
+
if #available(macOS 27.0, *) {
|
|
456
|
+
let attachments = imageSpecs.map { spec in
|
|
457
|
+
Attachment(imageURL: URL(fileURLWithPath: spec.path)).label(spec.label)
|
|
458
|
+
}
|
|
459
|
+
return Prompt {
|
|
460
|
+
for attachment in attachments { attachment }
|
|
461
|
+
text
|
|
462
|
+
}
|
|
463
|
+
}
|
|
464
|
+
return Prompt(text)
|
|
465
|
+
}
|
|
466
|
+
|
|
467
|
+
@available(macOS 26.0, *)
|
|
468
|
+
private func stringArray(_ value: Any?) -> [String] {
|
|
469
|
+
(value as? [Any] ?? []).compactMap { $0 as? String }
|
|
470
|
+
}
|
|
471
|
+
|
|
472
|
+
/// One request/response cycle. Kept separate from the transport so that
|
|
473
|
+
/// one-shot and serve modes cannot drift apart.
|
|
474
|
+
@available(macOS 26.0, *)
|
|
475
|
+
private func handle(
|
|
476
|
+
envelope: [String: Any],
|
|
477
|
+
schemaCache: inout [String: GenerationSchema]
|
|
478
|
+
) async {
|
|
479
|
+
await handleEnvelope(envelope: envelope, schemaCache: &schemaCache, streaming: false)
|
|
480
|
+
}
|
|
481
|
+
|
|
482
|
+
@available(macOS 26.0, *)
|
|
483
|
+
private func handleEnvelope(
|
|
484
|
+
envelope: [String: Any],
|
|
485
|
+
schemaCache: inout [String: GenerationSchema],
|
|
486
|
+
streaming: Bool
|
|
487
|
+
) async {
|
|
488
|
+
let useCase = envelope["useCase"] as? String ?? "general"
|
|
489
|
+
let guardrails = envelope["guardrails"] as? String ?? "default"
|
|
490
|
+
let model = modelFor(useCase: useCase, guardrails: guardrails)
|
|
491
|
+
|
|
492
|
+
switch model.availability {
|
|
493
|
+
case .available:
|
|
494
|
+
break
|
|
495
|
+
case .unavailable(let reason):
|
|
496
|
+
emit(["ok": false, "kind": "availability",
|
|
497
|
+
"error": "the on-device model is unavailable: \(describe(reason))"])
|
|
498
|
+
return
|
|
499
|
+
}
|
|
500
|
+
|
|
501
|
+
let instructions = envelope["instructions"] as? String ?? ""
|
|
502
|
+
let promptText = envelope["prompt"] as? String ?? ""
|
|
503
|
+
let imageSpecs = imageSpecsFrom(envelope["images"])
|
|
504
|
+
let toolNames = stringArray(envelope["tools"])
|
|
505
|
+
RequestContext.toolNames = toolNames
|
|
506
|
+
let sessionId = envelope["sessionId"] as? String
|
|
507
|
+
|
|
508
|
+
if !imageSpecs.isEmpty {
|
|
509
|
+
if #available(macOS 27.0, *) {
|
|
510
|
+
guard model.capabilities.contains(.vision) else {
|
|
511
|
+
emit(["ok": false, "kind": "unsupported",
|
|
512
|
+
"error": "this model does not support vision"])
|
|
513
|
+
return
|
|
514
|
+
}
|
|
515
|
+
for spec in imageSpecs where !FileManager.default.fileExists(atPath: spec.path) {
|
|
516
|
+
emit(["ok": false, "kind": "unsupported",
|
|
517
|
+
"error": "image not found: \(spec.path)"])
|
|
518
|
+
return
|
|
519
|
+
}
|
|
520
|
+
} else {
|
|
521
|
+
emit(["ok": false, "kind": "unsupported",
|
|
522
|
+
"error": "images need macOS 27 or later"])
|
|
523
|
+
return
|
|
524
|
+
}
|
|
525
|
+
}
|
|
526
|
+
|
|
527
|
+
let tools: [any Tool]
|
|
528
|
+
do {
|
|
529
|
+
tools = try toolsFor(toolNames)
|
|
530
|
+
} catch let err as ToolError {
|
|
531
|
+
switch err {
|
|
532
|
+
case .unknown(let m), .unsupported(let m):
|
|
533
|
+
emit(["ok": false, "kind": "unsupported", "error": m])
|
|
534
|
+
return
|
|
535
|
+
}
|
|
536
|
+
} catch {
|
|
537
|
+
emit(["ok": false, "kind": "unsupported", "error": "\(error)"])
|
|
538
|
+
return
|
|
539
|
+
}
|
|
540
|
+
if !tools.isEmpty {
|
|
541
|
+
guard model.capabilities.contains(.toolCalling) else {
|
|
542
|
+
emit(["ok": false, "kind": "unsupported",
|
|
543
|
+
"error": "this model does not support tool calling"])
|
|
544
|
+
return
|
|
545
|
+
}
|
|
546
|
+
}
|
|
547
|
+
|
|
548
|
+
// Token counting and prewarming share the model but not the generation path.
|
|
549
|
+
// An empty op defaults to generate; an unknown op is an error rather than
|
|
550
|
+
// silently running generate (which would hide a caller typo like
|
|
551
|
+
// "countToken").
|
|
552
|
+
let rawOp = envelope["op"] as? String
|
|
553
|
+
let op = (rawOp == nil || rawOp == "") ? "generate" : rawOp!
|
|
554
|
+
switch op {
|
|
555
|
+
case "generate", "stream":
|
|
556
|
+
break
|
|
557
|
+
case "countTokens":
|
|
558
|
+
guard #available(macOS 27.0, *) else {
|
|
559
|
+
emit(["ok": false, "kind": "unsupported",
|
|
560
|
+
"error": "counting tokens needs macOS 27 or later"])
|
|
561
|
+
return
|
|
562
|
+
}
|
|
563
|
+
do {
|
|
564
|
+
// Instructions and schema ride in the same window as the prompt, so
|
|
565
|
+
// a caller budgeting against contextSize needs all three counted.
|
|
566
|
+
var total = try await model.tokenCount(for: promptWith(text: promptText,
|
|
567
|
+
imageSpecs: imageSpecs))
|
|
568
|
+
if !instructions.isEmpty {
|
|
569
|
+
total += try await model.tokenCount(for: Instructions(instructions))
|
|
570
|
+
}
|
|
571
|
+
if !tools.isEmpty {
|
|
572
|
+
total += (try? await model.tokenCount(for: tools)) ?? 0
|
|
573
|
+
}
|
|
574
|
+
emit(["ok": true, "tokens": total, "contextSize": model.contextSize])
|
|
575
|
+
} catch {
|
|
576
|
+
emit(["ok": false, "kind": classify(error).kind, "error": "\(error)"])
|
|
577
|
+
}
|
|
578
|
+
return
|
|
579
|
+
|
|
580
|
+
case "prewarm":
|
|
581
|
+
// Loads model assets now so the first real call does not pay for it.
|
|
582
|
+
// Measured benefit is small once the assets are resident system-wide
|
|
583
|
+
// (0.31s vs 0.36s for an unprewarmed first call here); the win is on a
|
|
584
|
+
// genuinely cold system, where the first framework call took 7.8s.
|
|
585
|
+
// The session is discarded on purpose; see sessionFor above.
|
|
586
|
+
if tools.isEmpty {
|
|
587
|
+
LanguageModelSession(model: model, instructions: instructions).prewarm()
|
|
588
|
+
} else {
|
|
589
|
+
LanguageModelSession(model: model, tools: tools, instructions: instructions).prewarm()
|
|
590
|
+
}
|
|
591
|
+
emit(["ok": true, "prewarmed": true])
|
|
592
|
+
return
|
|
593
|
+
|
|
594
|
+
case "history":
|
|
595
|
+
guard let sid = sessionId, !sid.isEmpty else {
|
|
596
|
+
emit(["ok": false, "kind": "unsupported",
|
|
597
|
+
"error": "history needs a sessionId"])
|
|
598
|
+
return
|
|
599
|
+
}
|
|
600
|
+
if let entry = SessionStore.named[sid] {
|
|
601
|
+
emit(["ok": true, "sessionId": sid, "instructions": entry.instructions,
|
|
602
|
+
"history": entry.history])
|
|
603
|
+
} else {
|
|
604
|
+
let stored = SessionStore.loadHistory(sid)
|
|
605
|
+
emit(["ok": true, "sessionId": sid,
|
|
606
|
+
"instructions": stored.instructions ?? instructions,
|
|
607
|
+
"history": stored.history])
|
|
608
|
+
}
|
|
609
|
+
return
|
|
610
|
+
|
|
611
|
+
case "reset":
|
|
612
|
+
if let sid = sessionId, !sid.isEmpty {
|
|
613
|
+
SessionStore.named.removeValue(forKey: sid)
|
|
614
|
+
SessionStore.removeFile(sid)
|
|
615
|
+
} else {
|
|
616
|
+
SessionStore.named.removeAll()
|
|
617
|
+
}
|
|
618
|
+
emit(["ok": true, "reset": true])
|
|
619
|
+
return
|
|
620
|
+
|
|
621
|
+
default:
|
|
622
|
+
emit(["ok": false, "kind": "unsupported",
|
|
623
|
+
"error": "unsupported op \"\(op)\""])
|
|
624
|
+
return
|
|
625
|
+
}
|
|
626
|
+
|
|
627
|
+
// Streaming + guided generation do not mix in v1: the schema stream yields
|
|
628
|
+
// GeneratedContent snapshots whose partials are not plain-text deltas.
|
|
629
|
+
// Reject loudly rather than emitting misleading partial JSON.
|
|
630
|
+
let wantsStream = (op == "stream") || streaming
|
|
631
|
+
if wantsStream,
|
|
632
|
+
let schemaObject = envelope["schema"], !(schemaObject is NSNull) {
|
|
633
|
+
emit(["ok": false, "kind": "unsupported",
|
|
634
|
+
"error": "streaming with a schema is not supported; use generate for JSON"])
|
|
635
|
+
return
|
|
636
|
+
}
|
|
637
|
+
|
|
638
|
+
// No schema means a free-text call. api-scribe only ever asked for JSON and
|
|
639
|
+
// so required a schema on every request; `text()` needs the unconstrained
|
|
640
|
+
// `respond(to:options:)` overload instead.
|
|
641
|
+
var schema: GenerationSchema? = nil
|
|
642
|
+
if let schemaObject = envelope["schema"], !(schemaObject is NSNull) {
|
|
643
|
+
guard let schemaData = try? JSONSerialization.data(withJSONObject: schemaObject),
|
|
644
|
+
let schemaKey = String(data: schemaData, encoding: .utf8) else {
|
|
645
|
+
emit(["ok": false, "kind": "schema",
|
|
646
|
+
"error": "helper payload is missing a usable schema"])
|
|
647
|
+
return
|
|
648
|
+
}
|
|
649
|
+
// A changing GenerationSchema costs only ~0.15s per call, so per-request
|
|
650
|
+
// schemas are fine; this cache just makes a repeated one free.
|
|
651
|
+
if let cached = schemaCache[schemaKey] {
|
|
652
|
+
schema = cached
|
|
653
|
+
} else {
|
|
654
|
+
do {
|
|
655
|
+
let decoded = try JSONDecoder().decode(GenerationSchema.self, from: schemaData)
|
|
656
|
+
// Bound the cache: a long-lived server may see many schemas.
|
|
657
|
+
if schemaCache.count >= 64 { schemaCache.removeAll() }
|
|
658
|
+
schemaCache[schemaKey] = decoded
|
|
659
|
+
schema = decoded
|
|
660
|
+
} catch {
|
|
661
|
+
emit(["ok": false, "kind": "schema",
|
|
662
|
+
"error": "Apple rejected the response schema: \(error)"])
|
|
663
|
+
return
|
|
664
|
+
}
|
|
665
|
+
}
|
|
666
|
+
}
|
|
667
|
+
|
|
668
|
+
let options = GenerationOptions(
|
|
669
|
+
samplingMode: samplingFrom(envelope["sampling"]),
|
|
670
|
+
temperature: num(envelope["temperature"]),
|
|
671
|
+
maximumResponseTokens: intNum(envelope["maxTokens"])
|
|
672
|
+
)
|
|
673
|
+
let reuse = envelope["reuseSession"] as? Bool ?? false
|
|
674
|
+
// api-scribe passed false because it spelled the schema out in its own
|
|
675
|
+
// system prompt. A library cannot assume that, and a schema's `description`
|
|
676
|
+
// fields are how the decoder gets its generation guidance, so default true.
|
|
677
|
+
let includeSchema = envelope["includeSchemaInPrompt"] as? Bool ?? true
|
|
678
|
+
|
|
679
|
+
let session: LanguageModelSession
|
|
680
|
+
// Cross-process continuity: a fresh native session (new process, or param
|
|
681
|
+
// change) has no transcript, but the mirrored history survived on disk.
|
|
682
|
+
// Reprise the recent turns as prompt context so `run --session` remembers
|
|
683
|
+
// across CLI invocations too. Bounded to the last 10 turns; in-process
|
|
684
|
+
// calls skip this and use the native transcript alone.
|
|
685
|
+
var effectivePromptText = promptText
|
|
686
|
+
if let sid = sessionId, !sid.isEmpty {
|
|
687
|
+
let hadMemory = SessionStore.named[sid] != nil
|
|
688
|
+
session = namedSessionFor(id: sid, model: model, instructions: instructions,
|
|
689
|
+
useCase: useCase, guardrails: guardrails, tools: tools)
|
|
690
|
+
if !hadMemory, let entry = SessionStore.named[sid], !entry.history.isEmpty {
|
|
691
|
+
let recent = entry.history.suffix(10).map { turn -> String in
|
|
692
|
+
let who = (turn["role"] == "user") ? "User" : "Assistant"
|
|
693
|
+
return "\(who): \(turn["content"] ?? "")"
|
|
694
|
+
}.joined(separator: "\n")
|
|
695
|
+
effectivePromptText =
|
|
696
|
+
"Previous conversation:\n\(recent)\n\nCurrent request:\n\(promptText)"
|
|
697
|
+
}
|
|
698
|
+
} else if tools.isEmpty {
|
|
699
|
+
session = sessionFor(model: model, instructions: instructions, reuse: reuse)
|
|
700
|
+
} else {
|
|
701
|
+
session = LanguageModelSession(model: model, tools: tools, instructions: instructions)
|
|
702
|
+
}
|
|
703
|
+
let prompt = promptWith(text: effectivePromptText, imageSpecs: imageSpecs)
|
|
704
|
+
|
|
705
|
+
if wantsStream {
|
|
706
|
+
await handleStream(session: session, prompt: prompt, options: options,
|
|
707
|
+
sessionId: sessionId, promptText: promptText)
|
|
708
|
+
return
|
|
709
|
+
}
|
|
710
|
+
|
|
711
|
+
do {
|
|
712
|
+
let content: String
|
|
713
|
+
if let schema {
|
|
714
|
+
let response = try await session.respond(
|
|
715
|
+
to: prompt, schema: schema, includeSchemaInPrompt: includeSchema, options: options
|
|
716
|
+
)
|
|
717
|
+
content = response.content.jsonString
|
|
718
|
+
} else {
|
|
719
|
+
// Non-streaming. `session.streamResponse` is the streaming path;
|
|
720
|
+
// see handleStream below.
|
|
721
|
+
let response = try await session.respond(to: prompt, options: options)
|
|
722
|
+
content = response.content
|
|
723
|
+
}
|
|
724
|
+
recordTurn(sessionId: sessionId, prompt: promptText, content: content,
|
|
725
|
+
instructions: instructions)
|
|
726
|
+
emit(["ok": true, "content": content])
|
|
727
|
+
} catch {
|
|
728
|
+
let (kind, resetDate) = classify(error)
|
|
729
|
+
var payload: [String: Any] = ["ok": false, "kind": kind, "error": "\(error)"]
|
|
730
|
+
if let resetDate { payload["resetDate"] = resetDate }
|
|
731
|
+
emit(payload)
|
|
732
|
+
}
|
|
733
|
+
}
|
|
734
|
+
|
|
735
|
+
@available(macOS 26.0, *)
|
|
736
|
+
private func recordTurn(sessionId: String?, prompt: String, content: String,
|
|
737
|
+
instructions: String) {
|
|
738
|
+
guard let sid = sessionId, !sid.isEmpty else { return }
|
|
739
|
+
if var entry = SessionStore.named[sid] {
|
|
740
|
+
entry.history.append(["role": "user", "content": prompt])
|
|
741
|
+
entry.history.append(["role": "assistant", "content": content])
|
|
742
|
+
// Bound mirrored history: native transcript still holds the full
|
|
743
|
+
// context for this process lifetime; the mirror is for `history` and
|
|
744
|
+
// restart continuity, not inference.
|
|
745
|
+
if entry.history.count > 200 { entry.history.removeFirst(entry.history.count - 200) }
|
|
746
|
+
SessionStore.named[sid] = entry
|
|
747
|
+
SessionStore.saveHistory(id: sid, instructions: entry.instructions,
|
|
748
|
+
history: entry.history)
|
|
749
|
+
}
|
|
750
|
+
}
|
|
751
|
+
|
|
752
|
+
/// Streaming generation. Each snapshot's `content` is the cumulative partial,
|
|
753
|
+
/// so deltas are computed by stripping the previous prefix; when the model
|
|
754
|
+
/// revises earlier text (rare for plain prose) the whole new partial is sent
|
|
755
|
+
/// so the client never silently drops a correction.
|
|
756
|
+
@available(macOS 26.0, *)
|
|
757
|
+
private func handleStream(session: LanguageModelSession, prompt: Prompt,
|
|
758
|
+
options: GenerationOptions, sessionId: String?,
|
|
759
|
+
promptText: String) async {
|
|
760
|
+
do {
|
|
761
|
+
let stream = session.streamResponse(to: prompt, options: options)
|
|
762
|
+
var previous = ""
|
|
763
|
+
var full = ""
|
|
764
|
+
for try await snapshot in stream {
|
|
765
|
+
let current: String = snapshot.content
|
|
766
|
+
full = current
|
|
767
|
+
let delta: String
|
|
768
|
+
if current.hasPrefix(previous) {
|
|
769
|
+
delta = String(current.dropFirst(previous.count))
|
|
770
|
+
} else {
|
|
771
|
+
delta = current
|
|
772
|
+
}
|
|
773
|
+
previous = current
|
|
774
|
+
if !delta.isEmpty {
|
|
775
|
+
emit(["ok": true, "delta": delta, "done": false])
|
|
776
|
+
}
|
|
777
|
+
}
|
|
778
|
+
recordTurn(sessionId: sessionId, prompt: promptText, content: full,
|
|
779
|
+
instructions: "")
|
|
780
|
+
// Refresh persisted instructions for named sessions without clobbering.
|
|
781
|
+
emit(["ok": true, "content": full, "done": true])
|
|
782
|
+
} catch {
|
|
783
|
+
let (kind, resetDate) = classify(error)
|
|
784
|
+
var payload: [String: Any] = ["ok": false, "kind": kind, "error": "\(error)"]
|
|
785
|
+
if let resetDate { payload["resetDate"] = resetDate }
|
|
786
|
+
emit(payload)
|
|
787
|
+
}
|
|
788
|
+
}
|
|
789
|
+
|
|
790
|
+
private func describe(_ reason: SystemLanguageModel.Availability.UnavailableReason) -> String {
|
|
791
|
+
switch reason {
|
|
792
|
+
case .deviceNotEligible: return "deviceNotEligible"
|
|
793
|
+
case .appleIntelligenceNotEnabled: return "appleIntelligenceNotEnabled"
|
|
794
|
+
case .modelNotReady: return "modelNotReady"
|
|
795
|
+
@unknown default: return "unknown"
|
|
796
|
+
}
|
|
797
|
+
}
|
|
798
|
+
|
|
799
|
+
/// Private Cloud Compute quota, read without calling it.
|
|
800
|
+
///
|
|
801
|
+
/// PCC *inference* needs `com.apple.developer.private-cloud-compute`, which is
|
|
802
|
+
/// AMFI-restricted and unavailable to any installable package — that is why the
|
|
803
|
+
/// cloud tier goes through Shortcuts. But `quotaUsage` is readable from an
|
|
804
|
+
/// unentitled process, so a caller can see whether the cloud tier is worth
|
|
805
|
+
/// trying before spending a Shortcuts round trip on it.
|
|
806
|
+
@available(macOS 27.0, *)
|
|
807
|
+
private func cloudQuota() -> [String: Any] {
|
|
808
|
+
let pcc = PrivateCloudComputeLanguageModel()
|
|
809
|
+
var out: [String: Any] = ["isAvailable": pcc.isAvailable]
|
|
810
|
+
let usage = pcc.quotaUsage
|
|
811
|
+
switch usage.status {
|
|
812
|
+
case .belowLimit(let below):
|
|
813
|
+
out["status"] = "belowLimit"
|
|
814
|
+
out["approachingLimit"] = below.isApproachingLimit
|
|
815
|
+
case .limitReached:
|
|
816
|
+
out["status"] = "limitReached"
|
|
817
|
+
@unknown default:
|
|
818
|
+
out["status"] = "unknown"
|
|
819
|
+
}
|
|
820
|
+
if let reset = usage.resetDate {
|
|
821
|
+
out["resetDate"] = ISO8601DateFormatter().string(from: reset)
|
|
822
|
+
}
|
|
823
|
+
return out
|
|
824
|
+
}
|
|
825
|
+
|
|
826
|
+
@main
|
|
827
|
+
struct AppleLLMHelper {
|
|
828
|
+
static func main() async {
|
|
829
|
+
let model = SystemLanguageModel.default
|
|
830
|
+
|
|
831
|
+
if CommandLine.arguments.contains("--probe") {
|
|
832
|
+
var payload: [String: Any] = ["contextSize": model.contextSize]
|
|
833
|
+
switch model.availability {
|
|
834
|
+
case .available:
|
|
835
|
+
payload["available"] = true
|
|
836
|
+
case .unavailable(let reason):
|
|
837
|
+
payload["available"] = false
|
|
838
|
+
payload["reason"] = describe(reason)
|
|
839
|
+
}
|
|
840
|
+
if #available(macOS 27.0, *) {
|
|
841
|
+
payload["variant"] = model.variant.displayName
|
|
842
|
+
// What the model can actually do, rather than what the docs
|
|
843
|
+
// imply. On this machine the on-device model reports vision and
|
|
844
|
+
// tool calling but *not* reasoning, while PCC reports all four.
|
|
845
|
+
let capabilities = model.capabilities
|
|
846
|
+
payload["capabilities"] = [
|
|
847
|
+
"vision": capabilities.contains(.vision),
|
|
848
|
+
"guidedGeneration": capabilities.contains(.guidedGeneration),
|
|
849
|
+
"reasoning": capabilities.contains(.reasoning),
|
|
850
|
+
"toolCalling": capabilities.contains(.toolCalling),
|
|
851
|
+
]
|
|
852
|
+
payload["useCases"] = ["general", "contentTagging"]
|
|
853
|
+
payload["cloud"] = cloudQuota()
|
|
854
|
+
payload["features"] = [
|
|
855
|
+
"streaming": true,
|
|
856
|
+
"sessions": true,
|
|
857
|
+
"history": true,
|
|
858
|
+
"labelledAttachments": true,
|
|
859
|
+
"builtInTools": ["ocr", "barcode", "spotlight"],
|
|
860
|
+
]
|
|
861
|
+
}
|
|
862
|
+
emit(payload)
|
|
863
|
+
return
|
|
864
|
+
}
|
|
865
|
+
|
|
866
|
+
var schemaCache: [String: GenerationSchema] = [:]
|
|
867
|
+
|
|
868
|
+
// Serve mode: one request per stdin line, one response per stdout line,
|
|
869
|
+
// for as long as the parent keeps the pipe open. Spawning a process per
|
|
870
|
+
// request instead makes the model reload between calls, which measured
|
|
871
|
+
// at ~17s per request against ~1.5s once it is resident.
|
|
872
|
+
//
|
|
873
|
+
// Streaming is the one exception to one-line-per-request: op "stream"
|
|
874
|
+
// emits N {"delta","done":false} lines plus a final {"content",
|
|
875
|
+
// "done":true}. The parent must keep reading until done:true before
|
|
876
|
+
// sending the next request; the next stdin line is not consumed until
|
|
877
|
+
// the stream completes, so replies cannot interleave.
|
|
878
|
+
if CommandLine.arguments.contains("--serve") {
|
|
879
|
+
while let line = readLine(strippingNewline: true) {
|
|
880
|
+
if line.isEmpty { continue }
|
|
881
|
+
guard let data = line.data(using: .utf8),
|
|
882
|
+
let envelope = (try? JSONSerialization.jsonObject(with: data)) as? [String: Any]
|
|
883
|
+
else {
|
|
884
|
+
emit(["ok": false, "error": "helper could not parse a request line as JSON"])
|
|
885
|
+
continue
|
|
886
|
+
}
|
|
887
|
+
await handle(envelope: envelope, schemaCache: &schemaCache)
|
|
888
|
+
}
|
|
889
|
+
return
|
|
890
|
+
}
|
|
891
|
+
|
|
892
|
+
let input = FileHandle.standardInput.readDataToEndOfFile()
|
|
893
|
+
guard let envelope = (try? JSONSerialization.jsonObject(with: input)) as? [String: Any] else {
|
|
894
|
+
emit(["ok": false, "error": "helper could not parse its stdin payload as JSON"])
|
|
895
|
+
return
|
|
896
|
+
}
|
|
897
|
+
|
|
898
|
+
await handle(envelope: envelope, schemaCache: &schemaCache)
|
|
899
|
+
}
|
|
900
|
+
}
|