hoomanjs 1.42.1 → 1.43.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/acp/acp-agent.js +21 -3
- package/dist/acp/acp-agent.js.map +1 -1
- package/dist/chat/app.js +19 -2
- package/dist/chat/app.js.map +1 -1
- package/dist/chat/components/BottomChrome.d.ts +1 -0
- package/dist/chat/components/BottomChrome.js.map +1 -1
- package/dist/chat/components/StatusBar.d.ts +1 -0
- package/dist/chat/components/StatusBar.js +3 -0
- package/dist/chat/components/StatusBar.js.map +1 -1
- package/dist/configure/app.js +145 -3
- package/dist/configure/app.js.map +1 -1
- package/dist/configure/types.d.ts +3 -0
- package/dist/core/config.d.ts +67 -0
- package/dist/core/config.js +40 -0
- package/dist/core/config.js.map +1 -1
- package/dist/core/models/hub-download.d.ts +27 -0
- package/dist/core/models/hub-download.js +136 -0
- package/dist/core/models/hub-download.js.map +1 -0
- package/dist/core/models/index.js +1 -0
- package/dist/core/models/index.js.map +1 -1
- package/dist/core/models/llama-cpp/index.js +3 -0
- package/dist/core/models/llama-cpp/index.js.map +1 -1
- package/dist/core/models/llama-cpp/resolve-model.js +3 -132
- package/dist/core/models/llama-cpp/resolve-model.js.map +1 -1
- package/dist/core/models/llama-cpp/strands-llama-cpp.d.ts +6 -0
- package/dist/core/models/llama-cpp/strands-llama-cpp.js +6 -0
- package/dist/core/models/llama-cpp/strands-llama-cpp.js.map +1 -1
- package/dist/core/models/mlx/index.d.ts +13 -0
- package/dist/core/models/mlx/index.js +47 -0
- package/dist/core/models/mlx/index.js.map +1 -0
- package/dist/core/models/mlx/resolve-model.d.ts +31 -0
- package/dist/core/models/mlx/resolve-model.js +149 -0
- package/dist/core/models/mlx/resolve-model.js.map +1 -0
- package/dist/core/models/mlx/strands-mlx.d.ts +45 -0
- package/dist/core/models/mlx/strands-mlx.js +439 -0
- package/dist/core/models/mlx/strands-mlx.js.map +1 -0
- package/dist/core/models/types.d.ts +149 -4
- package/dist/core/models/types.js +27 -0
- package/dist/core/models/types.js.map +1 -1
- package/dist/core/skills/built-in/hooman-config/SKILL.md +34 -2
- package/dist/core/skills/built-in/hooman-config/providers.md +4 -2
- package/dist/core/utils/billing.d.ts +23 -2
- package/dist/core/utils/billing.js +25 -2
- package/dist/core/utils/billing.js.map +1 -1
- package/package.json +2 -1
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"index.js","sourceRoot":"","sources":["../../../../src/core/models/mlx/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,eAAe,EAAE,MAAM,kBAAkB,CAAC;AACnD,OAAO,EAAE,uBAAuB,EAAE,MAAM,aAAa,CAAC;AACtD,OAAO,EAAE,4BAA4B,EAAE,MAAM,aAAa,CAAC;AAG3D;;;;;;;;;GASG;AACH,MAAM,UAAU,MAAM,CACpB,eAAmC,EACnC,UAAsB;IAEtB,oEAAoE;IACpE,yEAAyE;IACzE,kDAAkD;IAClD,MAAM,SAAS,GAAG,eAAe,CAAC,SAAS,CAAC;IAC5C,MAAM,MAAM,GAAG,SAAS,EAAE,MAAM,CAAC;IACjC,wEAAwE;IACxE,yEAAyE;IACzE,gDAAgD;IAChD,MAAM,WAAW,GAAG,eAAe,CAAC,WAAW,CAAC;IAChD,MAAM,KAAK,GAAG,IAAI,eAAe,CAAC;QAChC,OAAO,EAAE,UAAU,CAAC,KAAK;QACzB,GAAG,CAAC,eAAe,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,OAAO,EAAE,eAAe,CAAC,OAAO,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC;QACxE,GAAG,CAAC,WAAW,KAAK,SAAS;YAC3B,CAAC,CAAC,EAAE,WAAW,EAAE,WAAW,KAAK,IAAI,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,WAAW,EAAE;YAC7D,CAAC,CAAC,EAAE,CAAC;QACP,GAAG,CAAC,SAAS,KAAK,SAAS;YACzB,CAAC,CAAC;gBACE,SAAS,EAAE,EAAE,GAAG,CAAC,MAAM,KAAK,SAAS,CAAC,CAAC,CAAC,EAAE,MAAM,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE;gBAC1D,mBAAmB,EAAE,uBAAuB,CAAC,MAAM,IAAI,QAAQ,CAAC;aACjE;YACH,CAAC,CAAC,EAAE,CAAC;QACP,GAAG,CAAC,UAAU,CAAC,WAAW,KAAK,SAAS;YACtC,CAAC,CAAC,EAAE,WAAW,EAAE,UAAU,CAAC,WAAW,EAAE;YACzC,CAAC,CAAC,EAAE,CAAC;QACP,GAAG,CAAC,UAAU,CAAC,SAAS,KAAK,SAAS;YACpC,CAAC,CAAC,EAAE,SAAS,EAAE,UAAU,CAAC,SAAS,EAAE;YACrC,CAAC,CAAC,EAAE,CAAC;KACR,CAAC,CAAC;IACH,kEAAkE;IAClE,4BAA4B,CAAC,KAAK,CAAC,CAAC;IACpC,OAAO,KAAK,CAAC;AACf,CAAC"}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* MLX model repos are cached under `~/.hooman/cache/huggingface` (HF cache
|
|
3
|
+
* layout), shared with the llama-cpp provider's GGUF downloads.
|
|
4
|
+
*/
|
|
5
|
+
export declare const mlxCacheDir: () => string;
|
|
6
|
+
export type ParsedModelSpec = {
|
|
7
|
+
kind: "local";
|
|
8
|
+
path: string;
|
|
9
|
+
} | {
|
|
10
|
+
kind: "hub";
|
|
11
|
+
repo: string;
|
|
12
|
+
};
|
|
13
|
+
/**
|
|
14
|
+
* Parse an LLM `model` value into either a local MLX model directory or a
|
|
15
|
+
* Hugging Face repo designation. Accepted shapes (an optional `hf:` prefix is
|
|
16
|
+
* stripped):
|
|
17
|
+
* - `/abs/path/to/model-dir`, `./rel/model-dir`, `~/models/model-dir`
|
|
18
|
+
* (a directory containing `config.json` + safetensors weights)
|
|
19
|
+
* - `owner/repo` (an MLX-format repo, e.g. from `mlx-community`)
|
|
20
|
+
*/
|
|
21
|
+
export declare function parseModelSpec(model: string): ParsedModelSpec;
|
|
22
|
+
/**
|
|
23
|
+
* Resolve a model spec to a local MLX model directory, downloading the repo
|
|
24
|
+
* (config, safetensors weights, tokenizer files) from the Hugging Face Hub
|
|
25
|
+
* into the Hooman cache when needed. Weight shards are reported as shards of
|
|
26
|
+
* one download via `subscribeModelDownloadProgress`; the smaller JSON/
|
|
27
|
+
* tokenizer files download silently unless they exceed the reporter's
|
|
28
|
+
* blob-size threshold. Returns the snapshot directory containing
|
|
29
|
+
* `config.json`, which `mlex.js`'s `MlexModel.load` consumes directly.
|
|
30
|
+
*/
|
|
31
|
+
export declare function resolveModelDir(model: string, hfToken?: string): Promise<string>;
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
import { existsSync } from "fs";
|
|
2
|
+
import { homedir } from "os";
|
|
3
|
+
import { dirname, join, sep } from "path";
|
|
4
|
+
import { listFiles, modelInfo } from "@huggingface/hub";
|
|
5
|
+
import { cachePath } from "../../utils/paths.js";
|
|
6
|
+
import { downloadFileWithProgress } from "../hub-download.js";
|
|
7
|
+
/**
|
|
8
|
+
* MLX model repos are cached under `~/.hooman/cache/huggingface` (HF cache
|
|
9
|
+
* layout), shared with the llama-cpp provider's GGUF downloads.
|
|
10
|
+
*/
|
|
11
|
+
export const mlxCacheDir = () => join(cachePath(), "huggingface");
|
|
12
|
+
function expandHome(p) {
|
|
13
|
+
if (p === "~" || p.startsWith(`~${sep}`) || p.startsWith("~/")) {
|
|
14
|
+
return join(homedir(), p.slice(1));
|
|
15
|
+
}
|
|
16
|
+
return p;
|
|
17
|
+
}
|
|
18
|
+
/**
|
|
19
|
+
* Parse an LLM `model` value into either a local MLX model directory or a
|
|
20
|
+
* Hugging Face repo designation. Accepted shapes (an optional `hf:` prefix is
|
|
21
|
+
* stripped):
|
|
22
|
+
* - `/abs/path/to/model-dir`, `./rel/model-dir`, `~/models/model-dir`
|
|
23
|
+
* (a directory containing `config.json` + safetensors weights)
|
|
24
|
+
* - `owner/repo` (an MLX-format repo, e.g. from `mlx-community`)
|
|
25
|
+
*/
|
|
26
|
+
export function parseModelSpec(model) {
|
|
27
|
+
const spec = (model.startsWith("hf:") ? model.slice(3) : model).trim();
|
|
28
|
+
if (spec.length === 0) {
|
|
29
|
+
throw new Error("MLX model is not configured");
|
|
30
|
+
}
|
|
31
|
+
const expanded = expandHome(spec);
|
|
32
|
+
const looksLikePath = spec.startsWith("/") ||
|
|
33
|
+
spec.startsWith("./") ||
|
|
34
|
+
spec.startsWith("../") ||
|
|
35
|
+
spec.startsWith("~");
|
|
36
|
+
if (looksLikePath || existsSync(join(expanded, "config.json"))) {
|
|
37
|
+
return { kind: "local", path: expanded };
|
|
38
|
+
}
|
|
39
|
+
const segments = spec.split("/").filter((s) => s.length > 0);
|
|
40
|
+
if (segments.length === 2) {
|
|
41
|
+
return { kind: "hub", repo: spec };
|
|
42
|
+
}
|
|
43
|
+
throw new Error(`Invalid MLX model "${model}". Use a local MLX model directory ` +
|
|
44
|
+
`or an "owner/repo" Hugging Face repo (MLX format, e.g. mlx-community/...).`);
|
|
45
|
+
}
|
|
46
|
+
/**
|
|
47
|
+
* Repo files the MLX runtime needs: model config + weights + tokenizer
|
|
48
|
+
* assets. Everything else (README, images, .gitattributes) is skipped.
|
|
49
|
+
*/
|
|
50
|
+
const MODEL_FILE_EXTENSIONS = [
|
|
51
|
+
".safetensors",
|
|
52
|
+
".json",
|
|
53
|
+
".jinja",
|
|
54
|
+
".txt",
|
|
55
|
+
".model",
|
|
56
|
+
];
|
|
57
|
+
function isModelFile(path) {
|
|
58
|
+
const lower = path.toLowerCase();
|
|
59
|
+
if (lower === ".gitattributes") {
|
|
60
|
+
return false;
|
|
61
|
+
}
|
|
62
|
+
return MODEL_FILE_EXTENSIONS.some((ext) => lower.endsWith(ext));
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* Resolve a model spec to a local MLX model directory, downloading the repo
|
|
66
|
+
* (config, safetensors weights, tokenizer files) from the Hugging Face Hub
|
|
67
|
+
* into the Hooman cache when needed. Weight shards are reported as shards of
|
|
68
|
+
* one download via `subscribeModelDownloadProgress`; the smaller JSON/
|
|
69
|
+
* tokenizer files download silently unless they exceed the reporter's
|
|
70
|
+
* blob-size threshold. Returns the snapshot directory containing
|
|
71
|
+
* `config.json`, which `mlex.js`'s `MlexModel.load` consumes directly.
|
|
72
|
+
*/
|
|
73
|
+
export async function resolveModelDir(model, hfToken) {
|
|
74
|
+
const parsed = parseModelSpec(model);
|
|
75
|
+
if (parsed.kind === "local") {
|
|
76
|
+
if (!existsSync(join(parsed.path, "config.json"))) {
|
|
77
|
+
throw new Error(`MLX model directory not found (no config.json): ${parsed.path}`);
|
|
78
|
+
}
|
|
79
|
+
return parsed.path;
|
|
80
|
+
}
|
|
81
|
+
const accessToken = hfToken?.trim() || process.env.HF_TOKEN?.trim();
|
|
82
|
+
const credentials = accessToken ? { accessToken } : {};
|
|
83
|
+
const cacheDir = mlxCacheDir();
|
|
84
|
+
// Pin every file to the repo's current head commit: the HF cache layout
|
|
85
|
+
// names snapshot dirs after the revision each file resolved to, so
|
|
86
|
+
// un-pinned multi-file downloads would scatter across snapshot dirs and
|
|
87
|
+
// never form one complete model directory.
|
|
88
|
+
const info = await modelInfo({
|
|
89
|
+
name: parsed.repo,
|
|
90
|
+
additionalFields: ["sha"],
|
|
91
|
+
...(accessToken ? { accessToken } : {}),
|
|
92
|
+
});
|
|
93
|
+
const revision = info.sha;
|
|
94
|
+
if (typeof revision !== "string" || revision.length === 0) {
|
|
95
|
+
throw new Error(`Cannot resolve the current revision of Hugging Face repo "${parsed.repo}".`);
|
|
96
|
+
}
|
|
97
|
+
const files = [];
|
|
98
|
+
for await (const entry of listFiles({
|
|
99
|
+
repo: parsed.repo,
|
|
100
|
+
recursive: true,
|
|
101
|
+
revision,
|
|
102
|
+
...(accessToken ? { accessToken } : {}),
|
|
103
|
+
})) {
|
|
104
|
+
if (entry.type === "file" && isModelFile(entry.path)) {
|
|
105
|
+
files.push(entry.path);
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
if (!files.includes("config.json")) {
|
|
109
|
+
throw new Error(`Hugging Face repo "${parsed.repo}" does not look like an MLX model ` +
|
|
110
|
+
`(no config.json). Use an MLX-format repo, e.g. from mlx-community.`);
|
|
111
|
+
}
|
|
112
|
+
const weights = files.filter((f) => f.toLowerCase().endsWith(".safetensors"));
|
|
113
|
+
if (weights.length === 0) {
|
|
114
|
+
throw new Error(`No .safetensors weights found in Hugging Face repo "${parsed.repo}".`);
|
|
115
|
+
}
|
|
116
|
+
// Small metadata files first (cheap, near-instant), then the weight shards
|
|
117
|
+
// with shard-indexed progress so the UI shows "shard i of n".
|
|
118
|
+
const metadata = files.filter((f) => !f.toLowerCase().endsWith(".safetensors"));
|
|
119
|
+
let configPath;
|
|
120
|
+
for (const filePath of metadata.sort()) {
|
|
121
|
+
const local = await downloadFileWithProgress({
|
|
122
|
+
repo: parsed.repo,
|
|
123
|
+
filePath,
|
|
124
|
+
cacheDir,
|
|
125
|
+
credentials,
|
|
126
|
+
model,
|
|
127
|
+
revision,
|
|
128
|
+
});
|
|
129
|
+
if (filePath === "config.json") {
|
|
130
|
+
configPath = local;
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
weights.sort();
|
|
134
|
+
for (let i = 0; i < weights.length; i++) {
|
|
135
|
+
await downloadFileWithProgress({
|
|
136
|
+
repo: parsed.repo,
|
|
137
|
+
filePath: weights[i],
|
|
138
|
+
cacheDir,
|
|
139
|
+
credentials,
|
|
140
|
+
model,
|
|
141
|
+
revision,
|
|
142
|
+
...(weights.length > 1
|
|
143
|
+
? { shard: { index: i + 1, total: weights.length } }
|
|
144
|
+
: {}),
|
|
145
|
+
});
|
|
146
|
+
}
|
|
147
|
+
return dirname(configPath);
|
|
148
|
+
}
|
|
149
|
+
//# sourceMappingURL=resolve-model.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"resolve-model.js","sourceRoot":"","sources":["../../../../src/core/models/mlx/resolve-model.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,UAAU,EAAE,MAAM,IAAI,CAAC;AAChC,OAAO,EAAE,OAAO,EAAE,MAAM,IAAI,CAAC;AAC7B,OAAO,EAAE,OAAO,EAAE,IAAI,EAAE,GAAG,EAAE,MAAM,MAAM,CAAC;AAC1C,OAAO,EAAE,SAAS,EAAE,SAAS,EAAE,MAAM,kBAAkB,CAAC;AACxD,OAAO,EAAE,SAAS,EAAE,MAAM,sBAAsB,CAAC;AACjD,OAAO,EAAE,wBAAwB,EAAE,MAAM,oBAAoB,CAAC;AAE9D;;;GAGG;AACH,MAAM,CAAC,MAAM,WAAW,GAAG,GAAG,EAAE,CAAC,IAAI,CAAC,SAAS,EAAE,EAAE,aAAa,CAAC,CAAC;AAKlE,SAAS,UAAU,CAAC,CAAS;IAC3B,IAAI,CAAC,KAAK,GAAG,IAAI,CAAC,CAAC,UAAU,CAAC,IAAI,GAAG,EAAE,CAAC,IAAI,CAAC,CAAC,UAAU,CAAC,IAAI,CAAC,EAAE,CAAC;QAC/D,OAAO,IAAI,CAAC,OAAO,EAAE,EAAE,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC;IACrC,CAAC;IACD,OAAO,CAAC,CAAC;AACX,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,cAAc,CAAC,KAAa;IAC1C,MAAM,IAAI,GAAG,CAAC,KAAK,CAAC,UAAU,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,IAAI,EAAE,CAAC;IACvE,IAAI,IAAI,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACtB,MAAM,IAAI,KAAK,CAAC,6BAA6B,CAAC,CAAC;IACjD,CAAC;IACD,MAAM,QAAQ,GAAG,UAAU,CAAC,IAAI,CAAC,CAAC;IAClC,MAAM,aAAa,GACjB,IAAI,CAAC,UAAU,CAAC,GAAG,CAAC;QACpB,IAAI,CAAC,UAAU,CAAC,IAAI,CAAC;QACrB,IAAI,CAAC,UAAU,CAAC,KAAK,CAAC;QACtB,IAAI,CAAC,UAAU,CAAC,GAAG,CAAC,CAAC;IACvB,IAAI,aAAa,IAAI,UAAU,CAAC,IAAI,CAAC,QAAQ,EAAE,aAAa,CAAC,CAAC,EAAE,CAAC;QAC/D,OAAO,EAAE,IAAI,EAAE,OAAO,EAAE,IAAI,EAAE,QAAQ,EAAE,CAAC;IAC3C,CAAC;IACD,MAAM,QAAQ,GAAG,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IAC7D,IAAI,QAAQ,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QAC1B,OAAO,EAAE,IAAI,EAAE,KAAK,EAAE,IAAI,EAAE,IAAI,EAAE,CAAC;IACrC,CAAC;IACD,MAAM,IAAI,KAAK,CACb,sBAAsB,KAAK,qCAAqC;QAC9D,4EAA4E,CAC/E,CAAC;AACJ,CAAC;AAED;;;GAGG;AACH,MAAM,qBAAqB,GAAG;IAC5B,cAAc;IACd,OAAO;IACP,QAAQ;IACR,MAAM;IACN,QAAQ;CACT,CAAC;AAEF,SAAS,WAAW,CAAC,IAAY;IAC/B,MAAM,KAAK,GAAG,IAAI,CAAC,WAAW,EAAE,CAAC;IACjC,IAAI,KAAK,KAAK,gBAAgB,EAAE,CAAC;QAC/B,OAAO,KAAK,CAAC;IACf,CAAC;IACD,OAAO,qBAAqB,CAAC,IAAI,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,KAAK,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC,CAAC;AAClE,CAAC;AAED;;;;;;;;GAQG;AACH,MAAM,CAAC,KAAK,UAAU,eAAe,CACnC,KAAa,EACb,OAAgB;IAEhB,MAAM,MAAM,GAAG,cAAc,CAAC,KAAK,CAAC,CAAC;IACrC,IAAI,MAAM,CAAC,IAAI,KAAK,OAAO,EAAE,CAAC;QAC5B,IAAI,CAAC,UAAU,CAAC,IAAI,CAAC,MAAM,CAAC,IAAI,EAAE,aAAa,CAAC,CAAC,EAAE,CAAC;YAClD,MAAM,IAAI,KAAK,CACb,mDAAmD,MAAM,CAAC,IAAI,EAAE,CACjE,CAAC;QACJ,CAAC;QACD,OAAO,MAAM,CAAC,IAAI,CAAC;IACrB,CAAC;IACD,MAAM,WAAW,GAAG,OAAO,EAAE,IAAI,EAAE,IAAI,OAAO,CAAC,GAAG,CAAC,QAAQ,EAAE,IAAI,EAAE,CAAC;IACpE,MAAM,WAAW,GAAG,WAAW,CAAC,CAAC,CAAC,EAAE,WAAW,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC;IACvD,MAAM,QAAQ,GAAG,WAAW,EAAE,CAAC;IAE/B,wEAAwE;IACxE,mEAAmE;IACnE,wEAAwE;IACxE,2CAA2C;IAC3C,MAAM,IAAI,GAAG,MAAM,SAAS,CAAC;QAC3B,IAAI,EAAE,MAAM,CAAC,IAAI;QACjB,gBAAgB,EAAE,CAAC,KAAK,CAAC;QACzB,GAAG,CAAC,WAAW,CAAC,CAAC,CAAC,EAAE,WAAW,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC;KACxC,CAAC,CAAC;IACH,MAAM,QAAQ,GAAG,IAAI,CAAC,GAAG,CAAC;IAC1B,IAAI,OAAO,QAAQ,KAAK,QAAQ,IAAI,QAAQ,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QAC1D,MAAM,IAAI,KAAK,CACb,6DAA6D,MAAM,CAAC,IAAI,IAAI,CAC7E,CAAC;IACJ,CAAC;IAED,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,IAAI,KAAK,EAAE,MAAM,KAAK,IAAI,SAAS,CAAC;QAClC,IAAI,EAAE,MAAM,CAAC,IAAI;QACjB,SAAS,EAAE,IAAI;QACf,QAAQ;QACR,GAAG,CAAC,WAAW,CAAC,CAAC,CAAC,EAAE,WAAW,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC;KACxC,CAAC,EAAE,CAAC;QACH,IAAI,KAAK,CAAC,IAAI,KAAK,MAAM,IAAI,WAAW,CAAC,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC;YACrD,KAAK,CAAC,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC;QACzB,CAAC;IACH,CAAC;IACD,IAAI,CAAC,KAAK,CAAC,QAAQ,CAAC,aAAa,CAAC,EAAE,CAAC;QACnC,MAAM,IAAI,KAAK,CACb,sBAAsB,MAAM,CAAC,IAAI,oCAAoC;YACnE,oEAAoE,CACvE,CAAC;IACJ,CAAC;IACD,MAAM,OAAO,GAAG,KAAK,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,WAAW,EAAE,CAAC,QAAQ,CAAC,cAAc,CAAC,CAAC,CAAC;IAC9E,IAAI,OAAO,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACzB,MAAM,IAAI,KAAK,CACb,uDAAuD,MAAM,CAAC,IAAI,IAAI,CACvE,CAAC;IACJ,CAAC;IAED,2EAA2E;IAC3E,8DAA8D;IAC9D,MAAM,QAAQ,GAAG,KAAK,CAAC,MAAM,CAC3B,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,WAAW,EAAE,CAAC,QAAQ,CAAC,cAAc,CAAC,CACjD,CAAC;IACF,IAAI,UAA8B,CAAC;IACnC,KAAK,MAAM,QAAQ,IAAI,QAAQ,CAAC,IAAI,EAAE,EAAE,CAAC;QACvC,MAAM,KAAK,GAAG,MAAM,wBAAwB,CAAC;YAC3C,IAAI,EAAE,MAAM,CAAC,IAAI;YACjB,QAAQ;YACR,QAAQ;YACR,WAAW;YACX,KAAK;YACL,QAAQ;SACT,CAAC,CAAC;QACH,IAAI,QAAQ,KAAK,aAAa,EAAE,CAAC;YAC/B,UAAU,GAAG,KAAK,CAAC;QACrB,CAAC;IACH,CAAC;IACD,OAAO,CAAC,IAAI,EAAE,CAAC;IACf,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,OAAO,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;QACxC,MAAM,wBAAwB,CAAC;YAC7B,IAAI,EAAE,MAAM,CAAC,IAAI;YACjB,QAAQ,EAAE,OAAO,CAAC,CAAC,CAAE;YACrB,QAAQ;YACR,WAAW;YACX,KAAK;YACL,QAAQ;YACR,GAAG,CAAC,OAAO,CAAC,MAAM,GAAG,CAAC;gBACpB,CAAC,CAAC,EAAE,KAAK,EAAE,EAAE,KAAK,EAAE,CAAC,GAAG,CAAC,EAAE,KAAK,EAAE,OAAO,CAAC,MAAM,EAAE,EAAE;gBACpD,CAAC,CAAC,EAAE,CAAC;SACR,CAAC,CAAC;IACL,CAAC;IACD,OAAO,OAAO,CAAC,UAAW,CAAC,CAAC;AAC9B,CAAC"}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import { Model } from "@strands-agents/sdk";
|
|
2
|
+
import type { BaseModelConfig, Message, StreamOptions } from "@strands-agents/sdk";
|
|
3
|
+
import type { ModelStreamEvent } from "@strands-agents/sdk";
|
|
4
|
+
import type { MlxPromptCacheConfig } from "../types.js";
|
|
5
|
+
export interface MlxModelConfig extends BaseModelConfig {
|
|
6
|
+
/**
|
|
7
|
+
* Model spec: `owner/repo` Hugging Face repo in MLX format (e.g.
|
|
8
|
+
* `mlx-community/...`) or a local MLX model directory containing
|
|
9
|
+
* `config.json` + safetensors weights.
|
|
10
|
+
*/
|
|
11
|
+
modelId?: string;
|
|
12
|
+
/** Hugging Face access token for gated/private repos (falls back to `HF_TOKEN`). */
|
|
13
|
+
hfToken?: string;
|
|
14
|
+
/**
|
|
15
|
+
* Whether turns may reuse KV state from mlex's internal prompt-cache pool
|
|
16
|
+
* (prefix matching against previous calls), applied once when the model
|
|
17
|
+
* is loaded. `undefined`/`false` disables caching entirely (every
|
|
18
|
+
* generate call forwards `promptCache: false`); an object (even `{}`)
|
|
19
|
+
* enables it, with its fields overriding mlex's own pool-sizing defaults.
|
|
20
|
+
*/
|
|
21
|
+
promptCache?: MlxPromptCacheConfig | false;
|
|
22
|
+
/**
|
|
23
|
+
* Thinking controls. Presence enables reasoning (`enableThinking: true` on
|
|
24
|
+
* the chat template) with `effort` capping the reasoning span via
|
|
25
|
+
* `reasoningBudgetTokens`. Absence disables it: the template renders with
|
|
26
|
+
* thinking off and reasoning content is dropped.
|
|
27
|
+
*/
|
|
28
|
+
reasoning?: {
|
|
29
|
+
effort?: "minimal" | "low" | "medium" | "high";
|
|
30
|
+
};
|
|
31
|
+
/** Cap on reasoning-span tokens while thinking is enabled. */
|
|
32
|
+
thoughtBudgetTokens?: number;
|
|
33
|
+
}
|
|
34
|
+
/** Strands {@link Model} backed by in-process Apple MLX via `mlex.js`. */
|
|
35
|
+
export declare class StrandsMlxModel extends Model<MlxModelConfig> {
|
|
36
|
+
private config;
|
|
37
|
+
private modelPromise;
|
|
38
|
+
constructor(config: MlxModelConfig);
|
|
39
|
+
updateConfig(modelConfig: MlxModelConfig): void;
|
|
40
|
+
getConfig(): MlxModelConfig;
|
|
41
|
+
private getModel;
|
|
42
|
+
private initModel;
|
|
43
|
+
private buildGenerateOptions;
|
|
44
|
+
stream(messages: Message[], options?: StreamOptions): AsyncIterable<ModelStreamEvent>;
|
|
45
|
+
}
|
|
@@ -0,0 +1,439 @@
|
|
|
1
|
+
import { Model, ModelError } from "@strands-agents/sdk";
|
|
2
|
+
import { ModelContentBlockDeltaEvent, ModelContentBlockStartEvent, ModelContentBlockStopEvent, ModelMessageStartEvent, ModelMessageStopEvent, ModelMetadataEvent, } from "@strands-agents/sdk";
|
|
3
|
+
import { resolveModelDir } from "./resolve-model.js";
|
|
4
|
+
/**
|
|
5
|
+
* mlex.js caps generation at 256 tokens when `maxTokens` is unset — far too
|
|
6
|
+
* low for agent turns — so apply our own default instead.
|
|
7
|
+
*/
|
|
8
|
+
const DEFAULT_MAX_TOKENS = 8192;
|
|
9
|
+
/**
|
|
10
|
+
* Loaded MLX models are expensive (weights in unified memory), so share them
|
|
11
|
+
* process-wide keyed by resolved model directory. mlex.js is stateless like
|
|
12
|
+
* the OpenAI/Anthropic APIs — every `generate` call takes the full
|
|
13
|
+
* transcript, and an internal prompt-cache pool transparently reuses KV
|
|
14
|
+
* state for whatever prefix a previous call already computed.
|
|
15
|
+
*/
|
|
16
|
+
const loadedModelPromises = new Map();
|
|
17
|
+
/**
|
|
18
|
+
* Streamed token classification is best-effort at token granularity, so the
|
|
19
|
+
* reasoning-span markers themselves (`<think>`/`</think>`, Gemma4's channel
|
|
20
|
+
* markers) can arrive inside `kind: "reasoning"` deltas. The final result's
|
|
21
|
+
* `reasoning` field is marker-stripped — only the incremental stream carries
|
|
22
|
+
* them — so strip them here to keep the reasoning block to the chain of
|
|
23
|
+
* thought.
|
|
24
|
+
*/
|
|
25
|
+
const REASONING_MARKER_RE = /<\/?think>|<\|channel\|?>thought|<\/?channel\|>/g;
|
|
26
|
+
function extractSystemText(system) {
|
|
27
|
+
if (system === undefined) {
|
|
28
|
+
return undefined;
|
|
29
|
+
}
|
|
30
|
+
if (typeof system === "string") {
|
|
31
|
+
return system;
|
|
32
|
+
}
|
|
33
|
+
const parts = [];
|
|
34
|
+
for (const block of system) {
|
|
35
|
+
if (block.type === "textBlock") {
|
|
36
|
+
parts.push(block.text);
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
const joined = parts.join("\n").trim();
|
|
40
|
+
return joined.length > 0 ? joined : undefined;
|
|
41
|
+
}
|
|
42
|
+
function toolResultToText(block) {
|
|
43
|
+
const parts = [];
|
|
44
|
+
for (const c of block.content) {
|
|
45
|
+
if (c.type === "textBlock") {
|
|
46
|
+
parts.push(c.text);
|
|
47
|
+
}
|
|
48
|
+
else if (c.type === "jsonBlock") {
|
|
49
|
+
parts.push(JSON.stringify(c.json));
|
|
50
|
+
}
|
|
51
|
+
else if (c.type === "imageBlock") {
|
|
52
|
+
parts.push("(mlx: image tool results are not supported)");
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
const joined = parts.join("\n");
|
|
56
|
+
if (joined.length > 0) {
|
|
57
|
+
return joined;
|
|
58
|
+
}
|
|
59
|
+
return block.status === "error" ? "(tool error)" : "";
|
|
60
|
+
}
|
|
61
|
+
/**
|
|
62
|
+
* Convert Strands conversation history to mlex.js `JsChatMessage[]` — the
|
|
63
|
+
* same OpenAI-style layout: assistant toolUse blocks become `toolCalls` on
|
|
64
|
+
* the assistant message and the matching toolResult blocks (which Strands
|
|
65
|
+
* puts on the following user message) become `role: "tool"` messages
|
|
66
|
+
* referencing the call id. Image blocks with inline bytes are attached as
|
|
67
|
+
* `images` when the loaded checkpoint accepts them.
|
|
68
|
+
*/
|
|
69
|
+
function strandsMessagesToHistory(messages, systemText, supportsImages) {
|
|
70
|
+
const history = [];
|
|
71
|
+
if (systemText) {
|
|
72
|
+
history.push({ role: "system", content: systemText });
|
|
73
|
+
}
|
|
74
|
+
const appendText = (msg, text) => {
|
|
75
|
+
msg.content = msg.content.length > 0 ? `${msg.content}\n${text}` : text;
|
|
76
|
+
};
|
|
77
|
+
const blockToText = (block) => {
|
|
78
|
+
if (block.type === "textBlock") {
|
|
79
|
+
return block.text;
|
|
80
|
+
}
|
|
81
|
+
if (block.type === "imageBlock" ||
|
|
82
|
+
block.type === "videoBlock" ||
|
|
83
|
+
block.type === "documentBlock") {
|
|
84
|
+
return `(mlx: ${block.type} content is not supported by this model)`;
|
|
85
|
+
}
|
|
86
|
+
return undefined;
|
|
87
|
+
};
|
|
88
|
+
for (const msg of messages) {
|
|
89
|
+
for (const block of msg.content) {
|
|
90
|
+
if (msg.role === "assistant") {
|
|
91
|
+
if (block.type === "toolUseBlock") {
|
|
92
|
+
const b = block;
|
|
93
|
+
const call = {
|
|
94
|
+
id: b.toolUseId,
|
|
95
|
+
name: b.name,
|
|
96
|
+
argumentsJson: JSON.stringify(b.input ?? {}),
|
|
97
|
+
};
|
|
98
|
+
const last = history.at(-1);
|
|
99
|
+
if (last?.role === "assistant") {
|
|
100
|
+
last.toolCalls = [...(last.toolCalls ?? []), call];
|
|
101
|
+
}
|
|
102
|
+
else {
|
|
103
|
+
history.push({ role: "assistant", content: "", toolCalls: [call] });
|
|
104
|
+
}
|
|
105
|
+
continue;
|
|
106
|
+
}
|
|
107
|
+
if (block.type === "reasoningBlock") {
|
|
108
|
+
// Prior turns' chain of thought is not replayed into the context.
|
|
109
|
+
continue;
|
|
110
|
+
}
|
|
111
|
+
const text = blockToText(block);
|
|
112
|
+
if (text !== undefined && text.length > 0) {
|
|
113
|
+
const last = history.at(-1);
|
|
114
|
+
if (last?.role === "assistant") {
|
|
115
|
+
appendText(last, text);
|
|
116
|
+
}
|
|
117
|
+
else {
|
|
118
|
+
history.push({ role: "assistant", content: text });
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
continue;
|
|
122
|
+
}
|
|
123
|
+
// user role
|
|
124
|
+
if (block.type === "toolResultBlock") {
|
|
125
|
+
const b = block;
|
|
126
|
+
history.push({
|
|
127
|
+
role: "tool",
|
|
128
|
+
content: toolResultToText(b),
|
|
129
|
+
toolCallId: b.toolUseId,
|
|
130
|
+
});
|
|
131
|
+
continue;
|
|
132
|
+
}
|
|
133
|
+
if (block.type === "imageBlock" && supportsImages) {
|
|
134
|
+
const source = block.source;
|
|
135
|
+
if (source.bytes !== undefined) {
|
|
136
|
+
const last = history.at(-1);
|
|
137
|
+
const target = last?.role === "user" && last.toolCalls === undefined
|
|
138
|
+
? last
|
|
139
|
+
: (history.push({ role: "user", content: "" }),
|
|
140
|
+
history.at(-1));
|
|
141
|
+
target.images = [...(target.images ?? []), Buffer.from(source.bytes)];
|
|
142
|
+
continue;
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
const text = blockToText(block);
|
|
146
|
+
if (text !== undefined && text.length > 0) {
|
|
147
|
+
const last = history.at(-1);
|
|
148
|
+
if (last?.role === "user" && last.toolCalls === undefined) {
|
|
149
|
+
appendText(last, text);
|
|
150
|
+
}
|
|
151
|
+
else {
|
|
152
|
+
history.push({ role: "user", content: text });
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
return history;
|
|
158
|
+
}
|
|
159
|
+
/**
|
|
160
|
+
* Convert Strands tool specs to mlex.js `JsTool`s. mlex takes standard JSON
|
|
161
|
+
* Schema `parameters` directly — no GBNF conversion or schema mangling.
|
|
162
|
+
*/
|
|
163
|
+
function strandsToolsToDefinitions(toolSpecs) {
|
|
164
|
+
if (!toolSpecs?.length) {
|
|
165
|
+
return undefined;
|
|
166
|
+
}
|
|
167
|
+
return toolSpecs.map((spec) => ({
|
|
168
|
+
name: spec.name,
|
|
169
|
+
...(spec.description ? { description: spec.description } : {}),
|
|
170
|
+
parameters: spec.inputSchema ?? { type: "object", properties: {} },
|
|
171
|
+
}));
|
|
172
|
+
}
|
|
173
|
+
/** Strands {@link Model} backed by in-process Apple MLX via `mlex.js`. */
|
|
174
|
+
export class StrandsMlxModel extends Model {
|
|
175
|
+
config;
|
|
176
|
+
modelPromise;
|
|
177
|
+
constructor(config) {
|
|
178
|
+
super();
|
|
179
|
+
this.config = { ...config };
|
|
180
|
+
}
|
|
181
|
+
updateConfig(modelConfig) {
|
|
182
|
+
const modelKey = (c) => JSON.stringify([c.modelId, c.promptCache ?? null]);
|
|
183
|
+
const before = modelKey(this.config);
|
|
184
|
+
this.config = { ...this.config, ...modelConfig };
|
|
185
|
+
if (modelKey(this.config) !== before) {
|
|
186
|
+
// Weights stay cached process-wide; just re-resolve on the next stream.
|
|
187
|
+
this.modelPromise = undefined;
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
getConfig() {
|
|
191
|
+
return { ...this.config };
|
|
192
|
+
}
|
|
193
|
+
getModel() {
|
|
194
|
+
this.modelPromise ??= this.initModel();
|
|
195
|
+
return this.modelPromise;
|
|
196
|
+
}
|
|
197
|
+
async initModel() {
|
|
198
|
+
const modelId = this.config.modelId;
|
|
199
|
+
if (!modelId) {
|
|
200
|
+
throw new ModelError("MLX model is not configured");
|
|
201
|
+
}
|
|
202
|
+
const modelDir = await resolveModelDir(modelId, this.config.hfToken);
|
|
203
|
+
const promptCache = this.config.promptCache;
|
|
204
|
+
const cacheEnabled = promptCache !== undefined && promptCache !== false;
|
|
205
|
+
// The prompt-cache pool is sized once at load time; key the shared
|
|
206
|
+
// instance by pool config too so two provider configs pointing at the
|
|
207
|
+
// same model directory with different sizing each get their own
|
|
208
|
+
// session instead of silently reusing whichever loaded first.
|
|
209
|
+
const cacheKey = `${modelDir}\u0000${JSON.stringify(cacheEnabled ? promptCache : false)}`;
|
|
210
|
+
let promise = loadedModelPromises.get(cacheKey);
|
|
211
|
+
if (!promise) {
|
|
212
|
+
promise = (async () => {
|
|
213
|
+
const { MlexModel } = await import("mlex.js");
|
|
214
|
+
return MlexModel.load(modelDir, cacheEnabled
|
|
215
|
+
? {
|
|
216
|
+
...(promptCache.maxEntries !== undefined
|
|
217
|
+
? { maxEntries: promptCache.maxEntries }
|
|
218
|
+
: {}),
|
|
219
|
+
...(promptCache.ttl !== undefined
|
|
220
|
+
? { ttlSeconds: promptCache.ttl }
|
|
221
|
+
: {}),
|
|
222
|
+
...(promptCache.minTokens !== undefined
|
|
223
|
+
? { minCacheableTokens: promptCache.minTokens }
|
|
224
|
+
: {}),
|
|
225
|
+
}
|
|
226
|
+
: undefined);
|
|
227
|
+
})();
|
|
228
|
+
loadedModelPromises.set(cacheKey, promise);
|
|
229
|
+
promise.catch(() => loadedModelPromises.delete(cacheKey));
|
|
230
|
+
}
|
|
231
|
+
return promise;
|
|
232
|
+
}
|
|
233
|
+
buildGenerateOptions(toolSpecs) {
|
|
234
|
+
const reasoningEnabled = this.config.reasoning !== undefined;
|
|
235
|
+
const tools = strandsToolsToDefinitions(toolSpecs);
|
|
236
|
+
return {
|
|
237
|
+
maxTokens: this.config.maxTokens ?? DEFAULT_MAX_TOKENS,
|
|
238
|
+
...(this.config.temperature !== undefined
|
|
239
|
+
? { temperature: this.config.temperature }
|
|
240
|
+
: {}),
|
|
241
|
+
...(this.config.topP !== undefined ? { topP: this.config.topP } : {}),
|
|
242
|
+
...(tools ? { tools } : {}),
|
|
243
|
+
...(this.config.promptCache === undefined || this.config.promptCache === false
|
|
244
|
+
? { promptCache: false }
|
|
245
|
+
: {}),
|
|
246
|
+
// Presence of `reasoning` opts into thinking with a token budget;
|
|
247
|
+
// absence pins it off (the template default for every supported
|
|
248
|
+
// family, made explicit).
|
|
249
|
+
enableThinking: reasoningEnabled,
|
|
250
|
+
...(reasoningEnabled && this.config.thoughtBudgetTokens !== undefined
|
|
251
|
+
? { reasoningBudgetTokens: this.config.thoughtBudgetTokens }
|
|
252
|
+
: {}),
|
|
253
|
+
};
|
|
254
|
+
}
|
|
255
|
+
async *stream(messages, options) {
|
|
256
|
+
let model;
|
|
257
|
+
try {
|
|
258
|
+
model = await this.getModel();
|
|
259
|
+
}
|
|
260
|
+
catch (e) {
|
|
261
|
+
// Let a failed init (bad model spec, download failure) be retried.
|
|
262
|
+
this.modelPromise = undefined;
|
|
263
|
+
if (e instanceof ModelError) {
|
|
264
|
+
throw e;
|
|
265
|
+
}
|
|
266
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
267
|
+
throw new ModelError(`MLX initialization failed: ${msg}`, { cause: e });
|
|
268
|
+
}
|
|
269
|
+
const systemText = extractSystemText(options?.systemPrompt);
|
|
270
|
+
const history = strandsMessagesToHistory(messages, systemText, model.supportsImages());
|
|
271
|
+
const generateOptions = this.buildGenerateOptions(options?.toolSpecs);
|
|
272
|
+
// Adapt mlex's onToken callback + result promise to a pull-based queue
|
|
273
|
+
// the generator below can drain with backpressure-free yields.
|
|
274
|
+
const queue = [];
|
|
275
|
+
let notify;
|
|
276
|
+
const push = (item) => {
|
|
277
|
+
queue.push(item);
|
|
278
|
+
notify?.();
|
|
279
|
+
notify = undefined;
|
|
280
|
+
};
|
|
281
|
+
const waitForItem = () => queue.length > 0
|
|
282
|
+
? Promise.resolve()
|
|
283
|
+
: new Promise((resolve) => {
|
|
284
|
+
notify = resolve;
|
|
285
|
+
});
|
|
286
|
+
model
|
|
287
|
+
.generate(history, generateOptions, (err, token) => {
|
|
288
|
+
if (err) {
|
|
289
|
+
push({ error: err });
|
|
290
|
+
}
|
|
291
|
+
else {
|
|
292
|
+
push({ token });
|
|
293
|
+
}
|
|
294
|
+
})
|
|
295
|
+
.then((result) => push({ result }), (error) => push({ error }));
|
|
296
|
+
yield new ModelMessageStartEvent({
|
|
297
|
+
type: "modelMessageStartEvent",
|
|
298
|
+
role: "assistant",
|
|
299
|
+
});
|
|
300
|
+
let textBlockOpen = false;
|
|
301
|
+
let reasoningBlockOpen = false;
|
|
302
|
+
const closeOpenBlock = () => {
|
|
303
|
+
if (textBlockOpen || reasoningBlockOpen) {
|
|
304
|
+
textBlockOpen = false;
|
|
305
|
+
reasoningBlockOpen = false;
|
|
306
|
+
return new ModelContentBlockStopEvent({
|
|
307
|
+
type: "modelContentBlockStopEvent",
|
|
308
|
+
});
|
|
309
|
+
}
|
|
310
|
+
return undefined;
|
|
311
|
+
};
|
|
312
|
+
try {
|
|
313
|
+
let finalResult;
|
|
314
|
+
while (finalResult === undefined) {
|
|
315
|
+
await waitForItem();
|
|
316
|
+
while (queue.length > 0 && finalResult === undefined) {
|
|
317
|
+
const item = queue.shift();
|
|
318
|
+
if ("error" in item) {
|
|
319
|
+
const e = item.error;
|
|
320
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
321
|
+
throw new ModelError(`MLX generation error: ${msg}`, { cause: e });
|
|
322
|
+
}
|
|
323
|
+
if ("result" in item) {
|
|
324
|
+
finalResult = item.result;
|
|
325
|
+
break;
|
|
326
|
+
}
|
|
327
|
+
const token = item.token;
|
|
328
|
+
if (token.kind === "toolCall") {
|
|
329
|
+
// Raw, not-yet-parsed tool-call syntax; the parsed calls arrive
|
|
330
|
+
// on the final result.
|
|
331
|
+
continue;
|
|
332
|
+
}
|
|
333
|
+
const isReasoning = token.kind === "reasoning";
|
|
334
|
+
const text = isReasoning
|
|
335
|
+
? token.text.replace(REASONING_MARKER_RE, "")
|
|
336
|
+
: token.text;
|
|
337
|
+
if (text.length === 0) {
|
|
338
|
+
continue;
|
|
339
|
+
}
|
|
340
|
+
// Templates emit a newline right after the thinking marker; don't
|
|
341
|
+
// open the reasoning block on whitespace alone.
|
|
342
|
+
if (isReasoning && !reasoningBlockOpen && text.trim().length === 0) {
|
|
343
|
+
continue;
|
|
344
|
+
}
|
|
345
|
+
if (isReasoning ? textBlockOpen : reasoningBlockOpen) {
|
|
346
|
+
const stop = closeOpenBlock();
|
|
347
|
+
if (stop) {
|
|
348
|
+
yield stop;
|
|
349
|
+
}
|
|
350
|
+
}
|
|
351
|
+
if (!textBlockOpen && !reasoningBlockOpen) {
|
|
352
|
+
yield new ModelContentBlockStartEvent({
|
|
353
|
+
type: "modelContentBlockStartEvent",
|
|
354
|
+
});
|
|
355
|
+
if (isReasoning) {
|
|
356
|
+
reasoningBlockOpen = true;
|
|
357
|
+
}
|
|
358
|
+
else {
|
|
359
|
+
textBlockOpen = true;
|
|
360
|
+
}
|
|
361
|
+
}
|
|
362
|
+
yield new ModelContentBlockDeltaEvent({
|
|
363
|
+
type: "modelContentBlockDeltaEvent",
|
|
364
|
+
delta: isReasoning
|
|
365
|
+
? { type: "reasoningContentDelta", text }
|
|
366
|
+
: { type: "textDelta", text },
|
|
367
|
+
});
|
|
368
|
+
}
|
|
369
|
+
}
|
|
370
|
+
const stop = closeOpenBlock();
|
|
371
|
+
if (stop) {
|
|
372
|
+
yield stop;
|
|
373
|
+
}
|
|
374
|
+
for (const call of finalResult.toolCalls) {
|
|
375
|
+
let input = {};
|
|
376
|
+
try {
|
|
377
|
+
input = JSON.parse(call.argumentsJson) ?? {};
|
|
378
|
+
}
|
|
379
|
+
catch {
|
|
380
|
+
// Unparseable arguments degrade to an empty object; the tool's own
|
|
381
|
+
// schema validation will surface the problem to the model.
|
|
382
|
+
}
|
|
383
|
+
yield new ModelContentBlockStartEvent({
|
|
384
|
+
type: "modelContentBlockStartEvent",
|
|
385
|
+
start: {
|
|
386
|
+
type: "toolUseStart",
|
|
387
|
+
name: call.name,
|
|
388
|
+
toolUseId: call.id,
|
|
389
|
+
},
|
|
390
|
+
});
|
|
391
|
+
yield new ModelContentBlockDeltaEvent({
|
|
392
|
+
type: "modelContentBlockDeltaEvent",
|
|
393
|
+
delta: {
|
|
394
|
+
type: "toolUseInputDelta",
|
|
395
|
+
input: JSON.stringify(input),
|
|
396
|
+
},
|
|
397
|
+
});
|
|
398
|
+
yield new ModelContentBlockStopEvent({
|
|
399
|
+
type: "modelContentBlockStopEvent",
|
|
400
|
+
});
|
|
401
|
+
}
|
|
402
|
+
// Native finish reason from mlex ("stop" | "length" | "toolCalls" |
|
|
403
|
+
// "aborted"), mapped onto Strands' stop reasons. "aborted" (the
|
|
404
|
+
// onToken callback stopped generation early — unused by this
|
|
405
|
+
// provider) degrades to endTurn.
|
|
406
|
+
const stopReason = finalResult.finishReason === "toolCalls"
|
|
407
|
+
? "toolUse"
|
|
408
|
+
: finalResult.finishReason === "length"
|
|
409
|
+
? "maxTokens"
|
|
410
|
+
: "endTurn";
|
|
411
|
+
yield new ModelMessageStopEvent({
|
|
412
|
+
type: "modelMessageStopEvent",
|
|
413
|
+
stopReason,
|
|
414
|
+
});
|
|
415
|
+
// mlex reports OpenAI-style usage: `promptTokens` is the full prompt
|
|
416
|
+
// and `cachedTokens` a subset of it served from the prompt-cache pool.
|
|
417
|
+
// The factory marks this model total-inclusive so billing meters
|
|
418
|
+
// normalize it to the additive shape.
|
|
419
|
+
const { promptTokens, cachedTokens, completionTokens } = finalResult.usage;
|
|
420
|
+
yield new ModelMetadataEvent({
|
|
421
|
+
type: "modelMetadataEvent",
|
|
422
|
+
usage: {
|
|
423
|
+
inputTokens: promptTokens,
|
|
424
|
+
outputTokens: completionTokens,
|
|
425
|
+
totalTokens: promptTokens + completionTokens,
|
|
426
|
+
...(cachedTokens > 0 ? { cacheReadInputTokens: cachedTokens } : {}),
|
|
427
|
+
},
|
|
428
|
+
});
|
|
429
|
+
}
|
|
430
|
+
catch (e) {
|
|
431
|
+
if (e instanceof ModelError) {
|
|
432
|
+
throw e;
|
|
433
|
+
}
|
|
434
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
435
|
+
throw new ModelError(`MLX generation error: ${msg}`, { cause: e });
|
|
436
|
+
}
|
|
437
|
+
}
|
|
438
|
+
}
|
|
439
|
+
//# sourceMappingURL=strands-mlx.js.map
|