pi-freeflow 1.6.0 → 1.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # pi-freeflow 🌊
2
2
 
3
- > **21 free models. Up to 1M context. Zero API keys. Infinite scale via your own relay pool.**
3
+ > **25 free models. Up to 1M context. Zero API keys. Infinite scale via your own relay pool.**
4
4
 
5
5
  Thin by design: model list + dumb relay + log. Host `pi-ai` owns thinking, normalization & provider magic. We just make it free, fast, and unbreakable.
6
6
 
@@ -18,7 +18,7 @@ Join devs bypassing rate limits with their own relay pools. BYO, add as many as
18
18
 
19
19
  | Feature | Description | Value | Cost |
20
20
  | :--- | :--- | :--- | :--- |
21
- | **21 Curated Free Models** | 7 OpenCode Zen + 14 KiloCode Gateway models, up to 1M context & 512K output | Ceiling Unlocked | **$0** |
21
+ | **25 Curated Free Models** | 7 OpenCode Zen + 18 KiloCode Gateway models, up to 1M context & 512K output | Ceiling Unlocked | **$0** |
22
22
  | **BYO Relay Pool** | Round-robin load balancing across your Cloudflare Workers & Vercel Edges | Zero Rate Limits | **$0** (your free tiers) |
23
23
  | **Adaptive Health & Error Detection** | Auto-cooldown on 429 rate limits, 504 timeouts, and socket drops | 0ms Wasted Latency | **$0** |
24
24
  | **Stream Truncation Resilience** | Stateful SSE terminal tracking (`response.failed` / `response.incomplete` injection) | Zero Host Crashes | **$0** |
@@ -32,7 +32,7 @@ Philosophy: **Thin by design.** We only ship model list + relay proxy + log. Hos
32
32
 
33
33
  ---
34
34
 
35
- ### 21 Curated Models, One Command
35
+ ### 25 Curated Models, One Command
36
36
 
37
37
  ```bash
38
38
  /model → freeflow → pick
@@ -43,35 +43,39 @@ Optimized for deep reasoning, long-horizon coding & autonomous agentic workflows
43
43
 
44
44
  | Model ID | Creator / Lab | Context | Max Output | Thinking | Vision |
45
45
  | :--- | :--- | :--- | :--- | :--- | :--- |
46
- | `muse-spark-1.2-contributor-free` | Meta Superintelligence Labs | **1M** (1.048.576) | **131K** (131.072) | `minimal / low / medium / high / xhigh / max` | ✅ |
47
- | `mimo-v2.5-free` | Xiaomi MiMo | **1M** (1.048.576) | **131K** (131.072) | `low / medium / high` | ✅ |
48
- | `laguna-s-2.1-free` | Poolside | **1M** (1.048.576) | **131K** (131.072) | `low / high / max` | ❌ |
49
- | `nemotron-3.5-lightning-free` | NVIDIA | **1M** (1.000.000) | **262K** (262.144) | `low / high / max` | ❌ |
50
- | `nemotron-3-ultra-free` | NVIDIA | **1M** (1.000.000) | **128K** (128.000) | `low / high / max` | ❌ |
51
- | `hy3-free` | Tencent Hunyuan | **262K** (262.144) | **262K** (262.144) | `low / high / max` | ❌ |
46
+ | `muse-spark-1.2-contributor-free` | Meta Superintelligence Labs | **1M** (1.048.576) | **131K** (131.072) | `minimal xhigh` | ✅ |
47
+ | `mimo-v2.5-free` | Xiaomi MiMo | **1M** (1.048.576) | **131K** (131.072) | `minimal xhigh`\* | ✅ |
48
+ | `laguna-s-2.1-free` | Poolside | **262K** (262.144) | **32K** (32.768) | `minimal xhigh` | ❌ |
49
+ | `nemotron-3.5-lightning-free` | NVIDIA | **1M** (1.000.000) | **262K** (262.144) | `minimal xhigh` | ❌ |
50
+ | `nemotron-3-ultra-free` | NVIDIA | **1M** (1.000.000) | **128K** (128.000) | `minimal xhigh` | ❌ |
51
+ | `hy3-free` | Tencent Hunyuan | **262K** (262.144) | **128K** (128.000) | `minimal xhigh` | ❌ |
52
52
  | `big-pickle` | Big Pickle | **200K** (200.000) | **32K** (32.000) | `high / max` | ❌ |
53
53
 
54
- #### KiloCode Gateway (14 Models), OpenRouter Compatible
54
+ #### KiloCode Gateway (18 Models), OpenRouter Compatible
55
55
  Keyless access with `Bearer kilo-free`. Clean slash-free and colon-free CLI aliases supported.
56
56
 
57
57
  | Model ID | Creator / Lab | Context | Max Output | Thinking | Vision |
58
58
  | :--- | :--- | :--- | :--- | :--- | :--- |
59
59
  | `dots-3-note-preview` (`dots-studio/...:free`) | Dots Studio | **512K** (512.000) | **512K** (512.000) | `minimal…xhigh`\* | ✅ |
60
60
  | `step-3.7-flash` (`stepfun/...:free`) | StepFun | **262K** (262.144) | **262K** (262.144) | `minimal…xhigh`\* | ✅ |
61
- | `nemotron-3-nano-omni` (`nvidia/...:free`) | NVIDIA | **256K** (256.000) | **65K** (65.536) | `minimal…xhigh`\* | ✅ |
62
- | `nemotron-3-ultra-550b` (`nvidia/...:free`) | NVIDIA | **1M** (1.000.000) | **65K** (65.536) | `minimal…xhigh`\* | ❌ |
63
- | `nvidia/nemotron-3.5-lightning:free` | NVIDIA | **1M** (1.000.000) | **131K** (131.072) | `minimal…xhigh`\* | ❌ |
61
+ | `nemotron-3-nano-omni` (`nvidia/...:free`) | NVIDIA | **256K** (256.000) | **131K** (131.072) | `minimal…xhigh`\* | ✅ |
62
+ | `nemotron-3-ultra-550b` (`nvidia/...:free`) | NVIDIA | **1M** (1.000.000) | **128K** (128.000) | `minimal…xhigh`\* | ❌ |
63
+ | `nvidia/nemotron-3.5-lightning:free` | NVIDIA | **1M** (1.000.000) | **262K** (262.144) | `minimal…xhigh`\* | ❌ |
64
64
  | `nemotron-3-super` (`nvidia/...:free`) | NVIDIA | **262K** (262.144) | **262K** (262.144) | `minimal…xhigh`\* | ❌ |
65
- | `hy3:free` (`tencent/hy3:free`) | Tencent Hunyuan | **262K** (262.144) | **262K** (262.144) | *(none sent)* | ❌ |
65
+ | `hy3:free` (`tencent/hy3:free`) | Tencent Hunyuan | **262K** (262.144) | **128K** (128.000) | `minimal…xhigh`\* | ❌ |
66
66
  | `north-mini-code` (`cohere/...:free`) | Cohere | **256K** (256.000) | **64K** (64.000) | `minimal…xhigh`\* | ❌ |
67
- | `laguna-s-2.1:free` (`poolside/...:free`) | Poolside | **1M** (1.048.576) | **131K** (131.072) | `minimal…xhigh`\* | ❌ |
67
+ | `laguna-s-2.1:free` (`poolside/...:free`) | Poolside | **262K** (262.144) | **32K** (32.768) | `minimal…xhigh`\* | ❌ |
68
68
  | `laguna-xs-2.1:free` (`poolside/...:free`) | Poolside | **262K** (262.144) | **32K** (32.768) | `minimal…xhigh`\* | ❌ |
69
- | `lfm-2.5` (`liquid/lfm-2.5-2.6b:free`) | Liquid AI | **128K** (128.000) | **32K** (32.768) | `minimal…xhigh`\* | ❌ |
70
- | `kilo-auto` (`kilo-auto/free`) | Kilo Gateway Auto | **256K** (256.000) | **10K** (10.000) | *(non-thinking)* | ❌ |
71
- | `openrouter` (`openrouter/free`) | OpenRouter Free | **200K** (200.000) | **65K** (65.536) | *(non-thinking)* | ✅ |
69
+ | `lfm-2.5` (`liquid/lfm-2.5-2.6b:free`) | Liquid AI | **65K** (65.536) | **32K** (32.768) | `minimal…xhigh`\* | ❌ |
70
+ | `kilo-auto` (`kilo-auto/free`) | Kilo Gateway Auto | **256K** (256.000) | **10K** (10.000) | `minimal…xhigh`\* | ❌ |
71
+ | `openrouter` (`openrouter/free`) | OpenRouter Free | **200K** (200.000) | **65K** (65.536) | `minimal…xhigh`\* | ✅ |
72
72
  | `content-safety` (`nvidia/...:free`) | NVIDIA | **128K** (128.000) | **8K** (8.192) | ❌ *(non-thinking)* | ✅ |
73
+ | `longcat-2.0` (`meituan/longcat-2.0-free`) | Meituan | **1M** (1.048.756) | **262K** (262.144) | `minimal…xhigh`\* | ❌ |
74
+ | `minimax-m2.7` (`minimax/minimax-m2.7:free`) | MiniMax | **196K** (196.608) | **196K** (196.608) | `minimal…xhigh`\* | ❌ |
75
+ | `minimax-m3` (`minimax/minimax-m3:free`) | MiniMax | **1M** (1.048.576) | **512K** (524.288) | `minimal…xhigh`\* | ❌ |
76
+ | `ling-3.0-flash-fin` (`inclusionai/ling-3.0-flash-fin:free`) | Inclusion AI | **262K** (262.144) | **32K** (32.768) | `minimal…xhigh`\* | ❌ |
73
77
 
74
- \* Levels are forwarded as-is through the OpenRouter-style nested `reasoning` parameter; effort mapping is decided by each model. hy3 accepts no reasoning parameter today.
78
+ \* Levels are forwarded as-is through the OpenRouter-style nested `reasoning` parameter; effort mapping is decided by each model. (Verified live 2026-08-29: hy3 accepts flat `reasoning_effort`/nested `reasoning` and returns thinking — README previously said otherwise.) MiMo collapses `minimal→low` and `xhigh→high` upstream, so its selector shows 5 labels but only 3 distinct effort values.
75
79
 
76
80
  ---
77
81
 
@@ -119,33 +123,64 @@ Manage your relay pool directly from the OMP / Pi terminal:
119
123
 
120
124
  ### Quick Start in 30 Seconds
121
125
 
122
- #### 1. Install
123
-
124
- **Oh My Pi (Recommended):**
126
+ **On Oh My Pi (OMP):**
125
127
  ```bash
126
128
  omp plugin install pi-freeflow
127
- # or local dev
129
+ # or local dev (repo checkout)
128
130
  omp plugin link /path/to/pi-freeflow
129
131
  ```
130
132
 
131
- **Pi:**
133
+ **On Pi:**
132
134
  ```bash
133
135
  pi install npm:pi-freeflow
134
136
  ```
135
137
 
138
+ > Both commands fetch the **same npm package** from the registry — pi-freeflow
139
+ > is an extension loaded by the host, not a standalone CLI. It works on **OMP
140
+ > and Pi only** (they share the extension API).
141
+
136
142
  #### 2. Pick a Model
137
143
 
144
+ **OMP — interactive:**
138
145
  ```bash
139
146
  omp
140
147
  /model → freeflow → muse-spark-1.2-contributor-free (1M) → max
148
+ ```
141
149
 
142
- # or CLI
150
+ **OMP one-shot CLI:**
151
+ ```bash
143
152
  omp -p --model freeflow/muse-spark-1.2-contributor-free "build me a SaaS"
144
- # or with short alias & thinking level
145
- omp -p --model freeflow/step-3.7-flash:high "solve this bug"
153
+ omp -p --model freeflow/step-3.7-flash:high "solve this bug" # alias + thinking level
154
+ ```
155
+
156
+ **Pi — interactive:**
157
+ ```bash
158
+ pi
159
+ /model → freeflow → pick
160
+ ```
161
+
162
+ **Pi — one-shot CLI:**
163
+ ```bash
164
+ pi -p --model freeflow/step-3.7-flash:high "solve this bug"
165
+ ```
166
+
167
+ > Model IDs accept a full canonical ID, a short alias (see the tables above),
168
+ > and an optional `:effort` suffix (`:minimal` … `:xhigh`, `:max` where
169
+ > supported). The host resolves the rest — you only type `freeflow/<name>`.
170
+
171
+ #### 3. Manage Your Relay Pool (OMP & Pi both)
172
+
173
+ ```bash
174
+ /freeflow status # active relay, pool status, candidates
175
+ /freeflow list # relays with health badges
176
+ /freeflow add <url> [label]
177
+ /freeflow deploy # guided deploy: vercel|cloudflare|deno
178
+ /freeflow logs [n] # tail proxy logs
146
179
  ```
147
180
 
148
- #### 3. Add Your Free Relays (Scale Infinitely)
181
+ These slash commands work **identically in OMP and Pi** — the extension
182
+ registers the same `/freeflow` command set in both hosts.
183
+ #### 4. Add Your Free Relays (Scale Infinitely)
149
184
 
150
185
  Default ships direct. Add relays via `/freeflow add <url> [label]`.
151
186
 
@@ -227,7 +262,7 @@ Log rotation at 10MB. Clean, parseable, real-time HTTP lifecycle tracking.
227
262
 
228
263
  This package stays thin. It ships three things: a model catalog, a relay proxy, and a log. There is no build step. The only runtime dependency is `undici`, which powers the upstream fetch agent. Thinking and prompt normalization stay with the host (`pi-ai`).
229
264
 
230
- Current size: about 11.3k lines including tests. 226 tests pass, typecheck clean.
265
+ Current size: about 11.3k lines including tests. 230 tests pass, typecheck clean.
231
266
 
232
267
  ---
233
268
 
@@ -247,6 +282,18 @@ Deleted in 1.3.0. If zai/qwen/deepseek thinking broke before, it's fixed now bec
247
282
 
248
283
  **Why is context free?**
249
284
  We use OpenCode Zen & Kilo free tiers. You pay only with your own Cloudflare/Vercel free tiers for egress.
285
+ **Why is it installed via npm?**
286
+ The npm package is the **distribution channel** only — both hosts resolve it internally:
287
+ `omp plugin install pi-freeflow` and `pi install npm:pi-freeflow` install the same
288
+ package from the npm registry. pi-freeflow is an **extension, not a standalone CLI** —
289
+ the host (OMP or Pi) loads and runs it. A plain `npm install` just downloads the
290
+ files; it is not a supported way to run the extension.
291
+
292
+ **Which hosts can use it?**
293
+ Oh My Pi (OMP) and Pi only. They share the same extension API
294
+ (`extensions/index.ts` declares both `omp` and `pi` extension entries), so one
295
+ package serves both. Other AI agents (OpenCode, KiloCode, Cursor, ...) have their
296
+ own plugin systems and do not load this extension.
250
297
 
251
298
  ---
252
299
 
@@ -267,7 +314,7 @@ cd pi-freeflow
267
314
  pnpm install
268
315
 
269
316
  # run all three before opening a PR
270
- pnpm test # 226 tests across 30 test files
317
+ pnpm test # 228 tests across 30 test files
271
318
  pnpm typecheck # tsc --noEmit, must pass clean
272
319
  pnpm smoke # verifies extensions/index.ts loads without crashing
273
320
  ```
@@ -277,7 +324,7 @@ pnpm smoke # verifies extensions/index.ts loads without crashing
277
324
  ```
278
325
  src/
279
326
  ├── index.ts # extension entry, lifecycle hooks
280
- ├── models.ts # 21-model catalog definitions
327
+ ├── models.ts # 25-model catalog definitions
281
328
  ├── catalog.ts # model catalog cache (24h disk)
282
329
  ├── proxy.ts # local proxy server (127.0.0.1:28180)
283
330
  ├── relay.ts # relay selection & round-robin
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "pi-freeflow",
3
3
  "type": "module",
4
- "version": "1.6.0",
4
+ "version": "1.7.1",
5
5
  "description": "Thin provider for OMP/Pi — model list + dumb relay proxy + log; host pi-ai owns thinking/normalization",
6
6
  "main": "extensions/index.ts",
7
7
  "types": "src/index.ts",
package/src/catalog.ts CHANGED
@@ -34,7 +34,7 @@ import type {
34
34
  export const DEAD_MODEL_IDS = new Set<string>(["deepseek-v4-flash-free", "x-preview-f-free"]);
35
35
  /**
36
36
  * In-memory cache of currently active/available free models.
37
- * Initialized with all 21 verified models for 0ms instant availability.
37
+ * Initialized with all 25 verified models for 0ms instant availability.
38
38
  */
39
39
  let aliveCatalog: RegisteredModel[] = ALL_MODELS.map((m) => ({
40
40
  ...m,
@@ -328,6 +328,6 @@ export async function refreshCatalog(force = false): Promise<RegisteredModel[]>
328
328
  } catch (err) {
329
329
  logDebug("Failed reading stale catalog cache", { error: String(err) });
330
330
  }
331
- // No valid cache — return in-memory static 21 (host will refresh if needed)
331
+ // No valid cache — return in-memory static 25 (host will refresh if needed)
332
332
  return aliveCatalog;
333
333
  }
package/src/index.ts CHANGED
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * pi-freeflow — Modular, high-resiliency LLM extension for Pi & Oh My Pi (OMP)
3
3
  *
4
- * Provides access to 21 free models (7 OpenCode Zen + 14 KiloCode Gateway) with:
4
+ * Provides access to 25 free models (7 OpenCode Zen + 18 KiloCode Gateway) with:
5
5
  * - Single-port daemon reuse on 28180 across concurrent subagents
6
6
  * - Multi-cloud rolling egress relays (Vercel Edge, Cloudflare, Deno)
7
7
  * - 0ms instant startup with verified static catalog and background live health checks
@@ -310,7 +310,7 @@ export default async function (pi: ExtensionAPI): Promise<void> {
310
310
  ? `Relay pool ready: ${freshRelayState.relays.length} relay(s). Run /freeflow for pool management.`
311
311
  : "Relay pool empty — direct mode. Run /freeflow deploy to add your own egress.";
312
312
  ctx.ui?.notify?.(
313
- `freeflow ready: 21 free models via local proxy 127.0.0.1:28180. ${hint}`,
313
+ `freeflow ready: 25 free models via local proxy 127.0.0.1:28180. ${hint}`,
314
314
  "info",
315
315
  );
316
316
  }
package/src/models.ts CHANGED
@@ -1,12 +1,12 @@
1
1
  /**
2
2
  * Static model definitions and upstream routing catalogs for pi-freeflow
3
3
  *
4
- * Defines the 21 verified free models:
4
+ * Defines the 25 verified free models:
5
5
  * - 7 OpenCode Zen models (1 Responses API + 6 Chat Completions)
6
- * - 14 KiloCode Keyless Gateway models (10 OpenRouter format + 4 Standard format)
6
+ * - 18 KiloCode Keyless Gateway models (17 OpenRouter format + 1 Standard format)
7
7
  */
8
8
 
9
- import type { ModelDef, Upstream } from "./types.ts";
9
+ import type { ModelDef, ThinkingLevelMap, Upstream } from "./types.ts";
10
10
 
11
11
  /**
12
12
  * OpenCode Zen free models verified against the live catalog and inference APIs.
@@ -51,7 +51,7 @@ export const OPENCODE_MODELS: ModelDef[] = [
51
51
  name: "Hy3 (262K)",
52
52
  reasoning: true,
53
53
  contextWindow: 262_144,
54
- maxTokens: 262_144,
54
+ maxTokens: 128_000,
55
55
  input: ["text"],
56
56
  thinkingLevelMap: {
57
57
  off: null,
@@ -112,10 +112,10 @@ export const OPENCODE_MODELS: ModelDef[] = [
112
112
  },
113
113
  {
114
114
  id: "laguna-s-2.1-free",
115
- name: "Laguna S 2.1 (1M)",
115
+ name: "Laguna S 2.1 (256K)",
116
116
  reasoning: true,
117
- contextWindow: 1_048_576,
118
- maxTokens: 131_072,
117
+ contextWindow: 262_144,
118
+ maxTokens: 32_768,
119
119
  input: ["text"],
120
120
  thinkingLevelMap: {
121
121
  off: null,
@@ -128,6 +128,23 @@ export const OPENCODE_MODELS: ModelDef[] = [
128
128
  },
129
129
  ];
130
130
 
131
+ /**
132
+ * Shared effort map for Kilo reasoning models — verified live 2026-08-29:
133
+ * gateway accepts flat reasoning_effort minimal..xhigh for every reasoning
134
+ * model; stepfun/step-3.7-flash measured monotonic 77→313 thinking chars
135
+ * across minimal→xhigh. Declaring the map locks the picker (instead of
136
+ * host guessing) and matches the OpenCode-model pattern.
137
+ */
138
+ const KILO_REASONING_MAP: ThinkingLevelMap = {
139
+ off: null,
140
+ minimal: "minimal",
141
+ low: "low",
142
+ medium: "medium",
143
+ high: "high",
144
+ xhigh: "xhigh",
145
+ max: null,
146
+ };
147
+
131
148
  /**
132
149
  * KiloCode Gateway free models (keyless — https://kilo.ai/docs/gateway).
133
150
  * Endpoint: https://api.kilo.ai/api/gateway/chat/completions
@@ -141,6 +158,7 @@ export const KILO_MODELS: ModelDef[] = [
141
158
  maxTokens: 512_000,
142
159
  input: ["text", "image"],
143
160
  thinkingFormat: "openrouter",
161
+ thinkingLevelMap: KILO_REASONING_MAP,
144
162
  },
145
163
  {
146
164
  id: "stepfun/step-3.7-flash:free",
@@ -150,24 +168,27 @@ export const KILO_MODELS: ModelDef[] = [
150
168
  maxTokens: 262_144,
151
169
  input: ["text", "image"],
152
170
  thinkingFormat: "openrouter",
171
+ thinkingLevelMap: KILO_REASONING_MAP,
153
172
  },
154
173
  {
155
174
  id: "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free",
156
175
  name: "Nemotron 3 Nano Omni",
157
176
  reasoning: true,
158
177
  contextWindow: 256_000,
159
- maxTokens: 65_536,
178
+ maxTokens: 131_072,
160
179
  input: ["text", "image"],
161
180
  thinkingFormat: "openrouter",
181
+ thinkingLevelMap: KILO_REASONING_MAP,
162
182
  },
163
183
  {
164
184
  id: "nvidia/nemotron-3-ultra-550b-a55b:free",
165
185
  name: "Nemotron 3 Ultra 550B (1M)",
166
186
  reasoning: true,
167
187
  contextWindow: 1_000_000,
168
- maxTokens: 65_536,
188
+ maxTokens: 128_000,
169
189
  input: ["text"],
170
190
  thinkingFormat: "openrouter",
191
+ thinkingLevelMap: KILO_REASONING_MAP,
171
192
  },
172
193
  {
173
194
  id: "nvidia/nemotron-3.5-lightning:free",
@@ -177,6 +198,7 @@ export const KILO_MODELS: ModelDef[] = [
177
198
  maxTokens: 262_144,
178
199
  input: ["text"],
179
200
  thinkingFormat: "openrouter",
201
+ thinkingLevelMap: KILO_REASONING_MAP,
180
202
  },
181
203
  {
182
204
  id: "nvidia/nemotron-3-super-120b-a12b:free",
@@ -186,15 +208,17 @@ export const KILO_MODELS: ModelDef[] = [
186
208
  maxTokens: 262_144,
187
209
  input: ["text"],
188
210
  thinkingFormat: "openrouter",
211
+ thinkingLevelMap: KILO_REASONING_MAP,
189
212
  },
190
213
  {
191
214
  id: "tencent/hy3:free",
192
215
  name: "Tencent Hy3 (Kilo)",
193
216
  reasoning: true,
194
217
  contextWindow: 262_144,
195
- maxTokens: 262_144,
218
+ maxTokens: 128_000,
196
219
  input: ["text"],
197
220
  thinkingFormat: "openrouter",
221
+ thinkingLevelMap: KILO_REASONING_MAP,
198
222
  },
199
223
  {
200
224
  id: "cohere/north-mini-code:free",
@@ -204,15 +228,17 @@ export const KILO_MODELS: ModelDef[] = [
204
228
  maxTokens: 64_000,
205
229
  input: ["text"],
206
230
  thinkingFormat: "openrouter",
231
+ thinkingLevelMap: KILO_REASONING_MAP,
207
232
  },
208
233
  {
209
234
  id: "poolside/laguna-s-2.1:free",
210
235
  name: "Laguna S 2.1 (Kilo)",
211
236
  reasoning: true,
212
- contextWindow: 1_048_576,
213
- maxTokens: 131_072,
237
+ contextWindow: 262_144,
238
+ maxTokens: 32_768,
214
239
  input: ["text"],
215
240
  thinkingFormat: "openrouter",
241
+ thinkingLevelMap: KILO_REASONING_MAP,
216
242
  },
217
243
  {
218
244
  id: "poolside/laguna-xs-2.1:free",
@@ -222,31 +248,37 @@ export const KILO_MODELS: ModelDef[] = [
222
248
  maxTokens: 32_768,
223
249
  input: ["text"],
224
250
  thinkingFormat: "openrouter",
251
+ thinkingLevelMap: KILO_REASONING_MAP,
225
252
  },
226
253
  {
227
254
  id: "liquid/lfm-2.5-2.6b:free",
228
255
  name: "Liquid LFM 2.5",
229
256
  reasoning: true,
230
- contextWindow: 128_000,
257
+ contextWindow: 65_536,
231
258
  maxTokens: 32_768,
232
259
  input: ["text"],
233
260
  thinkingFormat: "openrouter",
261
+ thinkingLevelMap: KILO_REASONING_MAP,
234
262
  },
235
263
  {
236
264
  id: "kilo-auto/free",
237
265
  name: "Kilo Auto",
238
- reasoning: false,
266
+ reasoning: true,
239
267
  contextWindow: 256_000,
240
268
  maxTokens: 10_000,
241
269
  input: ["text"],
270
+ thinkingFormat: "openrouter",
271
+ thinkingLevelMap: KILO_REASONING_MAP,
242
272
  },
243
273
  {
244
274
  id: "openrouter/free",
245
275
  name: "OpenRouter Auto",
246
- reasoning: false,
276
+ reasoning: true,
247
277
  contextWindow: 200_000,
248
278
  maxTokens: 65_536,
249
279
  input: ["text", "image"],
280
+ thinkingFormat: "openrouter",
281
+ thinkingLevelMap: KILO_REASONING_MAP,
250
282
  },
251
283
  {
252
284
  id: "nvidia/nemotron-3.5-content-safety:free",
@@ -256,6 +288,46 @@ export const KILO_MODELS: ModelDef[] = [
256
288
  maxTokens: 8_192,
257
289
  input: ["text", "image"],
258
290
  },
291
+ {
292
+ id: "meituan/longcat-2.0-free",
293
+ name: "LongCat 2.0 (1M)",
294
+ reasoning: true,
295
+ contextWindow: 1_048_756,
296
+ maxTokens: 262_144,
297
+ input: ["text"],
298
+ thinkingFormat: "openrouter",
299
+ thinkingLevelMap: KILO_REASONING_MAP,
300
+ },
301
+ {
302
+ id: "minimax/minimax-m2.7:free",
303
+ name: "MiniMax M2.7 (free)",
304
+ reasoning: true,
305
+ contextWindow: 196_608,
306
+ maxTokens: 196_608,
307
+ input: ["text"],
308
+ thinkingFormat: "openrouter",
309
+ thinkingLevelMap: KILO_REASONING_MAP,
310
+ },
311
+ {
312
+ id: "minimax/minimax-m3:free",
313
+ name: "MiniMax M3 (free)",
314
+ reasoning: true,
315
+ contextWindow: 1_048_576,
316
+ maxTokens: 524_288,
317
+ input: ["text"],
318
+ thinkingFormat: "openrouter",
319
+ thinkingLevelMap: KILO_REASONING_MAP,
320
+ },
321
+ {
322
+ id: "inclusionai/ling-3.0-flash-fin:free",
323
+ name: "Ling 3.0 Flash Fin",
324
+ reasoning: true,
325
+ contextWindow: 262_144,
326
+ maxTokens: 32_768,
327
+ input: ["text"],
328
+ thinkingFormat: "openrouter",
329
+ thinkingLevelMap: KILO_REASONING_MAP,
330
+ },
259
331
  ];
260
332
 
261
333
  /**
@@ -271,10 +343,15 @@ export const MODEL_ALIASES: Record<string, string> = {
271
343
  "step-3.7-flash": "stepfun/step-3.7-flash:free",
272
344
  "nemotron-3-nano-omni": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free",
273
345
  "nemotron-3-ultra-550b": "nvidia/nemotron-3-ultra-550b-a55b:free",
346
+ "nemotron-3.5-lightning": "nvidia/nemotron-3.5-lightning:free",
274
347
  "nemotron-3-super": "nvidia/nemotron-3-super-120b-a12b:free",
275
348
  "north-mini-code": "cohere/north-mini-code:free",
276
349
  "lfm-2.5": "liquid/lfm-2.5-2.6b:free",
277
350
  "content-safety": "nvidia/nemotron-3.5-content-safety:free",
351
+ "longcat-2.0": "meituan/longcat-2.0-free",
352
+ "minimax-m2.7": "minimax/minimax-m2.7:free",
353
+ "minimax-m3": "minimax/minimax-m3:free",
354
+ "ling-3.0-flash-fin": "inclusionai/ling-3.0-flash-fin:free",
278
355
  // provider-prefixed short aliases (slash-normalized)
279
356
  "hy3:free": "tencent/hy3:free",
280
357
  "laguna-s-2.1:free": "poolside/laguna-s-2.1:free",
@@ -302,7 +379,7 @@ export const KILO_MODEL_IDS = new Set<string>([
302
379
  ]);
303
380
 
304
381
  /**
305
- * Combined list of all 21 static free models (canonical)
382
+ * Combined list of all 25 static free models (canonical)
306
383
  */
307
384
  export const ALL_MODELS: ModelDef[] = [...OPENCODE_MODELS, ...KILO_MODELS];
308
385