@mindstudio-ai/remy 0.1.311 → 0.1.313

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -122,8 +122,9 @@ declare class HeadlessSession {
122
122
  /** Apply queued tool block updates to state.messages. Safe to call any time. */
123
123
  private applyPendingBlockUpdates;
124
124
  /**
125
- * Forced compaction gate. If lastContextSize exceeds the threshold, compact
126
- * before letting the upcoming turn run. Coalesces with any in-flight
125
+ * Forced compaction gate. If lastContextSize exceeds the upcoming turn's
126
+ * model threshold (per-model — see ModelContextLimits in models/surfaces),
127
+ * compact before letting the turn run. Coalesces with any in-flight
127
128
  * compaction (e.g., one already started by /compact or a tool call). No
128
129
  * timeout — compaction takes as long as it takes.
129
130
  *
package/dist/headless.js CHANGED
@@ -486,36 +486,59 @@ var MODEL_SURFACES = {
486
486
  userPickable: false
487
487
  }
488
488
  };
489
+ var DEFAULT_SUGGEST_COMPACT_AT = 3e5;
490
+ var TEXT_MODELS = {
491
+ // Anthropic 1M-context, flat-priced.
492
+ "claude-5-opus": { forceCompactAt: 85e4 },
493
+ "claude-4-8-opus": { forceCompactAt: 85e4 },
494
+ "claude-4-7-opus": { forceCompactAt: 85e4 },
495
+ // Anthropic 1M-context with 2x long-context pricing above 200K input.
496
+ "claude-4-6-opus": { forceCompactAt: 85e4, suggestCompactAt: 18e4 },
497
+ "claude-4-6-sonnet": { forceCompactAt: 85e4, suggestCompactAt: 18e4 },
498
+ "claude-fable-5": { forceCompactAt: 85e4 },
499
+ "claude-fable-5-1": { forceCompactAt: 85e4 },
500
+ "claude-5-sonnet": { forceCompactAt: 85e4 },
501
+ // OpenAI gpt-5.5/5.6: ~1M window, but the usable input ceiling under
502
+ // `truncation: 'auto'` is ~794K (output + reasoning reserve), and all
503
+ // rates double above 272K input.
504
+ "gpt-5.5": { forceCompactAt: 6e5, suggestCompactAt: 25e4 },
505
+ "gpt-5.6-sol": { forceCompactAt: 6e5, suggestCompactAt: 25e4 },
506
+ "gpt-5.6-terra": { forceCompactAt: 6e5, suggestCompactAt: 25e4 },
507
+ "gpt-5.6-luna": { forceCompactAt: 6e5, suggestCompactAt: 25e4 },
508
+ // Google ~1M-context; only 3.1-pro is tiered (higher rates above 200K).
509
+ "gemini-3-pro": { forceCompactAt: 85e4 },
510
+ "gemini-3.1-pro": { forceCompactAt: 85e4, suggestCompactAt: 18e4 },
511
+ "gemini-3-flash": { forceCompactAt: 85e4 },
512
+ "gemini-3.5-flash": { forceCompactAt: 85e4 },
513
+ "gemini-3.7-flash": { forceCompactAt: 85e4 },
514
+ // 256K window; its 200K pricing tier sits above the gate, so no nudge.
515
+ "grok-build-0.1": { forceCompactAt: 18e4 },
516
+ "grok-4.5": { forceCompactAt: 4e5 },
517
+ // 500K window
518
+ "grok-4.6": { forceCompactAt: 4e5 },
519
+ // 500K window
520
+ "glm-5.2": { forceCompactAt: 85e4 },
521
+ "muse-spark-1.1": { forceCompactAt: 85e4 },
522
+ "kimi-k2-7-code": { forceCompactAt: 2e5 },
523
+ // 262K window
524
+ "kimi-k3": { forceCompactAt: 85e4 },
525
+ "deepseek-v4-flash-0731": { forceCompactAt: 85e4 },
526
+ "qwen3.8-2.4t-a95b-deepinfra": { forceCompactAt: 2e5 },
527
+ // 262K window
528
+ "qwen3.8-27b-deepinfra": { forceCompactAt: 2e5 },
529
+ // 262K window
530
+ "minimax-m3": { forceCompactAt: 42e4 }
531
+ // 524K window
532
+ };
533
+ var DEFAULT_CONTEXT_LIMITS = { forceCompactAt: 85e4 };
534
+ function getContextLimits(modelId) {
535
+ return TEXT_MODELS[modelId] ?? DEFAULT_CONTEXT_LIMITS;
536
+ }
537
+ function getSuggestCompactAt(modelId) {
538
+ return getContextLimits(modelId).suggestCompactAt ?? DEFAULT_SUGGEST_COMPACT_AT;
539
+ }
489
540
  var ALLOWED_MODELS_BY_TYPE = {
490
- text: [
491
- "claude-5-opus",
492
- "claude-4-8-opus",
493
- "claude-4-7-opus",
494
- "claude-4-6-opus",
495
- "claude-4-6-sonnet",
496
- "claude-fable-5",
497
- "claude-5-sonnet",
498
- "gpt-5.5",
499
- "gpt-5.6-sol",
500
- "gpt-5.6-terra",
501
- "gpt-5.6-luna",
502
- "gemini-3-pro",
503
- "gemini-3.1-pro",
504
- "gemini-3-flash",
505
- "gemini-3.5-flash",
506
- "gemini-3.7-flash",
507
- "grok-build-0.1",
508
- "grok-4.5",
509
- "grok-4.6",
510
- "glm-5.2",
511
- "muse-spark-1.1",
512
- "kimi-k2-7-code",
513
- "kimi-k3",
514
- "deepseek-v4-flash-0731",
515
- "qwen3.8-2.4t-a95b-deepinfra",
516
- "qwen3.8-27b-deepinfra",
517
- "minimax-m3"
518
- ]
541
+ text: Object.keys(TEXT_MODELS)
519
542
  // vision: undefined — unconstrained
520
543
  // image_generation: undefined — unconstrained
521
544
  };
@@ -558,6 +581,11 @@ function getEffectiveModelSurfaces() {
558
581
  function resolveModel(surfaceId, models, fallback) {
559
582
  return models?.[surfaceId] ?? fallback ?? orgDefaultModels[surfaceId] ?? MODEL_SURFACES[surfaceId].default;
560
583
  }
584
+ function resolveParentModel(models, fallback, buildModel) {
585
+ const override = buildModel ? filterModelPicks({ parent: buildModel }).parent : void 0;
586
+ const baseline = resolveModel("parent", models, fallback);
587
+ return { baseline, effective: override ?? baseline };
588
+ }
561
589
 
562
590
  // src/orgContext.ts
563
591
  var log3 = createLogger("orgContext");
@@ -771,7 +799,7 @@ ${entries.join("\n\n")}
771
799
  // src/prompt/skills/_catalog.ts
772
800
  var INTRO = `Platform capabilities most apps don't use, so their references are kept out of this prompt rather than competing for your attention on every task \u2014 not because they're marginal.
773
801
 
774
- Read what follows as part of what the platform can do, not as a lookup table. Recognising that one of these fits a feature is your job, and proposing one is fair game \u2014 several of them are the difference between an app that works and an app worth showing off. When a trigger fires, load the reference with loadSkill before writing the code rather than after. Loading is cheap and expected; guessing at one of these APIs is not.
802
+ Read what follows as part of what the platform can do, not as a lookup table. Recognising that one of these fits a feature is your job, and proposing one is fair game \u2014 several of them are the difference between an app that works and an app worth showing off. When a trigger fires, load the reference with loadSkill before writing the spec or code rather than after. Loading is cheap and expected; guessing at one of these APIs is not.
775
803
 
776
804
  A loaded reference drops out of the conversation once it ages out. Re-read it at the path listed with readFile whenever you need it again.`;
777
805
  var catalog = buildSkillCatalog({
@@ -8615,6 +8643,12 @@ function classify(line, matches) {
8615
8643
  tailClean = SEPARATORS_ONLY.test(gap);
8616
8644
  }
8617
8645
  }
8646
+ function chipCase(label) {
8647
+ if (label.startsWith("`")) {
8648
+ return label;
8649
+ }
8650
+ return label.charAt(0).toUpperCase() + label.slice(1);
8651
+ }
8618
8652
  function parseSuggestions(raw) {
8619
8653
  if (!raw.includes(SUGGEST_MARKER)) {
8620
8654
  return { text: raw, suggestions: [] };
@@ -8644,7 +8678,7 @@ function parseSuggestions(raw) {
8644
8678
  const key = `${m.label}::${m.message}`;
8645
8679
  if (!seen.has(key)) {
8646
8680
  seen.add(key);
8647
- suggestions.push({ label: m.label, message: m.message });
8681
+ suggestions.push({ label: chipCase(m.label), message: m.message });
8648
8682
  }
8649
8683
  }
8650
8684
  let rebuilt = "";
@@ -8744,10 +8778,12 @@ async function runTurn(params) {
8744
8778
  onBackgroundComplete
8745
8779
  } = params;
8746
8780
  const tools2 = getToolDefinitions();
8747
- const buildModelOverride = buildModel ? filterModelPicks({ parent: buildModel }).parent : void 0;
8748
- const baseline = resolveModel("parent", state.models, model);
8749
- const parentModel = buildModelOverride ?? baseline;
8750
- const modelOverride = buildModelOverride && buildModelOverride !== baseline ? { from: baseline } : void 0;
8781
+ const { baseline, effective: parentModel } = resolveParentModel(
8782
+ state.models,
8783
+ model,
8784
+ buildModel
8785
+ );
8786
+ const modelOverride = parentModel !== baseline ? { from: baseline } : void 0;
8751
8787
  const totalAttachments = entries.reduce(
8752
8788
  (n, e) => n + (e.attachments?.length ?? 0),
8753
8789
  0
@@ -8755,7 +8791,7 @@ async function runTurn(params) {
8755
8791
  log15.info("Turn started", {
8756
8792
  requestId,
8757
8793
  model,
8758
- buildModel: buildModelOverride,
8794
+ buildModel: modelOverride ? parentModel : void 0,
8759
8795
  toolCount: tools2.length,
8760
8796
  ...entries.length > 1 && { entryCount: entries.length },
8761
8797
  ...totalAttachments > 0 && { attachmentCount: totalAttachments }
@@ -9590,12 +9626,13 @@ function loadPassiveResults() {
9590
9626
  }
9591
9627
  return [];
9592
9628
  }
9593
- function writeStats(stats, queue, passiveResults) {
9629
+ function writeStats(stats, queue, passiveResults, suggestCompactAt) {
9594
9630
  try {
9595
9631
  writeFileAtomicSync(
9596
9632
  STATS_FILE,
9597
9633
  JSON.stringify({
9598
9634
  ...stats,
9635
+ suggestCompactAt,
9599
9636
  queue,
9600
9637
  passiveResults
9601
9638
  })
@@ -9797,7 +9834,6 @@ var USER_FACING_TOOLS = /* @__PURE__ */ new Set([
9797
9834
  "confirmDestructiveAction",
9798
9835
  "presentPublishPlan"
9799
9836
  ]);
9800
- var FORCED_COMPACTION_THRESHOLD_TOKENS = 85e4;
9801
9837
  var HeadlessSession = class {
9802
9838
  // Configuration
9803
9839
  opts;
@@ -10070,7 +10106,15 @@ var HeadlessSession = class {
10070
10106
  /** Persist sessionStats + queue snapshot + passive pen to .remy-stats.json. */
10071
10107
  persistStats() {
10072
10108
  this.sessionStats.updatedAt = Date.now();
10073
- writeStats(this.sessionStats, this.queue.snapshot(), this.passivePen);
10109
+ const suggestCompactAt = getSuggestCompactAt(
10110
+ resolveParentModel(this.state.models, this.opts.model).effective
10111
+ );
10112
+ writeStats(
10113
+ this.sessionStats,
10114
+ this.queue.snapshot(),
10115
+ this.passivePen,
10116
+ suggestCompactAt
10117
+ );
10074
10118
  }
10075
10119
  //////////////////////////////////////////////////////////////////////////////
10076
10120
  // Background completions (tool-block mutation; message delivery via queue)
@@ -10100,8 +10144,9 @@ var HeadlessSession = class {
10100
10144
  saveSession(this.state);
10101
10145
  }
10102
10146
  /**
10103
- * Forced compaction gate. If lastContextSize exceeds the threshold, compact
10104
- * before letting the upcoming turn run. Coalesces with any in-flight
10147
+ * Forced compaction gate. If lastContextSize exceeds the upcoming turn's
10148
+ * model threshold (per-model — see ModelContextLimits in models/surfaces),
10149
+ * compact before letting the turn run. Coalesces with any in-flight
10105
10150
  * compaction (e.g., one already started by /compact or a tool call). No
10106
10151
  * timeout — compaction takes as long as it takes.
10107
10152
  *
@@ -10112,13 +10157,15 @@ var HeadlessSession = class {
10112
10157
  * On compaction failure we don't bail — the turn proceeds and surfaces any
10113
10158
  * downstream overflow through the existing "prompt is too long" path.
10114
10159
  */
10115
- async runForcedCompactionIfNeeded(requestId) {
10116
- if (this.sessionStats.lastContextSize <= FORCED_COMPACTION_THRESHOLD_TOKENS) {
10160
+ async runForcedCompactionIfNeeded(requestId, parentModel) {
10161
+ const threshold = getContextLimits(parentModel).forceCompactAt;
10162
+ if (this.sessionStats.lastContextSize <= threshold) {
10117
10163
  return;
10118
10164
  }
10119
10165
  log17.info("Forced compaction gate triggered", {
10120
10166
  contextSize: this.sessionStats.lastContextSize,
10121
- threshold: FORCED_COMPACTION_THRESHOLD_TOKENS,
10167
+ threshold,
10168
+ model: parentModel,
10122
10169
  requestId
10123
10170
  });
10124
10171
  try {
@@ -10600,7 +10647,12 @@ var HeadlessSession = class {
10600
10647
  }
10601
10648
  return steered;
10602
10649
  };
10603
- await this.runForcedCompactionIfNeeded(requestId);
10650
+ const parentModel = resolveParentModel(
10651
+ this.state.models,
10652
+ this.opts.model,
10653
+ params.buildModel
10654
+ ).effective;
10655
+ await this.runForcedCompactionIfNeeded(requestId, parentModel);
10604
10656
  try {
10605
10657
  await runTurn({
10606
10658
  state: this.state,
package/dist/index.js CHANGED
@@ -2111,6 +2111,12 @@ var init_compaction = __esm({
2111
2111
  });
2112
2112
 
2113
2113
  // src/models/surfaces.ts
2114
+ function getContextLimits(modelId) {
2115
+ return TEXT_MODELS[modelId] ?? DEFAULT_CONTEXT_LIMITS;
2116
+ }
2117
+ function getSuggestCompactAt(modelId) {
2118
+ return getContextLimits(modelId).suggestCompactAt ?? DEFAULT_SUGGEST_COMPACT_AT;
2119
+ }
2114
2120
  function setOrgDefaultModels(models) {
2115
2121
  orgDefaultModels = models;
2116
2122
  }
@@ -2149,7 +2155,12 @@ function getEffectiveModelSurfaces() {
2149
2155
  function resolveModel(surfaceId, models, fallback) {
2150
2156
  return models?.[surfaceId] ?? fallback ?? orgDefaultModels[surfaceId] ?? MODEL_SURFACES[surfaceId].default;
2151
2157
  }
2152
- var MODEL_SURFACES, ALLOWED_MODELS_BY_TYPE, orgDefaultModels;
2158
+ function resolveParentModel(models, fallback, buildModel) {
2159
+ const override = buildModel ? filterModelPicks({ parent: buildModel }).parent : void 0;
2160
+ const baseline = resolveModel("parent", models, fallback);
2161
+ return { baseline, effective: override ?? baseline };
2162
+ }
2163
+ var MODEL_SURFACES, DEFAULT_SUGGEST_COMPACT_AT, TEXT_MODELS, DEFAULT_CONTEXT_LIMITS, ALLOWED_MODELS_BY_TYPE, orgDefaultModels;
2153
2164
  var init_surfaces = __esm({
2154
2165
  "src/models/surfaces.ts"() {
2155
2166
  "use strict";
@@ -2241,36 +2252,53 @@ var init_surfaces = __esm({
2241
2252
  userPickable: false
2242
2253
  }
2243
2254
  };
2255
+ DEFAULT_SUGGEST_COMPACT_AT = 3e5;
2256
+ TEXT_MODELS = {
2257
+ // Anthropic 1M-context, flat-priced.
2258
+ "claude-5-opus": { forceCompactAt: 85e4 },
2259
+ "claude-4-8-opus": { forceCompactAt: 85e4 },
2260
+ "claude-4-7-opus": { forceCompactAt: 85e4 },
2261
+ // Anthropic 1M-context with 2x long-context pricing above 200K input.
2262
+ "claude-4-6-opus": { forceCompactAt: 85e4, suggestCompactAt: 18e4 },
2263
+ "claude-4-6-sonnet": { forceCompactAt: 85e4, suggestCompactAt: 18e4 },
2264
+ "claude-fable-5": { forceCompactAt: 85e4 },
2265
+ "claude-fable-5-1": { forceCompactAt: 85e4 },
2266
+ "claude-5-sonnet": { forceCompactAt: 85e4 },
2267
+ // OpenAI gpt-5.5/5.6: ~1M window, but the usable input ceiling under
2268
+ // `truncation: 'auto'` is ~794K (output + reasoning reserve), and all
2269
+ // rates double above 272K input.
2270
+ "gpt-5.5": { forceCompactAt: 6e5, suggestCompactAt: 25e4 },
2271
+ "gpt-5.6-sol": { forceCompactAt: 6e5, suggestCompactAt: 25e4 },
2272
+ "gpt-5.6-terra": { forceCompactAt: 6e5, suggestCompactAt: 25e4 },
2273
+ "gpt-5.6-luna": { forceCompactAt: 6e5, suggestCompactAt: 25e4 },
2274
+ // Google ~1M-context; only 3.1-pro is tiered (higher rates above 200K).
2275
+ "gemini-3-pro": { forceCompactAt: 85e4 },
2276
+ "gemini-3.1-pro": { forceCompactAt: 85e4, suggestCompactAt: 18e4 },
2277
+ "gemini-3-flash": { forceCompactAt: 85e4 },
2278
+ "gemini-3.5-flash": { forceCompactAt: 85e4 },
2279
+ "gemini-3.7-flash": { forceCompactAt: 85e4 },
2280
+ // 256K window; its 200K pricing tier sits above the gate, so no nudge.
2281
+ "grok-build-0.1": { forceCompactAt: 18e4 },
2282
+ "grok-4.5": { forceCompactAt: 4e5 },
2283
+ // 500K window
2284
+ "grok-4.6": { forceCompactAt: 4e5 },
2285
+ // 500K window
2286
+ "glm-5.2": { forceCompactAt: 85e4 },
2287
+ "muse-spark-1.1": { forceCompactAt: 85e4 },
2288
+ "kimi-k2-7-code": { forceCompactAt: 2e5 },
2289
+ // 262K window
2290
+ "kimi-k3": { forceCompactAt: 85e4 },
2291
+ "deepseek-v4-flash-0731": { forceCompactAt: 85e4 },
2292
+ "qwen3.8-2.4t-a95b-deepinfra": { forceCompactAt: 2e5 },
2293
+ // 262K window
2294
+ "qwen3.8-27b-deepinfra": { forceCompactAt: 2e5 },
2295
+ // 262K window
2296
+ "minimax-m3": { forceCompactAt: 42e4 }
2297
+ // 524K window
2298
+ };
2299
+ DEFAULT_CONTEXT_LIMITS = { forceCompactAt: 85e4 };
2244
2300
  ALLOWED_MODELS_BY_TYPE = {
2245
- text: [
2246
- "claude-5-opus",
2247
- "claude-4-8-opus",
2248
- "claude-4-7-opus",
2249
- "claude-4-6-opus",
2250
- "claude-4-6-sonnet",
2251
- "claude-fable-5",
2252
- "claude-5-sonnet",
2253
- "gpt-5.5",
2254
- "gpt-5.6-sol",
2255
- "gpt-5.6-terra",
2256
- "gpt-5.6-luna",
2257
- "gemini-3-pro",
2258
- "gemini-3.1-pro",
2259
- "gemini-3-flash",
2260
- "gemini-3.5-flash",
2261
- "gemini-3.7-flash",
2262
- "grok-build-0.1",
2263
- "grok-4.5",
2264
- "grok-4.6",
2265
- "glm-5.2",
2266
- "muse-spark-1.1",
2267
- "kimi-k2-7-code",
2268
- "kimi-k3",
2269
- "deepseek-v4-flash-0731",
2270
- "qwen3.8-2.4t-a95b-deepinfra",
2271
- "qwen3.8-27b-deepinfra",
2272
- "minimax-m3"
2273
- ]
2301
+ text: Object.keys(TEXT_MODELS)
2274
2302
  // vision: undefined — unconstrained
2275
2303
  // image_generation: undefined — unconstrained
2276
2304
  };
@@ -3163,7 +3191,7 @@ var init_catalog = __esm({
3163
3191
  init_assets();
3164
3192
  INTRO = `Platform capabilities most apps don't use, so their references are kept out of this prompt rather than competing for your attention on every task \u2014 not because they're marginal.
3165
3193
 
3166
- Read what follows as part of what the platform can do, not as a lookup table. Recognising that one of these fits a feature is your job, and proposing one is fair game \u2014 several of them are the difference between an app that works and an app worth showing off. When a trigger fires, load the reference with loadSkill before writing the code rather than after. Loading is cheap and expected; guessing at one of these APIs is not.
3194
+ Read what follows as part of what the platform can do, not as a lookup table. Recognising that one of these fits a feature is your job, and proposing one is fair game \u2014 several of them are the difference between an app that works and an app worth showing off. When a trigger fires, load the reference with loadSkill before writing the spec or code rather than after. Loading is cheap and expected; guessing at one of these APIs is not.
3167
3195
 
3168
3196
  A loaded reference drops out of the conversation once it ages out. Re-read it at the path listed with readFile whenever you need it again.`;
3169
3197
  catalog = buildSkillCatalog({
@@ -8920,6 +8948,12 @@ function classify(line, matches) {
8920
8948
  tailClean = SEPARATORS_ONLY.test(gap);
8921
8949
  }
8922
8950
  }
8951
+ function chipCase(label) {
8952
+ if (label.startsWith("`")) {
8953
+ return label;
8954
+ }
8955
+ return label.charAt(0).toUpperCase() + label.slice(1);
8956
+ }
8923
8957
  function parseSuggestions(raw) {
8924
8958
  if (!raw.includes(SUGGEST_MARKER)) {
8925
8959
  return { text: raw, suggestions: [] };
@@ -8949,7 +8983,7 @@ function parseSuggestions(raw) {
8949
8983
  const key = `${m.label}::${m.message}`;
8950
8984
  if (!seen.has(key)) {
8951
8985
  seen.add(key);
8952
- suggestions.push({ label: m.label, message: m.message });
8986
+ suggestions.push({ label: chipCase(m.label), message: m.message });
8953
8987
  }
8954
8988
  }
8955
8989
  let rebuilt = "";
@@ -9400,10 +9434,12 @@ async function runTurn(params) {
9400
9434
  onBackgroundComplete
9401
9435
  } = params;
9402
9436
  const tools2 = getToolDefinitions();
9403
- const buildModelOverride = buildModel ? filterModelPicks({ parent: buildModel }).parent : void 0;
9404
- const baseline = resolveModel("parent", state.models, model);
9405
- const parentModel = buildModelOverride ?? baseline;
9406
- const modelOverride = buildModelOverride && buildModelOverride !== baseline ? { from: baseline } : void 0;
9437
+ const { baseline, effective: parentModel } = resolveParentModel(
9438
+ state.models,
9439
+ model,
9440
+ buildModel
9441
+ );
9442
+ const modelOverride = parentModel !== baseline ? { from: baseline } : void 0;
9407
9443
  const totalAttachments = entries.reduce(
9408
9444
  (n, e) => n + (e.attachments?.length ?? 0),
9409
9445
  0
@@ -9411,7 +9447,7 @@ async function runTurn(params) {
9411
9447
  log14.info("Turn started", {
9412
9448
  requestId,
9413
9449
  model,
9414
- buildModel: buildModelOverride,
9450
+ buildModel: modelOverride ? parentModel : void 0,
9415
9451
  toolCount: tools2.length,
9416
9452
  ...entries.length > 1 && { entryCount: entries.length },
9417
9453
  ...totalAttachments > 0 && { attachmentCount: totalAttachments }
@@ -10550,12 +10586,13 @@ function loadPassiveResults() {
10550
10586
  }
10551
10587
  return [];
10552
10588
  }
10553
- function writeStats(stats, queue, passiveResults) {
10589
+ function writeStats(stats, queue, passiveResults, suggestCompactAt) {
10554
10590
  try {
10555
10591
  writeFileAtomicSync(
10556
10592
  STATS_FILE,
10557
10593
  JSON.stringify({
10558
10594
  ...stats,
10595
+ suggestCompactAt,
10559
10596
  queue,
10560
10597
  passiveResults
10561
10598
  })
@@ -10766,7 +10803,7 @@ var headless_exports = {};
10766
10803
  __export(headless_exports, {
10767
10804
  HeadlessSession: () => HeadlessSession
10768
10805
  });
10769
- var log17, EXTERNAL_TOOL_TIMEOUT_MS, LONG_RUNNING_TOOLS, LONG_RUNNING_TOOL_TIMEOUT_MS, USER_FACING_TOOLS, FORCED_COMPACTION_THRESHOLD_TOKENS, HeadlessSession;
10806
+ var log17, EXTERNAL_TOOL_TIMEOUT_MS, LONG_RUNNING_TOOLS, LONG_RUNNING_TOOL_TIMEOUT_MS, USER_FACING_TOOLS, HeadlessSession;
10770
10807
  var init_headless = __esm({
10771
10808
  "src/headless/index.ts"() {
10772
10809
  "use strict";
@@ -10797,7 +10834,6 @@ var init_headless = __esm({
10797
10834
  "confirmDestructiveAction",
10798
10835
  "presentPublishPlan"
10799
10836
  ]);
10800
- FORCED_COMPACTION_THRESHOLD_TOKENS = 85e4;
10801
10837
  HeadlessSession = class {
10802
10838
  // Configuration
10803
10839
  opts;
@@ -11070,7 +11106,15 @@ var init_headless = __esm({
11070
11106
  /** Persist sessionStats + queue snapshot + passive pen to .remy-stats.json. */
11071
11107
  persistStats() {
11072
11108
  this.sessionStats.updatedAt = Date.now();
11073
- writeStats(this.sessionStats, this.queue.snapshot(), this.passivePen);
11109
+ const suggestCompactAt = getSuggestCompactAt(
11110
+ resolveParentModel(this.state.models, this.opts.model).effective
11111
+ );
11112
+ writeStats(
11113
+ this.sessionStats,
11114
+ this.queue.snapshot(),
11115
+ this.passivePen,
11116
+ suggestCompactAt
11117
+ );
11074
11118
  }
11075
11119
  //////////////////////////////////////////////////////////////////////////////
11076
11120
  // Background completions (tool-block mutation; message delivery via queue)
@@ -11100,8 +11144,9 @@ var init_headless = __esm({
11100
11144
  saveSession(this.state);
11101
11145
  }
11102
11146
  /**
11103
- * Forced compaction gate. If lastContextSize exceeds the threshold, compact
11104
- * before letting the upcoming turn run. Coalesces with any in-flight
11147
+ * Forced compaction gate. If lastContextSize exceeds the upcoming turn's
11148
+ * model threshold (per-model — see ModelContextLimits in models/surfaces),
11149
+ * compact before letting the turn run. Coalesces with any in-flight
11105
11150
  * compaction (e.g., one already started by /compact or a tool call). No
11106
11151
  * timeout — compaction takes as long as it takes.
11107
11152
  *
@@ -11112,13 +11157,15 @@ var init_headless = __esm({
11112
11157
  * On compaction failure we don't bail — the turn proceeds and surfaces any
11113
11158
  * downstream overflow through the existing "prompt is too long" path.
11114
11159
  */
11115
- async runForcedCompactionIfNeeded(requestId) {
11116
- if (this.sessionStats.lastContextSize <= FORCED_COMPACTION_THRESHOLD_TOKENS) {
11160
+ async runForcedCompactionIfNeeded(requestId, parentModel) {
11161
+ const threshold = getContextLimits(parentModel).forceCompactAt;
11162
+ if (this.sessionStats.lastContextSize <= threshold) {
11117
11163
  return;
11118
11164
  }
11119
11165
  log17.info("Forced compaction gate triggered", {
11120
11166
  contextSize: this.sessionStats.lastContextSize,
11121
- threshold: FORCED_COMPACTION_THRESHOLD_TOKENS,
11167
+ threshold,
11168
+ model: parentModel,
11122
11169
  requestId
11123
11170
  });
11124
11171
  try {
@@ -11600,7 +11647,12 @@ var init_headless = __esm({
11600
11647
  }
11601
11648
  return steered;
11602
11649
  };
11603
- await this.runForcedCompactionIfNeeded(requestId);
11650
+ const parentModel = resolveParentModel(
11651
+ this.state.models,
11652
+ this.opts.model,
11653
+ params.buildModel
11654
+ ).effective;
11655
+ await this.runForcedCompactionIfNeeded(requestId, parentModel);
11604
11656
  try {
11605
11657
  await runTurn({
11606
11658
  state: this.state,
@@ -224,7 +224,7 @@ The app projected as an MCP server for *external* AI agents to drive (Claude Des
224
224
 
225
225
  ## Agent (Conversational Interface)
226
226
 
227
- A conversational interface where the app's own LLM orchestrates its methods as tools — its own personality, system prompt, and model config (the inverse of MCP). Chat runs as the authenticated user, so every tool call carries that user's roles, and the config must declare an `auth` block (`{ "requireUser": boolean, "requireRole"?: string[] }`) gating who may chat at all. **Load the `agentInterfaces` skill** before authoring `src/interfaces/agent.md` or building the chat UI.
227
+ A conversational interface where the app's own LLM orchestrates its methods as tools — its own personality, system prompt, and model config (the inverse of MCP). Chat runs as the authenticated user, so every tool call carries that user's roles, and the config must declare an `auth` block (`{ "requireUser": boolean, "requireRole"?: string[] }`) gating who may chat at all. When the app includes a conversational surface, this is the default way to build it — not a custom chat UI over per-turn method calls. **Load the `agentInterfaces` skill** when a conversational feature enters the plan, and before authoring `src/interfaces/agent.md` or building the chat UI.
228
228
 
229
229
  ## Voice (Realtime Conversation)
230
230
 
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: Agent Interfaces
3
3
  what: Conversational AI as a first-class interface to the app — an LLM with authenticated, per-user access to the app's methods as tools, paired with a streaming chat UI. The platform handles auth, tool dispatch, threads, and streaming, so the work is authorship: who the agent is, which methods it can reach, and how each one is described to it. Any app whose methods do something interesting can be projected into a conversation this way, often as its most compelling surface. This reference covers the whole feature — writing the agent spec, compiling it, and building the chat frontend.
4
- when: Before authoring `src/interfaces/agent.md`, compiling `dist/interfaces/agent/`, or building an agent's chat UI.
4
+ when: When a conversational surface — a chat, an assistant — is part of the plan or spec, load it before deciding how that feature is built: this is the default architecture for conversation. Also before authoring `src/interfaces/agent.md`, compiling `dist/interfaces/agent/`, or building an agent's chat UI.
5
5
  ---
6
6
 
7
7
  # Building Agent Interfaces
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: Inbound Email
3
3
  what: The app has its own email address, and mail sent to it runs a method. Every address on the app's subdomain routes to one handler, so `support@`, `receipts@` and `anything@` all arrive without registering anything — branch on the recipient in code. Attachments arrive as CDN URLs the platform has already uploaded, threading headers come through intact so replies land in the same conversation, and a verified custom domain both receives and sends under the app's own brand. "Forward a receipt and the app files it" is a real feature that costs one method.
4
- when: Before writing an email-handler method, adding an `email` interface, or promising anything about what happens when a user emails the app.
4
+ when: The moment "email something to the app" would be a natural feature — load it before scoping that idea. Also before writing an email-handler method, adding an `email` interface, or promising anything about what happens when a user emails the app.
5
5
  ---
6
6
 
7
7
  # Inbound Email Interfaces
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: MCP Interfaces
3
3
  what: Ships the app as an MCP server, so external AI agents — Claude Desktop, Cursor, anyone's agent — can drive it as a tool surface. The platform hosts the server, handles auth, and derives every tool's input schema from the method contract, so there is no protocol code to write: the work is choosing which methods an outsider should see and describing them well enough for a stranger to use correctly. Cheap to add to an app that already has methods, and it puts the app inside the tools its users already work in.
4
- when: Before authoring `src/interfaces/mcp.md`, deciding which of the app's methods an external agent gets to see, or writing the MCP interface config.
4
+ when: The moment the plan wants the app reachable from Claude, Cursor, ChatGPT, or a user's own agents — load it before proposing how. Also before authoring `src/interfaces/mcp.md`, deciding which of the app's methods an external agent gets to see, or writing the MCP interface config.
5
5
  ---
6
6
 
7
7
  # MCP Interfaces
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: REST API
3
3
  what: A designed, documented REST surface over the app's methods — named routes, path and query params, resource groupings, and a generated OpenAPI spec. This is distinct from the method endpoints every app already has: those exist automatically and need nothing from you. This is for when the API itself is the product, or when something outside the app has to integrate against stable URLs rather than internal method names.
4
- when: Before authoring `src/interfaces/api.md` or adding an `api` interface with designed routes. Also load it when a method needs the raw HTTP request (headers, unparsed body).
4
+ when: The moment the plan calls for a public or partner-facing API, or anything outside the app integrating against stable URLs — load it before designing that surface. Also before authoring `src/interfaces/api.md` or adding an `api` interface with designed routes, and when a method needs the raw HTTP request (headers, unparsed body).
5
5
  ---
6
6
 
7
7
  # REST API Interfaces
@@ -128,6 +128,8 @@ Re-running it is free (documents are content-addressed), so it's safe to keep in
128
128
 
129
129
  Align scenario data to the vibe of the app - construct data that feels like it fits.
130
130
 
131
+ For seeded email addresses, use `remy@mindstudio.ai` or `@example.com` addresses — the platform sinks mail to both, whereas an invented domain that resolves to a real mail server bounces and damages sending reputation when the app later emails its seeded users.
132
+
131
133
  ### Scenario Images
132
134
 
133
135
  When scenarios seed data that includes image URLs (profile photos, product images, cover art, etc.), ask the `visualDesignExpert` to generate a small batch of images that fit the app's aesthetic before writing the scenario code. A handful of bespoke photos make scenarios feel dramatically more real than placeholder services. Use the CDN URLs directly in your `db.push()` calls.
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: Scheduled Jobs
3
3
  what: Methods that run on a schedule, declared as cron expressions in an interface config and synced to the platform on deploy. Nothing to host and no scheduler to run — a job is a method plus a schedule line.
4
- when: Before adding a `cron` interface or writing a method meant to run on a timer.
4
+ when: The moment anything should happen on a schedule — digests, syncs, reminders, cleanup — load it before designing that feature. Also before adding a `cron` interface or writing a method meant to run on a timer.
5
5
  ---
6
6
 
7
7
  # Cron Interfaces
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: Voice Interfaces
3
3
  what: Realtime voice conversation as a first-class interface — the user talks to the app and its voice agent talks back in sub-second, interruptible speech, calling the app's methods mid-conversation as the authenticated user. The platform handles the media transport, turn-taking, barge-in, and transcripts, so the work is authorship — a persona written for the ear, a small toolset where every tool carries a latency class, and descriptions that say results out loud. Any app whose methods do something interesting can pick up a voice, and it is often the most impressive surface it has.
4
- when: Before authoring `src/interfaces/voice.md`, choosing a voice model or pipeline, deciding which methods a voice agent gets, building the voice UI with `createVoiceClient()`, or working out why a voice agent behaved the way it did on a call.
4
+ when: The moment a feature wants to be spoken — a phone line, a talking assistant, hands-free operation — load it before deciding how that feature is built. Also before authoring `src/interfaces/voice.md`, choosing a voice model or pipeline, deciding which methods a voice agent gets, building the voice UI with `createVoiceClient()`, or working out why a voice agent behaved the way it did on a call.
5
5
  ---
6
6
 
7
7
  # Building Voice Interfaces
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: Webhooks
3
3
  what: Inbound HTTP endpoints that run a method synchronously, routed by a secret in the URL rather than by an auth header — which is what makes them the right fit for provider callbacks from Stripe, GitHub, Shopify, Slack or Twilio, since those senders can't present a bearer token. Signature verification works natively off the raw request body.
4
- when: Before adding a `webhook` interface or writing a method that receives a provider callback. Also load it before reaching for any confirmation-token, polling, or proxy workaround for inbound HTTP — those aren't needed here.
4
+ when: The moment an external service needs to call the app — Stripe events, GitHub pushes, Twilio callbacks — load it before designing the receiving path. Also before adding a `webhook` interface or writing a method that receives a provider callback, and before reaching for any confirmation-token, polling, or proxy workaround for inbound HTTP — those aren't needed here.
5
5
  ---
6
6
 
7
7
  # Webhook Interfaces
@@ -18,7 +18,7 @@ The scaffold starts with these spec files that cover the full picture of the app
18
18
  - **`src/interfaces/@brand/voice.md`** — voice and terminology: tone, error messages, word choices
19
19
  - **`src/roadmap/`** — feature roadmap. One file per feature (`type: roadmap`). See "Roadmap" below.
20
20
 
21
- These are starting points, not constraints. Create as many spec files as the project needs — the `src/` folder is your workspace and every `.md` file in it becomes compilation context. If the app has substantial content (presentation slides, copy, lesson plans, menu items, quiz questions), put it in its own file (`src/content.md`, `src/slides.md`, `src/menu.md`, etc.) rather than cramming it into `app.md` or `web.md`. If the domain is complex, split `app.md` into multiple files by area (`src/billing.md`, `src/approvals.md`). Add interface specs for other interface types (`api.md`, `webhook.md`, `cron.md`, `email.md`, `mcp.md`, `agent.md`, `voice.md`) if the app uses them. Each of those has a skill carrying its spec format and config — `restApi`, `webhooks`, `scheduledJobs`, `inboundEmail`, `mcpInterfaces`, `agentInterfaces`, `voiceInterfaces` — and you should load the relevant one before writing the spec rather than after, since the spec is what the config is compiled from. For external HTTP the choice is between two of them: the Webhook interface handles inbound provider webhooks (Stripe, GitHub) via secret-in-URL routing, while the API interface covers bearer-auth sync endpoints, public REST APIs, and batch tools. Organize however serves clarity — the platform reads the entire `src/` folder.
21
+ These are starting points, not constraints. Create as many spec files as the project needs — the `src/` folder is your workspace and every `.md` file in it becomes compilation context. If the app has substantial content (presentation slides, copy, lesson plans, menu items, quiz questions), put it in its own file (`src/content.md`, `src/slides.md`, `src/menu.md`, etc.) rather than cramming it into `app.md` or `web.md`. If the domain is complex, split `app.md` into multiple files by area (`src/billing.md`, `src/approvals.md`). Add interface specs for other interface types (`api.md`, `webhook.md`, `cron.md`, `email.md`, `mcp.md`, `agent.md`, `voice.md`) if the app uses them. Each of those has a skill carrying its spec format and config — `restApi`, `webhooks`, `scheduledJobs`, `inboundEmail`, `mcpInterfaces`, `agentInterfaces`, `voiceInterfaces` — and you should load the relevant one before writing the spec rather than after, since the spec is what the config is compiled from. When the app includes a conversational surface, the `agent` interface is the default way to build it — make that architecture call with the `agentInterfaces` skill loaded, and treat `voice` the same way for spoken features. For external HTTP the choice is between two of them: the Webhook interface handles inbound provider webhooks (Stripe, GitHub) via secret-in-URL routing, while the API interface covers bearer-auth sync endpoints, public REST APIs, and batch tools. Organize however serves clarity — the platform reads the entire `src/` folder.
22
22
 
23
23
  Remember: users care about look and feel as much as (and often more than) underlying data structures. Don't treat the brand and interface specs as an afterthought — for many users, the visual identity and voice are the first things they want to get right.
24
24
 
@@ -6,7 +6,7 @@ You are a browser smoke test agent. You verify that features work end to end by
6
6
  - Browser unavailability is an infrastructure issue, not a test failure. If `browserCommand` reports the browser is unavailable or drops mid-test, the test is **inconclusive** — do not retry, do not attribute it to app brokenness. Report "test inconclusive: browser unavailable" and stop.
7
7
 
8
8
  ## Tester Persona
9
- The user is watching the automation happen on their screen in real-time. When typing into forms or inputs, behave like a realistic user of this specific app. Use the app context (if provided) to understand the audience and tone. Type the way that audience would actually type — not formal, not robotic. The app developer's name is Remy - you must use that and the email remy@mindstudio.ai as the basis for any testing that requires a persona.
9
+ The user is watching the automation happen on their screen in real-time. When typing into forms or inputs, behave like a realistic user of this specific app. Use the app context (if provided) to understand the audience and tone. Type the way that audience would actually type — not formal, not robotic. The app developer's name is Remy - you must use that and the email remy@mindstudio.ai as the basis for any testing that requires a persona. When a form needs additional email addresses, use `@example.com` ones — the platform sinks mail to those and to remy@mindstudio.ai, but an invented domain on a real mail server bounces and damages sending reputation.
10
10
 
11
11
  ### Auth Testing
12
12
  When the content you need to test is behind authentication, use the `setupBrowser` tool to automatically pre-authenticate instead of manually navigating login flows. This mints a session cookie, reloads the page with the authenticated state, and optionally navigates to a starting path. Use `remy@mindstudio.ai` as the email. If the test requires a specific role, pass it in the `roles` array. For apps that use "Sign in with Remy" (delegated auth, no email/phone login), `setupBrowser` authenticates as the developer's own Remy identity automatically — call it the same way; the email is ignored for these apps, and `roles` still apply. Do not try to click through the "Sign in with Remy" button manually.
@@ -71,6 +71,8 @@ When a plan includes multiple screens/API calls, always note this item for the d
71
71
 
72
72
  Two things NOT to flag. A deterministic sequence that happens to touch several methods — if the order is known up front and no judgment is involved, imperative code is correct and cheaper. And the fire-and-forget background pattern, where a method kicks off `runTask()` and writes the result back in `.then()` with failures landing in `.catch()`: that write-back is doing status management the agent can't do for itself. That one is the recommended pattern, not a smell.
73
73
 
74
+ - **A task agent (or per-turn generation) powering a multi-turn chat.** The inverse of the item above. Multi-turn conversational features belong on the platform's Agent interface — native streaming, thread management, and tool dispatch. A custom chat UI calling a method per message rebuilds all of that by hand. Flag it and point at the `agent` interface (the developer's `agentInterfaces` skill has the full picture) unless there's a strong, specific reason the agent interface can't serve the feature.
75
+
74
76
  - **Task agent tool descriptions missing or recycled.** When a plan exposes app methods to `runTask()`, each entry should carry an inline `description` written for *that* task — when to call it, when not to, what to do with the result. Falling back to the method's own description is allowed but usually too generic, and the description is the main thing determining whether the agent uses the tool correctly. Flag entries with no description, or the same description pasted across different tasks.
75
77
 
76
78
  - **A method exposed as a task tool that itself calls `runTask()`.** Task agents can't nest through method tools — the inner call is rejected at runtime. Flag it and suggest flattening the decomposition. An inline function tool calling `runTask()` is different: it is NOT blocked by the platform (no depth cap applies), so accidental recursion runs unbounded and burns credits per turn. Flag it unless the nesting is clearly deliberate and bounded.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@mindstudio-ai/remy",
3
- "version": "0.1.311",
3
+ "version": "0.1.313",
4
4
  "description": "Remy coding agent",
5
5
  "repository": {
6
6
  "type": "git",