@yeaft/webchat-agent 1.0.507 → 1.0.509

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@yeaft/webchat-agent",
3
- "version": "1.0.507",
3
+ "version": "1.0.509",
4
4
  "description": "Remote worker agent for Yeaft Web Code Agent — connects the native Yeaft engine, CLI providers, and workbench tools",
5
5
  "main": "index.js",
6
6
  "type": "module",
@@ -11,12 +11,60 @@
11
11
  import { existsSync, readFileSync } from 'fs';
12
12
  import { join } from 'path';
13
13
  import { DEFAULT_YEAFT_DIR } from './init.js';
14
- import { normalizeProviderModels, parseModelRef, serializeModelForPersistence } from './models.js';
14
+ import { getModelEffortOptions, normalizeEffort, normalizeProviderModels, parseModelRef, resolveMaxOutputTokens, serializeModelForPersistence } from './models.js';
15
+ import { inferProtocolFromModelId } from './llm/router.js';
15
16
  import { clampYeaftField, normaliseTelemetrySection, normaliseYeaftSection } from './config.js';
16
17
  import { normaliseBrowserRuntimeSection, validateBrowserRuntimeUpdate } from '../browser-runtime/config.js';
17
18
  import { normalizePluginConfig } from './plugins.js';
18
19
  import { mutateAgentConfig, readAgentConfigForWrite } from './config-store.js';
19
- import { isGitHubCopilotProvider, serializeKnownProviderForPersistence } from './llm/known-providers.js';
20
+ import { isGitHubCopilotProvider, normalizeKnownProviderForRuntime, serializeKnownProviderForPersistence } from './llm/known-providers.js';
21
+
22
+ /** Agent-owned model catalog, with the same protocol/capability resolution as runtime. */
23
+ function quickSendModels(config) {
24
+ return (Array.isArray(config.providers) ? config.providers : []).flatMap(raw => {
25
+ const provider = normalizeKnownProviderForRuntime(raw);
26
+ return normalizeProviderModels(provider).map(model => ({
27
+ id: model.id,
28
+ ref: provider.name ? `${provider.name}/${model.id}` : model.id,
29
+ provider: provider.name,
30
+ label: model.id,
31
+ effortOptions: getModelEffortOptions(model.id, {
32
+ ...model,
33
+ protocol: model.protocol || provider.protocol || inferProtocolFromModelId(model.id) || 'openai-responses',
34
+ }),
35
+ maxOutput: resolveMaxOutputTokens(model.id, { modelInfo: model }),
36
+ }));
37
+ });
38
+ }
39
+
40
+ /** Validate inside mutateAgentConfig's lock, against the resulting provider catalog. */
41
+ function validateQuickSends(value, config) {
42
+ if (!Array.isArray(value) || value.length > 5) throw new Error('quickSends must be an array of at most 5 items');
43
+ const models = quickSendModels(config);
44
+ const ids = new Set();
45
+ return value.map(item => {
46
+ if (!item || typeof item !== 'object' || Array.isArray(item)) throw new Error('Each quick send must be an object');
47
+ const id = typeof item.id === 'string' ? item.id.trim() : '';
48
+ const name = typeof item.name === 'string' ? item.name.trim() : '';
49
+ if (!id || id.length > 128 || ids.has(id)) throw new Error('Quick send id must be unique and 1–128 characters');
50
+ if (!name || name.length > 80) throw new Error('Quick send name must be 1–80 characters');
51
+ ids.add(id);
52
+ const ref = typeof item.model === 'string' ? item.model.trim() : '';
53
+ const matches = models.filter(model => model.ref === ref);
54
+ const candidates = matches.length ? matches : models.filter(model => model.id === ref);
55
+ if (candidates.length !== 1) throw new Error(`Quick send model must exist and be unambiguous: ${ref}`);
56
+ const model = candidates[0];
57
+ const effort = item.effort ?? null;
58
+ if (effort !== null && (!normalizeEffort(effort) || !model.effortOptions.includes(effort))) {
59
+ throw new Error(`Quick send effort is not available for ${model.ref}`);
60
+ }
61
+ const maxOutputTokens = item.maxOutputTokens ?? null;
62
+ if (maxOutputTokens !== null && (!Number.isSafeInteger(maxOutputTokens) || maxOutputTokens <= 0 || maxOutputTokens > model.maxOutput)) {
63
+ throw new Error(`Quick send maxOutputTokens must be a positive integer no greater than ${model.maxOutput}`);
64
+ }
65
+ return { id, name, model: model.ref, effort, maxOutputTokens };
66
+ });
67
+ }
20
68
 
21
69
  /**
22
70
  * Read config.json before any public mutation. A missing file is a valid
@@ -44,7 +92,7 @@ function readLocalLlmConfig(dir) {
44
92
  const configPath = join(root, 'config.json');
45
93
 
46
94
  if (!existsSync(configPath)) {
47
- return { providers: [], primaryModel: null, fastModel: null, language: 'en', needsSetup: true };
95
+ return { providers: [], primaryModel: null, fastModel: null, language: 'en', quickSends: [], availableModels: [], needsSetup: true };
48
96
  }
49
97
 
50
98
  const raw = readFileSync(configPath, 'utf8');
@@ -56,6 +104,8 @@ function readLocalLlmConfig(dir) {
56
104
  fastModel: json.fastModel || null,
57
105
  language: json.language || 'en',
58
106
  debug: json.debug === true,
107
+ quickSends: Array.isArray(json.quickSends) ? json.quickSends : [],
108
+ availableModels: quickSendModels(json),
59
109
  needsSetup: providers.length === 0 || providers.every(p => p.apiKey === 'proxy' || p.apiKey === '' || (!p.apiKey && !p.credentialProvider)),
60
110
  };
61
111
  }
@@ -169,6 +219,7 @@ export function updateLlmConfig(update, dir) {
169
219
  if (update.providers !== undefined) normalizeManagedModelDefaults(existing);
170
220
  if (update.language !== undefined) existing.language = update.language;
171
221
  if (update.debug !== undefined) existing.debug = update.debug === true;
222
+ if (update.quickSends !== undefined) existing.quickSends = validateQuickSends(update.quickSends, existing);
172
223
 
173
224
  const agentConfig = {
174
225
  providers: Array.isArray(existing.providers) ? existing.providers : [],
@@ -176,6 +227,8 @@ export function updateLlmConfig(update, dir) {
176
227
  fastModel: existing.fastModel || null,
177
228
  language: existing.language || 'en',
178
229
  debug: existing.debug === true,
230
+ quickSends: Array.isArray(existing.quickSends) ? existing.quickSends : [],
231
+ availableModels: quickSendModels(existing),
179
232
  };
180
233
  return {
181
234
  ...agentConfig,
@@ -216,7 +269,7 @@ export function getYeaftSettings(dir) {
216
269
  * (LLM provider / model fields are untouched) and validates each field:
217
270
  * `maxConcurrentThreads` must be 1..50, `autoArchiveIdleDays` must be
218
271
  * 1..3650, `recentTurnsLimit` must be 1..500, `relatedTurnsLimit` must be
219
- * 0..10 (0 disables related recall). Dream limits are read-only
272
+ * 0..5 (0 disables related recall). Dream limits are read-only
220
273
  * runtime defaults here; invalid values are rejected
221
274
  * outright so the UI sees an error rather than silently reverting — a
222
275
  * silent revert would make "I set it to 100 and nothing happened"
@@ -256,8 +309,8 @@ export function updateYeaftSettings(update, dir) {
256
309
  if (update.relatedTurnsLimit !== undefined) {
257
310
  const value = clampYeaftField(update.relatedTurnsLimit, 'relatedTurnsLimit');
258
311
  const n = Number(update.relatedTurnsLimit);
259
- if (value === null || n < 0 || n > 10) {
260
- return { error: 'relatedTurnsLimit must be between 0 and 10' };
312
+ if (value === null || n < 0 || n > 5) {
313
+ return { error: 'relatedTurnsLimit must be between 0 and 5' };
261
314
  }
262
315
  }
263
316
 
package/yeaft/config.js CHANGED
@@ -55,8 +55,8 @@ const DEFAULTS = {
55
55
  // through history pagination/search. Range: 1–500.
56
56
  yeaftRecentTurnsLimit: 20,
57
57
  // Same-Session related Q&A turns, selected by deterministic full-text rules.
58
- // Range: 0–10; 0 disables related recall without changing recent history.
59
- yeaftRelatedTurnsLimit: 8,
58
+ // Range: 0–5; 0 disables related recall without changing recent history.
59
+ yeaftRelatedTurnsLimit: 5,
60
60
  // CLAUDE.md / AGENTS.md project-doc cap, in bytes. Mirrors Codex's
61
61
  // `project_doc_max_bytes`. 0 disables the feature (no project-doc
62
62
  // block is injected). Hand-edited values are NOT clamped — we let
@@ -306,7 +306,7 @@ export function clampYeaftField(v, field) {
306
306
  let hi;
307
307
  if (field === 'maxConcurrentThreads') { lo = 1; hi = 50; }
308
308
  else if (field === 'recentTurnsLimit') { lo = 1; hi = 500; }
309
- else if (field === 'relatedTurnsLimit') { lo = 0; hi = 10; }
309
+ else if (field === 'relatedTurnsLimit') { lo = 0; hi = 5; }
310
310
  else { lo = 1; hi = 3650; } // autoArchiveIdleDays
311
311
  return Math.min(hi, Math.max(lo, Math.floor(n)));
312
312
  }
@@ -10,7 +10,7 @@ import {
10
10
  normalizeLiteralSearch,
11
11
  } from './visible-entry.js';
12
12
  import { fingerprintConversationSources } from './history-index-state.js';
13
- import { extractRecallTerms, scoreRecallTurn, RECALL_LIMITS } from './recall-relevance.js';
13
+ import { extractRecallTerms, scoreRecallTurn, normalizeRecallLimit, RECALL_LIMITS } from './recall-relevance.js';
14
14
 
15
15
  const INDEX_SCHEMA_VERSION = 2;
16
16
  const SHORT_BLOOM_BYTES = 256;
@@ -433,7 +433,7 @@ function recallTurns(request) {
433
433
  const generation = Number(indexMeta.generation) || 0;
434
434
  const terms = extractRecallTerms(request.prompt);
435
435
  const cap = (value, fallback) => Math.min(fallback, Math.max(1, Math.floor(Number(value) || fallback)));
436
- const limit = cap(request.limit, 10);
436
+ const limit = normalizeRecallLimit(request.limit);
437
437
  const maxTurnRows = cap(request.maxTurnRows, RECALL_LIMITS.maxTurnRows);
438
438
  const maxTurnBytes = cap(request.maxTurnBytes, RECALL_LIMITS.maxTurnBytes);
439
439
  const maxReadBytes = cap(request.maxReadBytes, RECALL_LIMITS.maxReadBytes);
@@ -445,6 +445,7 @@ function recallTurns(request) {
445
445
  limits: { ...RECALL_LIMITS, maxTurnRows, maxTurnBytes, maxReadBytes, limit },
446
446
  ...indexStats(indexMeta),
447
447
  };
448
+ if (limit === 0) return { turns: [], meta: { ...meta, status: 'disabled', reason: 'disabled' } };
448
449
  const sourceBefore = currentSourceToken();
449
450
  if (sourceBefore.fingerprint !== indexMeta.raw_source_fingerprint) {
450
451
  return { turns: [], meta: { ...meta, status: 'not_ready', reason: 'stale_result' } };
@@ -2,7 +2,7 @@ import { Worker } from 'node:worker_threads';
2
2
  import { existsSync, mkdirSync, readFileSync, rmSync } from 'node:fs';
3
3
  import { dirname } from 'node:path';
4
4
  import { writeAtomic } from '../storage/atomic.js';
5
- import { extractRecallTerms, scoreRecallTurn } from './recall-relevance.js';
5
+ import { extractRecallTerms, scoreRecallTurn, normalizeRecallLimit } from './recall-relevance.js';
6
6
  import {
7
7
  conversationIndexDatabasePath,
8
8
  conversationIndexManifestPath,
@@ -556,8 +556,8 @@ export async function searchConversationIndex(ownerRoot, sessionId, query, opts
556
556
  /**
557
557
  * Recall complete visible user/assistant turns from this owner-root Session.
558
558
  * beforeSeq excludes that user turn and all later rows (exclusive seq fence).
559
- * limit defaults to 8, capped at 10. maxTurnRows/maxTurnBytes/maxReadBytes may
560
- * only lower hard worker caps. Canonical message IDs identify entries; their
559
+ * limit defaults to 5, capped at 5; 0 disables recall.
560
+ * maxTurnRows/maxTurnBytes/maxReadBytes may only lower hard worker caps. Canonical message IDs identify entries; their
561
561
  * sourceMessageIds preserve all persisted identities, including aggregated VP
562
562
  * replies. The first cold call yields not_ready and starts a background build;
563
563
  * later calls wait up to 1500ms for readiness and the worker read, without
@@ -569,6 +569,8 @@ export async function searchConversationIndex(ownerRoot, sessionId, query, opts
569
569
  */
570
570
  export async function recallConversationTurns(ownerRoot, sessionId, prompt, opts = {}) {
571
571
  const terms = extractRecallTerms(prompt);
572
+ const limit = normalizeRecallLimit(opts.limit);
573
+ if (limit === 0) return { turns: [], meta: { status: 'disabled', reason: 'disabled', terms } };
572
574
  const eligibility = scoreRecallTurn(terms, terms.join(' '));
573
575
  if (!eligibility.score) {
574
576
  return { turns: [], meta: { status: 'ready', reason: eligibility.reason, terms } };
@@ -577,7 +579,7 @@ export async function recallConversationTurns(ownerRoot, sessionId, prompt, opts
577
579
  return await requestConversationHistoryIndex(ownerRoot, sessionId, 'recall-turns', {
578
580
  prompt: typeof prompt === 'string' ? prompt.slice(0, 4096) : '',
579
581
  beforeSeq: opts.beforeSeq,
580
- limit: Math.min(10, Math.max(1, Math.floor(Number(opts.limit) || 8))),
582
+ limit,
581
583
  maxTurnRows: opts.maxTurnRows,
582
584
  maxTurnBytes: opts.maxTurnBytes,
583
585
  maxReadBytes: opts.maxReadBytes,
@@ -1564,6 +1564,43 @@ export class ConversationStore {
1564
1564
  return pairSanitize(messages);
1565
1565
  }
1566
1566
 
1567
+ /**
1568
+ * Provider-only chronological history. UI pages deliberately cap raw rows;
1569
+ * that cap is not a turn limit and must not truncate a tool-heavy query's
1570
+ * context. Yield between bounded scan batches; never mutate the transcript.
1571
+ * Stable user identities, not equal prompt text, define human turns.
1572
+ * @returns {Promise<object[]>} Complete past turns before the durable user fence.
1573
+ */
1574
+ async loadProviderHistoryBySession(sessionId, turnsLimit = 20, { beforeSeq = Infinity } = {}) {
1575
+ if (!sessionId || !(turnsLimit > 0)) return [];
1576
+ const kept = [];
1577
+ const pending = [];
1578
+ const identities = new Set();
1579
+ let scanned = 0;
1580
+ let bytes = 0;
1581
+ for (const row of this.#iterateSessionRows(sessionId, { beforeSeq, desc: true })) {
1582
+ scanned += 1;
1583
+ bytes += Buffer.byteLength(JSON.stringify(row));
1584
+ if (scanned > 32768 || bytes > 64 * 1024 * 1024) {
1585
+ const error = new Error('Recent history scan limit reached before completing the requested turn window');
1586
+ error.code = 'HISTORY_RECENT_SCAN_LIMIT';
1587
+ throw error;
1588
+ }
1589
+ if (scanned % 64 === 0) await new Promise(resolve => setImmediate(resolve));
1590
+ if (!row || row.sessionId !== sessionId || isHiddenConversationRow(row)) continue;
1591
+ if (row.role === 'user') {
1592
+ const identity = row.clientMessageId ? `client:${row.clientMessageId}` : `message:${row.id}`;
1593
+ identities.add(identity);
1594
+ kept.push(...pending.splice(0), row);
1595
+ // The requested oldest user closes the reverse scan. Do not read the
1596
+ // previous turn's tool tail just to discover one more user boundary.
1597
+ if (identities.size >= turnsLimit) break;
1598
+ } else pending.push(row);
1599
+ }
1600
+ // Pending rows without their opening user are not a complete past turn.
1601
+ return pairSanitize(kept.reverse());
1602
+ }
1603
+
1567
1604
  /**
1568
1605
  * Load every hot message stamped with `sessionId`.
1569
1606
  *
@@ -4,7 +4,35 @@ const MAX_PROMPT_CHARS = 4096;
4
4
  const MAX_TERMS = 8;
5
5
  const STOP_WORDS = new Set(`a an and are as at be been but by can could did do does for from had has have how i if in is it its me my of on or our please should so that the their them there these they this to was we were what when where which who why will with would you your about again before earlier previous remember recall history message messages conversation turn turns tell show find help need want use using make get know explain answer question code file project problem fix work task test tests implementation implement change thanks continue revisit discuss discussed discussion decide decided follow-up
6
6
  的 了 是 在 和 与 或 我 你 他 她 它 我们 你们 他们 这个 那个 什么 怎么 如何 为什么 请 请问 帮 帮我 帮忙 可以 能 不能 是否 需要 想 要 再 还 又 也 就 都 把 将 给 对 从 到 上 下 中 里 有 没有 一下 一些 一个 这些 那些 之前 以前 上次 刚才 历史 记得 回忆 召回 消息 对话 问题 回答 内容 事情 继续 现在 今天 昨天 后来 然后 相关 具体 代码 文件 项目 实现 修改 功能 测试 方案 方法 工作 任务 谢谢 好的 好 看看 查找 搜索 查询 处理 解决 进行 使用 讨论 提到 记忆 总结 回顾 提醒`.split(/\s+/u));
7
+ for (const word of `system service config configuration settings build run error errors issue issues status result results request requests response responses check checks update updates version versions default option options limit limits page data user users model models tool tools server client input output changes details new old
8
+ 系统 服务 配置 设置 构建 运行 错误 报错 状态 结果 请求 响应 检查 更新 版本 默认 选项 限制 页面 数据 用户 模型 工具 服务端 客户端 输入 输出 改动 详细 新 旧`.split(/\s+/u)) STOP_WORDS.add(word);
7
9
  const segmenter = new Intl.Segmenter('zh', { granularity: 'word' });
10
+ const MIN_COVERAGE = 0.6;
11
+ const MIN_DISTINCTIVENESS = 0.5;
12
+ const MIN_SCORE = 6.5;
13
+
14
+ // Paths and concrete issue IDs are strong anchors even when a bounded sample
15
+ // contains many discussions of that exact reference. Mere camel/snake casing
16
+ // does not make an otherwise common word distinctive.
17
+ function isExactReference(term) {
18
+ return term.includes('/')
19
+ || /\.[a-z][a-z0-9]{0,7}$/iu.test(term)
20
+ || /^[a-z][a-z0-9]*(?:[-_:][a-z0-9]+)*[-_:]\d[\da-z_-]*$/iu.test(term)
21
+ || /^(?:[a-z0-9]+[-_:])*[a-f0-9]{8,}$/iu.test(term);
22
+ }
23
+
24
+ function isGenericTerm(term) {
25
+ if (STOP_WORDS.has(term.toLocaleLowerCase())) return true;
26
+ const words = term.replace(/([a-z])([A-Z])/gu, '$1 $2').toLocaleLowerCase().split(/[\s_.:@-]+/u);
27
+ return words.every(word => STOP_WORDS.has(word));
28
+ }
29
+
30
+ /** Automatic recall accepts zero and never exceeds five, even for raw callers. */
31
+ export function normalizeRecallLimit(value) {
32
+ const number = typeof value === 'number' || typeof value === 'string' && value.trim()
33
+ ? Number(value) : NaN;
34
+ return Number.isFinite(number) ? Math.min(5, Math.max(0, Math.floor(number))) : 5;
35
+ }
8
36
 
9
37
  function isIdentifier(term) {
10
38
  return /[\p{L}\d][_.:/@-][\p{L}\d]/u.test(term)
@@ -20,7 +48,7 @@ export function extractRecallTerms(prompt) {
20
48
  ? prompt.slice(0, MAX_PROMPT_CHARS).replace(/@vp-[A-Za-z0-9_-]+\b/gu, ' ') : '';
21
49
  const tokens = [];
22
50
  // Keep paths, issue IDs, snake_case and camelCase intact before segmentation.
23
- const rest = text.replace(/[\p{L}\p{N}]+(?:[_.:/@-][\p{L}\p{N}]+)+|[A-Za-z][A-Za-z0-9]*/gu, token => {
51
+ const rest = text.replace(/(?:\.{0,2}\/)?[\p{L}\p{N}]+(?:[_.:/@-][\p{L}\p{N}]+)+|[A-Za-z][A-Za-z0-9]*/gu, token => {
24
52
  tokens.push(token);
25
53
  return ' ';
26
54
  });
@@ -44,7 +72,7 @@ export function extractRecallTerms(prompt) {
44
72
  const seen = new Set();
45
73
  const terms = tokens.filter(term => {
46
74
  const key = term.toLocaleLowerCase();
47
- if (seen.has(key) || STOP_WORDS.has(key) || /^\d+$/u.test(key)
75
+ if (seen.has(key) || isGenericTerm(term) || /^\d+$/u.test(key)
48
76
  || Array.from(key).length < 2 || key.length > 96) return false;
49
77
  seen.add(key);
50
78
  return true;
@@ -59,17 +87,21 @@ export function extractRecallTerms(prompt) {
59
87
  * Returns explainable rejection reasons rather than weak positive matches.
60
88
  */
61
89
  export function scoreRecallTurn(promptOrTerms, text, stats = {}) {
62
- const terms = (Array.isArray(promptOrTerms) ? promptOrTerms : extractRecallTerms(promptOrTerms)).slice(0, MAX_TERMS);
90
+ const terms = (Array.isArray(promptOrTerms) ? promptOrTerms : extractRecallTerms(promptOrTerms))
91
+ .filter(term => typeof term === 'string' && !isGenericTerm(term)).slice(0, MAX_TERMS);
63
92
  const body = String(text || '').toLocaleLowerCase();
64
93
  const matchedTerms = terms.filter(term => {
65
94
  const key = term.toLocaleLowerCase();
66
- if (/^[a-z0-9_]+$/u.test(key)) {
95
+ if (/^[a-z0-9_]+$/u.test(key) || isIdentifier(term)) {
67
96
  const escaped = key.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
68
- return new RegExp(`(?<![a-z0-9_])${escaped}(?![a-z0-9_])`, 'u').test(body);
97
+ const boundary = isIdentifier(term) ? 'a-z0-9_.:/@-' : 'a-z0-9_';
98
+ const suffix = isIdentifier(term) ? '(?![a-z0-9_/@-]|[.:][a-z0-9_])' : `(?![${boundary}])`;
99
+ return new RegExp(`(?<![${boundary}])${escaped}${suffix}`, 'u').test(body);
69
100
  }
70
101
  return body.includes(key);
71
102
  });
72
103
  const identifierMatches = matchedTerms.filter(isIdentifier);
104
+ const exactReferenceMatches = matchedTerms.filter(isExactReference);
73
105
  const independentTerms = matchedTerms.filter(term => !matchedTerms.some(other => (
74
106
  other !== term && other.toLocaleLowerCase().includes(term.toLocaleLowerCase())
75
107
  )));
@@ -79,16 +111,17 @@ export function scoreRecallTurn(promptOrTerms, text, stats = {}) {
79
111
  const frequency = stats.termDocumentFrequency?.[term.toLocaleLowerCase()];
80
112
  return sampleSize >= 8 && Number.isFinite(frequency) ? 1 - frequency / sampleSize : 1;
81
113
  })) : 0;
114
+ const rawScore = Math.round((independentTerms.length * 2 + identifierMatches.length * 4
115
+ + coverage * 2 + distinctiveness) * 100) / 100;
82
116
  let reason = 'relevant';
83
117
  if (!terms.length) reason = 'generic_prompt';
84
118
  else if (!matchedTerms.length) reason = 'no_match';
85
119
  else if (!identifierMatches.length && independentTerms.length < 2) reason = 'insufficient_keywords';
86
- else if (!identifierMatches.length && coverage < 0.5) reason = 'low_coverage';
87
- else if (!identifierMatches.length && distinctiveness < 0.15) reason = 'low_distinctiveness';
88
- const score = reason === 'relevant'
89
- ? Math.round((independentTerms.length * 2 + identifierMatches.length * 4 + coverage * 2 + distinctiveness) * 100) / 100
90
- : 0;
91
- return { score, matchedTerms, reason, coverage, distinctiveness, identifierMatches, sampleSize };
120
+ else if (coverage < MIN_COVERAGE) reason = 'low_coverage';
121
+ else if (!exactReferenceMatches.length && distinctiveness < MIN_DISTINCTIVENESS) reason = 'low_distinctiveness';
122
+ else if (rawScore < MIN_SCORE) reason = 'low_score';
123
+ const score = reason === 'relevant' ? rawScore : 0;
124
+ return { score, matchedTerms, reason, coverage, distinctiveness, identifierMatches, exactReferenceMatches, sampleSize };
92
125
  }
93
126
 
94
127
  export const RECALL_LIMITS = Object.freeze({
package/yeaft/engine.js CHANGED
@@ -46,7 +46,7 @@ import { perfNowMs, recordAgentPerfTrace } from './perf-trace.js';
46
46
  const MAIN_THREAD_ID = 'main';
47
47
  import { pickEffort, parseEffortPrefix, snapshotEffortDecision } from './effort.js';
48
48
  import { bindProviderState } from './llm/provider-state.js';
49
- import { DEFAULT_CONTEXT_WINDOW, normalizeEffort, resolveContextWindow, resolveModel } from './models.js';
49
+ import { DEFAULT_CONTEXT_WINDOW, getModelInfo, normalizeEffort, parseModelRef, resolveContextWindow, resolveMaxOutputTokens, resolveModel } from './models.js';
50
50
  import { lookupModelLimitSync } from './llm/models-dev.js';
51
51
  import { attachRouterPlan, extractPriorPlan, stripMetaForWire } from './router/continuity.js';
52
52
  import { resolveThinking } from './router/thinking.js';
@@ -1882,7 +1882,7 @@ export class Engine {
1882
1882
  }
1883
1883
  }
1884
1884
 
1885
- async *#queryLifecycle({ prompt, promptParts = null, messages = [], signal, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null } = {}) {
1885
+ async *#queryLifecycle({ prompt, promptParts = null, messages = [], signal, turnConfig = null, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null } = {}) {
1886
1886
  if (!prompt || typeof prompt !== 'string' || !prompt.trim()) {
1887
1887
  const error = new Error('prompt is required and must be a non-empty string');
1888
1888
  yield {
@@ -1974,7 +1974,7 @@ export class Engine {
1974
1974
  try {
1975
1975
  this.#currentThreadId = threadId || MAIN_THREAD_ID;
1976
1976
  this.#currentCausalRootId = effectiveCausalRootId;
1977
- yield* this.#runQuery({ prompt: effectivePrompt, promptParts: effectivePromptParts, messages, signal: runSignal, userEffort: explicitUserEffort, scenario, isSubAgent, parentEffortDecision, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds, projectInstruction, projectLabel, vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted, currentUserMessage, causalRootId: effectiveCausalRootId, getCurrentTodos, setCurrentTodos, askUser, threadId: this.#currentThreadId, vpTurnId, drainPendingUserMessages, prepareProviderRequest, startProviderRequest, finishProviderRequest, failProviderRequest, closePendingUserInput, collabToolPolicy: effectiveCollabToolPolicy, explicitSkillName: parsedSkill.skillName, retryLifecycle });
1977
+ yield* this.#runQuery({ prompt: effectivePrompt, promptParts: effectivePromptParts, messages, signal: runSignal, turnConfig: turnConfig ? { model: turnConfig.model, effort: turnConfig.effort, maxOutputTokens: turnConfig.maxOutputTokens } : null, userEffort: explicitUserEffort, scenario, isSubAgent, parentEffortDecision, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds, projectInstruction, projectLabel, vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted, currentUserMessage, causalRootId: effectiveCausalRootId, getCurrentTodos, setCurrentTodos, askUser, threadId: this.#currentThreadId, vpTurnId, drainPendingUserMessages, prepareProviderRequest, startProviderRequest, finishProviderRequest, failProviderRequest, closePendingUserInput, collabToolPolicy: effectiveCollabToolPolicy, explicitSkillName: parsedSkill.skillName, retryLifecycle });
1978
1978
  } finally {
1979
1979
  // Closing the async generator at a visible retry boundary means the
1980
1980
  // continuation never reached a provider. Keep it out of history and
@@ -2029,7 +2029,7 @@ export class Engine {
2029
2029
  * in a try/finally without indenting the whole loop.
2030
2030
  * @private
2031
2031
  */
2032
- async *#runQuery({ prompt, promptParts = null, messages, signal, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null, explicitSkillName = null, retryLifecycle }) {
2032
+ async *#runQuery({ prompt, promptParts = null, messages, signal, turnConfig = null, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null, explicitSkillName = null, retryLifecycle }) {
2033
2033
 
2034
2034
  const effectiveCollabToolPolicy = collabToolPolicy === COLLAB_TOOL_POLICY.SINGLE_VP || collabToolPolicy === COLLAB_TOOL_POLICY.MULTI_VP
2035
2035
  ? collabToolPolicy
@@ -2103,13 +2103,18 @@ export class Engine {
2103
2103
  ...(internalTrigger ? { internal: true } : { userAuthored: true }),
2104
2104
  }, { sessionId: runtimeSessionId });
2105
2105
  }
2106
- // Native Session history, including internal wakeups, never comes from
2107
- // Dream. Work Center and scoped child agents keep their memory contracts.
2106
+ // Native Session history, including internal wakeups, comes from the
2107
+ // canonical message transcript. Dream memory loading is temporarily
2108
+ // disabled for every scenario (including Work Center and child agents)
2109
+ // while message-history recall is evaluated as its replacement. Keep the
2110
+ // downstream memory pipeline intact behind this single switch so it can be
2111
+ // restored without migrating or deleting persisted memory data.
2108
2112
  const useMessageHistory = scenario !== 'work-item' && !!runtimeSessionId && !vpPersona?.subAgent;
2109
- const useDreamMemory = scenario === 'work-item' || !!vpPersona?.subAgent
2110
- || (!runtimeSessionId && !internalTrigger);
2111
- const recentTurnCap = this.#config.yeaft?.recentTurnsLimit ?? 20;
2112
- const relatedTurnCap = this.#config.yeaft?.relatedTurnsLimit ?? 8;
2113
+ // const useDreamMemory = scenario === 'work-item' || !!vpPersona?.subAgent
2114
+ // || (!runtimeSessionId && !internalTrigger);
2115
+ const useDreamMemory = false;
2116
+ const recentTurnCap = Math.max(20, this.#config.yeaft?.recentTurnsLimit ?? 20);
2117
+ const relatedTurnCap = Math.min(5, this.#config.yeaft?.relatedTurnsLimit ?? 5);
2113
2118
  let relatedHistoryTurns = [];
2114
2119
  let historyRecallMeta = { source: 'messages', status: 'disabled' };
2115
2120
  if (useMessageHistory && !internalTrigger && this.#conversationStore?.loadRecentBySession) {
@@ -2125,7 +2130,9 @@ export class Engine {
2125
2130
  const beforeSeq = Number.isFinite(persistedQueryUser?.seq)
2126
2131
  ? persistedQueryUser.seq : parseSeqFromId(persistedQueryUser?.id);
2127
2132
  if (Number.isFinite(beforeSeq)) {
2128
- const tail = this.#conversationStore.loadRecentBySession(runtimeSessionId, recentTurnCap, { beforeSeq });
2133
+ const loadHistory = this.#conversationStore.loadProviderHistoryBySession
2134
+ || this.#conversationStore.loadRecentBySession;
2135
+ const tail = await loadHistory.call(this.#conversationStore, runtimeSessionId, recentTurnCap, { beforeSeq });
2129
2136
  messages = tail.filter(m => parseSeqFromId(m.id) < beforeSeq
2130
2137
  && (m.role !== 'tool' || !queryVpId || m.speakerVpId === queryVpId))
2131
2138
  .map(m => {
@@ -2550,7 +2557,7 @@ export class Engine {
2550
2557
  // `refreshConfig()` may publish a new Session model while a stream or a
2551
2558
  // tool is running. Apply it only before the next provider request; the
2552
2559
  // current request keeps the snapshot captured below.
2553
- let currentModel = this.#config.model;
2560
+ let currentModel = turnConfig?.model || this.#config.model;
2554
2561
  let primaryModelAtLastBoundary = currentModel;
2555
2562
  let cumulativeInputTokens = 0;
2556
2563
  let cumulativeOutputTokens = 0;
@@ -2595,7 +2602,7 @@ export class Engine {
2595
2602
  // Keep a retry fallback selected by this query; replacing it here would
2596
2603
  // turn an exhausted primary into an endless retry loop.
2597
2604
  if (currentModel === primaryModelAtLastBoundary) {
2598
- const refreshedPrimaryModel = this.#config.model;
2605
+ const refreshedPrimaryModel = turnConfig?.model || this.#config.model;
2599
2606
  if (refreshedPrimaryModel !== primaryModelAtLastBoundary) {
2600
2607
  currentModel = refreshedPrimaryModel;
2601
2608
  primaryModelAtLastBoundary = refreshedPrimaryModel;
@@ -2607,6 +2614,21 @@ export class Engine {
2607
2614
  // this request. Fallback retries intentionally retain their selected
2608
2615
  // model, but still use the current policy and configured effort.
2609
2616
  const requestConfig = { ...this.#config };
2617
+ // Overlay only the request snapshot. Never publish temporary settings via
2618
+ // refreshConfig or mutate the shared config / AdapterRouter catalog.
2619
+ if (turnConfig) {
2620
+ requestConfig.model = currentModel;
2621
+ const entry = requestConfig.availableModels?.find(model => model.ref === currentModel);
2622
+ requestConfig.modelInfo = getModelInfo(parseModelRef(currentModel).modelId, entry) || null;
2623
+ if (turnConfig.effort != null) requestConfig.modelEffort = turnConfig.effort;
2624
+ // Blank means the selected model's default cap, not the Session's
2625
+ // previous model budget. Re-resolve at every boundary (including fallback
2626
+ // retries / catalog refresh), without leaking the old global ceiling.
2627
+ const outputLimit = resolveMaxOutputTokens(parseModelRef(currentModel).modelId, {
2628
+ modelInfo: requestConfig.modelInfo,
2629
+ });
2630
+ requestConfig.maxOutputTokens = Math.min(turnConfig.maxOutputTokens ?? outputLimit, outputLimit);
2631
+ }
2610
2632
  // Capture the matching provider catalog in the same synchronous boundary
2611
2633
  // as config/model. Preflight may yield user/task events before the stream
2612
2634
  // is built, but one request must never mix two refresh revisions.
@@ -2803,7 +2825,7 @@ export class Engine {
2803
2825
  // effect at the next loop. A caller override or `/effort` prefix stays
2804
2826
  // fixed for this query and still wins over live Session config.
2805
2827
  const configuredEffort = normalizeEffort(requestConfig.modelEffort);
2806
- const requestUserEffort = userEffort || configuredEffort || null;
2828
+ const requestUserEffort = normalizeEffort(turnConfig?.effort) || userEffort || configuredEffort || null;
2807
2829
  let resolvedEffort = pickEffort({ scenario, toolLoopTurns, userEffort: requestUserEffort });
2808
2830
 
2809
2831
  // DESIGN.md §9.16: thinking-mode precedence chain. When a VP
@@ -2864,8 +2886,8 @@ export class Engine {
2864
2886
  const buckets = useMessageHistory ? buildHistoryBuckets(conversationMessages, {
2865
2887
  prompt,
2866
2888
  relatedTurns: relatedHistoryTurns,
2867
- recentTurnCap: requestConfig.yeaft?.recentTurnsLimit ?? 20,
2868
- relatedTurnCap: requestConfig.yeaft?.relatedTurnsLimit ?? 8,
2889
+ recentTurnCap: Math.max(20, requestConfig.yeaft?.recentTurnsLimit ?? 20),
2890
+ relatedTurnCap: Math.min(5, requestConfig.yeaft?.relatedTurnsLimit ?? 5),
2869
2891
  messageTokenBudget: historyBudget,
2870
2892
  currentTurnStartIndex: turnStartIdx,
2871
2893
  language: requestConfig.language,