@yeaft/webchat-agent 1.0.507 → 1.0.509
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/connection/message-router.js +1 -1
- package/local-runtime/server/handlers/agent-sync.js +1 -0
- package/local-runtime/server/handlers/client-conversation.js +38 -2
- package/local-runtime/server/handlers/client-misc.js +1 -1
- package/local-runtime/server/handlers/client-workbench.js +7 -1
- package/local-runtime/server/workbench-route.js +11 -4
- package/local-runtime/version.json +1 -1
- package/local-runtime/web/app.bundle.js +212 -99
- package/local-runtime/web/app.bundle.js.gz +0 -0
- package/local-runtime/web/index.html +2 -2
- package/local-runtime/web/style.bundle.css +1 -1
- package/local-runtime/web/style.bundle.css.gz +0 -0
- package/package.json +1 -1
- package/yeaft/config-api.js +59 -6
- package/yeaft/config.js +3 -3
- package/yeaft/conversation/history-index-worker.js +3 -2
- package/yeaft/conversation/history-index.js +6 -4
- package/yeaft/conversation/persist.js +37 -0
- package/yeaft/conversation/recall-relevance.js +44 -11
- package/yeaft/engine.js +38 -16
- package/yeaft/history-window.js +48 -53
- package/yeaft/prompts.js +3 -0
- package/yeaft/session.js +16 -45
- package/yeaft/sub-agent/runner.js +6 -4
- package/yeaft/web-bridge.js +130 -39
|
Binary file
|
package/package.json
CHANGED
package/yeaft/config-api.js
CHANGED
|
@@ -11,12 +11,60 @@
|
|
|
11
11
|
import { existsSync, readFileSync } from 'fs';
|
|
12
12
|
import { join } from 'path';
|
|
13
13
|
import { DEFAULT_YEAFT_DIR } from './init.js';
|
|
14
|
-
import { normalizeProviderModels, parseModelRef, serializeModelForPersistence } from './models.js';
|
|
14
|
+
import { getModelEffortOptions, normalizeEffort, normalizeProviderModels, parseModelRef, resolveMaxOutputTokens, serializeModelForPersistence } from './models.js';
|
|
15
|
+
import { inferProtocolFromModelId } from './llm/router.js';
|
|
15
16
|
import { clampYeaftField, normaliseTelemetrySection, normaliseYeaftSection } from './config.js';
|
|
16
17
|
import { normaliseBrowserRuntimeSection, validateBrowserRuntimeUpdate } from '../browser-runtime/config.js';
|
|
17
18
|
import { normalizePluginConfig } from './plugins.js';
|
|
18
19
|
import { mutateAgentConfig, readAgentConfigForWrite } from './config-store.js';
|
|
19
|
-
import { isGitHubCopilotProvider, serializeKnownProviderForPersistence } from './llm/known-providers.js';
|
|
20
|
+
import { isGitHubCopilotProvider, normalizeKnownProviderForRuntime, serializeKnownProviderForPersistence } from './llm/known-providers.js';
|
|
21
|
+
|
|
22
|
+
/** Agent-owned model catalog, with the same protocol/capability resolution as runtime. */
|
|
23
|
+
function quickSendModels(config) {
|
|
24
|
+
return (Array.isArray(config.providers) ? config.providers : []).flatMap(raw => {
|
|
25
|
+
const provider = normalizeKnownProviderForRuntime(raw);
|
|
26
|
+
return normalizeProviderModels(provider).map(model => ({
|
|
27
|
+
id: model.id,
|
|
28
|
+
ref: provider.name ? `${provider.name}/${model.id}` : model.id,
|
|
29
|
+
provider: provider.name,
|
|
30
|
+
label: model.id,
|
|
31
|
+
effortOptions: getModelEffortOptions(model.id, {
|
|
32
|
+
...model,
|
|
33
|
+
protocol: model.protocol || provider.protocol || inferProtocolFromModelId(model.id) || 'openai-responses',
|
|
34
|
+
}),
|
|
35
|
+
maxOutput: resolveMaxOutputTokens(model.id, { modelInfo: model }),
|
|
36
|
+
}));
|
|
37
|
+
});
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/** Validate inside mutateAgentConfig's lock, against the resulting provider catalog. */
|
|
41
|
+
function validateQuickSends(value, config) {
|
|
42
|
+
if (!Array.isArray(value) || value.length > 5) throw new Error('quickSends must be an array of at most 5 items');
|
|
43
|
+
const models = quickSendModels(config);
|
|
44
|
+
const ids = new Set();
|
|
45
|
+
return value.map(item => {
|
|
46
|
+
if (!item || typeof item !== 'object' || Array.isArray(item)) throw new Error('Each quick send must be an object');
|
|
47
|
+
const id = typeof item.id === 'string' ? item.id.trim() : '';
|
|
48
|
+
const name = typeof item.name === 'string' ? item.name.trim() : '';
|
|
49
|
+
if (!id || id.length > 128 || ids.has(id)) throw new Error('Quick send id must be unique and 1–128 characters');
|
|
50
|
+
if (!name || name.length > 80) throw new Error('Quick send name must be 1–80 characters');
|
|
51
|
+
ids.add(id);
|
|
52
|
+
const ref = typeof item.model === 'string' ? item.model.trim() : '';
|
|
53
|
+
const matches = models.filter(model => model.ref === ref);
|
|
54
|
+
const candidates = matches.length ? matches : models.filter(model => model.id === ref);
|
|
55
|
+
if (candidates.length !== 1) throw new Error(`Quick send model must exist and be unambiguous: ${ref}`);
|
|
56
|
+
const model = candidates[0];
|
|
57
|
+
const effort = item.effort ?? null;
|
|
58
|
+
if (effort !== null && (!normalizeEffort(effort) || !model.effortOptions.includes(effort))) {
|
|
59
|
+
throw new Error(`Quick send effort is not available for ${model.ref}`);
|
|
60
|
+
}
|
|
61
|
+
const maxOutputTokens = item.maxOutputTokens ?? null;
|
|
62
|
+
if (maxOutputTokens !== null && (!Number.isSafeInteger(maxOutputTokens) || maxOutputTokens <= 0 || maxOutputTokens > model.maxOutput)) {
|
|
63
|
+
throw new Error(`Quick send maxOutputTokens must be a positive integer no greater than ${model.maxOutput}`);
|
|
64
|
+
}
|
|
65
|
+
return { id, name, model: model.ref, effort, maxOutputTokens };
|
|
66
|
+
});
|
|
67
|
+
}
|
|
20
68
|
|
|
21
69
|
/**
|
|
22
70
|
* Read config.json before any public mutation. A missing file is a valid
|
|
@@ -44,7 +92,7 @@ function readLocalLlmConfig(dir) {
|
|
|
44
92
|
const configPath = join(root, 'config.json');
|
|
45
93
|
|
|
46
94
|
if (!existsSync(configPath)) {
|
|
47
|
-
return { providers: [], primaryModel: null, fastModel: null, language: 'en', needsSetup: true };
|
|
95
|
+
return { providers: [], primaryModel: null, fastModel: null, language: 'en', quickSends: [], availableModels: [], needsSetup: true };
|
|
48
96
|
}
|
|
49
97
|
|
|
50
98
|
const raw = readFileSync(configPath, 'utf8');
|
|
@@ -56,6 +104,8 @@ function readLocalLlmConfig(dir) {
|
|
|
56
104
|
fastModel: json.fastModel || null,
|
|
57
105
|
language: json.language || 'en',
|
|
58
106
|
debug: json.debug === true,
|
|
107
|
+
quickSends: Array.isArray(json.quickSends) ? json.quickSends : [],
|
|
108
|
+
availableModels: quickSendModels(json),
|
|
59
109
|
needsSetup: providers.length === 0 || providers.every(p => p.apiKey === 'proxy' || p.apiKey === '' || (!p.apiKey && !p.credentialProvider)),
|
|
60
110
|
};
|
|
61
111
|
}
|
|
@@ -169,6 +219,7 @@ export function updateLlmConfig(update, dir) {
|
|
|
169
219
|
if (update.providers !== undefined) normalizeManagedModelDefaults(existing);
|
|
170
220
|
if (update.language !== undefined) existing.language = update.language;
|
|
171
221
|
if (update.debug !== undefined) existing.debug = update.debug === true;
|
|
222
|
+
if (update.quickSends !== undefined) existing.quickSends = validateQuickSends(update.quickSends, existing);
|
|
172
223
|
|
|
173
224
|
const agentConfig = {
|
|
174
225
|
providers: Array.isArray(existing.providers) ? existing.providers : [],
|
|
@@ -176,6 +227,8 @@ export function updateLlmConfig(update, dir) {
|
|
|
176
227
|
fastModel: existing.fastModel || null,
|
|
177
228
|
language: existing.language || 'en',
|
|
178
229
|
debug: existing.debug === true,
|
|
230
|
+
quickSends: Array.isArray(existing.quickSends) ? existing.quickSends : [],
|
|
231
|
+
availableModels: quickSendModels(existing),
|
|
179
232
|
};
|
|
180
233
|
return {
|
|
181
234
|
...agentConfig,
|
|
@@ -216,7 +269,7 @@ export function getYeaftSettings(dir) {
|
|
|
216
269
|
* (LLM provider / model fields are untouched) and validates each field:
|
|
217
270
|
* `maxConcurrentThreads` must be 1..50, `autoArchiveIdleDays` must be
|
|
218
271
|
* 1..3650, `recentTurnsLimit` must be 1..500, `relatedTurnsLimit` must be
|
|
219
|
-
* 0..
|
|
272
|
+
* 0..5 (0 disables related recall). Dream limits are read-only
|
|
220
273
|
* runtime defaults here; invalid values are rejected
|
|
221
274
|
* outright so the UI sees an error rather than silently reverting — a
|
|
222
275
|
* silent revert would make "I set it to 100 and nothing happened"
|
|
@@ -256,8 +309,8 @@ export function updateYeaftSettings(update, dir) {
|
|
|
256
309
|
if (update.relatedTurnsLimit !== undefined) {
|
|
257
310
|
const value = clampYeaftField(update.relatedTurnsLimit, 'relatedTurnsLimit');
|
|
258
311
|
const n = Number(update.relatedTurnsLimit);
|
|
259
|
-
if (value === null || n < 0 || n >
|
|
260
|
-
return { error: 'relatedTurnsLimit must be between 0 and
|
|
312
|
+
if (value === null || n < 0 || n > 5) {
|
|
313
|
+
return { error: 'relatedTurnsLimit must be between 0 and 5' };
|
|
261
314
|
}
|
|
262
315
|
}
|
|
263
316
|
|
package/yeaft/config.js
CHANGED
|
@@ -55,8 +55,8 @@ const DEFAULTS = {
|
|
|
55
55
|
// through history pagination/search. Range: 1–500.
|
|
56
56
|
yeaftRecentTurnsLimit: 20,
|
|
57
57
|
// Same-Session related Q&A turns, selected by deterministic full-text rules.
|
|
58
|
-
// Range: 0–
|
|
59
|
-
yeaftRelatedTurnsLimit:
|
|
58
|
+
// Range: 0–5; 0 disables related recall without changing recent history.
|
|
59
|
+
yeaftRelatedTurnsLimit: 5,
|
|
60
60
|
// CLAUDE.md / AGENTS.md project-doc cap, in bytes. Mirrors Codex's
|
|
61
61
|
// `project_doc_max_bytes`. 0 disables the feature (no project-doc
|
|
62
62
|
// block is injected). Hand-edited values are NOT clamped — we let
|
|
@@ -306,7 +306,7 @@ export function clampYeaftField(v, field) {
|
|
|
306
306
|
let hi;
|
|
307
307
|
if (field === 'maxConcurrentThreads') { lo = 1; hi = 50; }
|
|
308
308
|
else if (field === 'recentTurnsLimit') { lo = 1; hi = 500; }
|
|
309
|
-
else if (field === 'relatedTurnsLimit') { lo = 0; hi =
|
|
309
|
+
else if (field === 'relatedTurnsLimit') { lo = 0; hi = 5; }
|
|
310
310
|
else { lo = 1; hi = 3650; } // autoArchiveIdleDays
|
|
311
311
|
return Math.min(hi, Math.max(lo, Math.floor(n)));
|
|
312
312
|
}
|
|
@@ -10,7 +10,7 @@ import {
|
|
|
10
10
|
normalizeLiteralSearch,
|
|
11
11
|
} from './visible-entry.js';
|
|
12
12
|
import { fingerprintConversationSources } from './history-index-state.js';
|
|
13
|
-
import { extractRecallTerms, scoreRecallTurn, RECALL_LIMITS } from './recall-relevance.js';
|
|
13
|
+
import { extractRecallTerms, scoreRecallTurn, normalizeRecallLimit, RECALL_LIMITS } from './recall-relevance.js';
|
|
14
14
|
|
|
15
15
|
const INDEX_SCHEMA_VERSION = 2;
|
|
16
16
|
const SHORT_BLOOM_BYTES = 256;
|
|
@@ -433,7 +433,7 @@ function recallTurns(request) {
|
|
|
433
433
|
const generation = Number(indexMeta.generation) || 0;
|
|
434
434
|
const terms = extractRecallTerms(request.prompt);
|
|
435
435
|
const cap = (value, fallback) => Math.min(fallback, Math.max(1, Math.floor(Number(value) || fallback)));
|
|
436
|
-
const limit =
|
|
436
|
+
const limit = normalizeRecallLimit(request.limit);
|
|
437
437
|
const maxTurnRows = cap(request.maxTurnRows, RECALL_LIMITS.maxTurnRows);
|
|
438
438
|
const maxTurnBytes = cap(request.maxTurnBytes, RECALL_LIMITS.maxTurnBytes);
|
|
439
439
|
const maxReadBytes = cap(request.maxReadBytes, RECALL_LIMITS.maxReadBytes);
|
|
@@ -445,6 +445,7 @@ function recallTurns(request) {
|
|
|
445
445
|
limits: { ...RECALL_LIMITS, maxTurnRows, maxTurnBytes, maxReadBytes, limit },
|
|
446
446
|
...indexStats(indexMeta),
|
|
447
447
|
};
|
|
448
|
+
if (limit === 0) return { turns: [], meta: { ...meta, status: 'disabled', reason: 'disabled' } };
|
|
448
449
|
const sourceBefore = currentSourceToken();
|
|
449
450
|
if (sourceBefore.fingerprint !== indexMeta.raw_source_fingerprint) {
|
|
450
451
|
return { turns: [], meta: { ...meta, status: 'not_ready', reason: 'stale_result' } };
|
|
@@ -2,7 +2,7 @@ import { Worker } from 'node:worker_threads';
|
|
|
2
2
|
import { existsSync, mkdirSync, readFileSync, rmSync } from 'node:fs';
|
|
3
3
|
import { dirname } from 'node:path';
|
|
4
4
|
import { writeAtomic } from '../storage/atomic.js';
|
|
5
|
-
import { extractRecallTerms, scoreRecallTurn } from './recall-relevance.js';
|
|
5
|
+
import { extractRecallTerms, scoreRecallTurn, normalizeRecallLimit } from './recall-relevance.js';
|
|
6
6
|
import {
|
|
7
7
|
conversationIndexDatabasePath,
|
|
8
8
|
conversationIndexManifestPath,
|
|
@@ -556,8 +556,8 @@ export async function searchConversationIndex(ownerRoot, sessionId, query, opts
|
|
|
556
556
|
/**
|
|
557
557
|
* Recall complete visible user/assistant turns from this owner-root Session.
|
|
558
558
|
* beforeSeq excludes that user turn and all later rows (exclusive seq fence).
|
|
559
|
-
* limit defaults to
|
|
560
|
-
* only lower hard worker caps. Canonical message IDs identify entries; their
|
|
559
|
+
* limit defaults to 5, capped at 5; 0 disables recall.
|
|
560
|
+
* maxTurnRows/maxTurnBytes/maxReadBytes may only lower hard worker caps. Canonical message IDs identify entries; their
|
|
561
561
|
* sourceMessageIds preserve all persisted identities, including aggregated VP
|
|
562
562
|
* replies. The first cold call yields not_ready and starts a background build;
|
|
563
563
|
* later calls wait up to 1500ms for readiness and the worker read, without
|
|
@@ -569,6 +569,8 @@ export async function searchConversationIndex(ownerRoot, sessionId, query, opts
|
|
|
569
569
|
*/
|
|
570
570
|
export async function recallConversationTurns(ownerRoot, sessionId, prompt, opts = {}) {
|
|
571
571
|
const terms = extractRecallTerms(prompt);
|
|
572
|
+
const limit = normalizeRecallLimit(opts.limit);
|
|
573
|
+
if (limit === 0) return { turns: [], meta: { status: 'disabled', reason: 'disabled', terms } };
|
|
572
574
|
const eligibility = scoreRecallTurn(terms, terms.join(' '));
|
|
573
575
|
if (!eligibility.score) {
|
|
574
576
|
return { turns: [], meta: { status: 'ready', reason: eligibility.reason, terms } };
|
|
@@ -577,7 +579,7 @@ export async function recallConversationTurns(ownerRoot, sessionId, prompt, opts
|
|
|
577
579
|
return await requestConversationHistoryIndex(ownerRoot, sessionId, 'recall-turns', {
|
|
578
580
|
prompt: typeof prompt === 'string' ? prompt.slice(0, 4096) : '',
|
|
579
581
|
beforeSeq: opts.beforeSeq,
|
|
580
|
-
limit
|
|
582
|
+
limit,
|
|
581
583
|
maxTurnRows: opts.maxTurnRows,
|
|
582
584
|
maxTurnBytes: opts.maxTurnBytes,
|
|
583
585
|
maxReadBytes: opts.maxReadBytes,
|
|
@@ -1564,6 +1564,43 @@ export class ConversationStore {
|
|
|
1564
1564
|
return pairSanitize(messages);
|
|
1565
1565
|
}
|
|
1566
1566
|
|
|
1567
|
+
/**
|
|
1568
|
+
* Provider-only chronological history. UI pages deliberately cap raw rows;
|
|
1569
|
+
* that cap is not a turn limit and must not truncate a tool-heavy query's
|
|
1570
|
+
* context. Yield between bounded scan batches; never mutate the transcript.
|
|
1571
|
+
* Stable user identities, not equal prompt text, define human turns.
|
|
1572
|
+
* @returns {Promise<object[]>} Complete past turns before the durable user fence.
|
|
1573
|
+
*/
|
|
1574
|
+
async loadProviderHistoryBySession(sessionId, turnsLimit = 20, { beforeSeq = Infinity } = {}) {
|
|
1575
|
+
if (!sessionId || !(turnsLimit > 0)) return [];
|
|
1576
|
+
const kept = [];
|
|
1577
|
+
const pending = [];
|
|
1578
|
+
const identities = new Set();
|
|
1579
|
+
let scanned = 0;
|
|
1580
|
+
let bytes = 0;
|
|
1581
|
+
for (const row of this.#iterateSessionRows(sessionId, { beforeSeq, desc: true })) {
|
|
1582
|
+
scanned += 1;
|
|
1583
|
+
bytes += Buffer.byteLength(JSON.stringify(row));
|
|
1584
|
+
if (scanned > 32768 || bytes > 64 * 1024 * 1024) {
|
|
1585
|
+
const error = new Error('Recent history scan limit reached before completing the requested turn window');
|
|
1586
|
+
error.code = 'HISTORY_RECENT_SCAN_LIMIT';
|
|
1587
|
+
throw error;
|
|
1588
|
+
}
|
|
1589
|
+
if (scanned % 64 === 0) await new Promise(resolve => setImmediate(resolve));
|
|
1590
|
+
if (!row || row.sessionId !== sessionId || isHiddenConversationRow(row)) continue;
|
|
1591
|
+
if (row.role === 'user') {
|
|
1592
|
+
const identity = row.clientMessageId ? `client:${row.clientMessageId}` : `message:${row.id}`;
|
|
1593
|
+
identities.add(identity);
|
|
1594
|
+
kept.push(...pending.splice(0), row);
|
|
1595
|
+
// The requested oldest user closes the reverse scan. Do not read the
|
|
1596
|
+
// previous turn's tool tail just to discover one more user boundary.
|
|
1597
|
+
if (identities.size >= turnsLimit) break;
|
|
1598
|
+
} else pending.push(row);
|
|
1599
|
+
}
|
|
1600
|
+
// Pending rows without their opening user are not a complete past turn.
|
|
1601
|
+
return pairSanitize(kept.reverse());
|
|
1602
|
+
}
|
|
1603
|
+
|
|
1567
1604
|
/**
|
|
1568
1605
|
* Load every hot message stamped with `sessionId`.
|
|
1569
1606
|
*
|
|
@@ -4,7 +4,35 @@ const MAX_PROMPT_CHARS = 4096;
|
|
|
4
4
|
const MAX_TERMS = 8;
|
|
5
5
|
const STOP_WORDS = new Set(`a an and are as at be been but by can could did do does for from had has have how i if in is it its me my of on or our please should so that the their them there these they this to was we were what when where which who why will with would you your about again before earlier previous remember recall history message messages conversation turn turns tell show find help need want use using make get know explain answer question code file project problem fix work task test tests implementation implement change thanks continue revisit discuss discussed discussion decide decided follow-up
|
|
6
6
|
的 了 是 在 和 与 或 我 你 他 她 它 我们 你们 他们 这个 那个 什么 怎么 如何 为什么 请 请问 帮 帮我 帮忙 可以 能 不能 是否 需要 想 要 再 还 又 也 就 都 把 将 给 对 从 到 上 下 中 里 有 没有 一下 一些 一个 这些 那些 之前 以前 上次 刚才 历史 记得 回忆 召回 消息 对话 问题 回答 内容 事情 继续 现在 今天 昨天 后来 然后 相关 具体 代码 文件 项目 实现 修改 功能 测试 方案 方法 工作 任务 谢谢 好的 好 看看 查找 搜索 查询 处理 解决 进行 使用 讨论 提到 记忆 总结 回顾 提醒`.split(/\s+/u));
|
|
7
|
+
for (const word of `system service config configuration settings build run error errors issue issues status result results request requests response responses check checks update updates version versions default option options limit limits page data user users model models tool tools server client input output changes details new old
|
|
8
|
+
系统 服务 配置 设置 构建 运行 错误 报错 状态 结果 请求 响应 检查 更新 版本 默认 选项 限制 页面 数据 用户 模型 工具 服务端 客户端 输入 输出 改动 详细 新 旧`.split(/\s+/u)) STOP_WORDS.add(word);
|
|
7
9
|
const segmenter = new Intl.Segmenter('zh', { granularity: 'word' });
|
|
10
|
+
const MIN_COVERAGE = 0.6;
|
|
11
|
+
const MIN_DISTINCTIVENESS = 0.5;
|
|
12
|
+
const MIN_SCORE = 6.5;
|
|
13
|
+
|
|
14
|
+
// Paths and concrete issue IDs are strong anchors even when a bounded sample
|
|
15
|
+
// contains many discussions of that exact reference. Mere camel/snake casing
|
|
16
|
+
// does not make an otherwise common word distinctive.
|
|
17
|
+
function isExactReference(term) {
|
|
18
|
+
return term.includes('/')
|
|
19
|
+
|| /\.[a-z][a-z0-9]{0,7}$/iu.test(term)
|
|
20
|
+
|| /^[a-z][a-z0-9]*(?:[-_:][a-z0-9]+)*[-_:]\d[\da-z_-]*$/iu.test(term)
|
|
21
|
+
|| /^(?:[a-z0-9]+[-_:])*[a-f0-9]{8,}$/iu.test(term);
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
function isGenericTerm(term) {
|
|
25
|
+
if (STOP_WORDS.has(term.toLocaleLowerCase())) return true;
|
|
26
|
+
const words = term.replace(/([a-z])([A-Z])/gu, '$1 $2').toLocaleLowerCase().split(/[\s_.:@-]+/u);
|
|
27
|
+
return words.every(word => STOP_WORDS.has(word));
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/** Automatic recall accepts zero and never exceeds five, even for raw callers. */
|
|
31
|
+
export function normalizeRecallLimit(value) {
|
|
32
|
+
const number = typeof value === 'number' || typeof value === 'string' && value.trim()
|
|
33
|
+
? Number(value) : NaN;
|
|
34
|
+
return Number.isFinite(number) ? Math.min(5, Math.max(0, Math.floor(number))) : 5;
|
|
35
|
+
}
|
|
8
36
|
|
|
9
37
|
function isIdentifier(term) {
|
|
10
38
|
return /[\p{L}\d][_.:/@-][\p{L}\d]/u.test(term)
|
|
@@ -20,7 +48,7 @@ export function extractRecallTerms(prompt) {
|
|
|
20
48
|
? prompt.slice(0, MAX_PROMPT_CHARS).replace(/@vp-[A-Za-z0-9_-]+\b/gu, ' ') : '';
|
|
21
49
|
const tokens = [];
|
|
22
50
|
// Keep paths, issue IDs, snake_case and camelCase intact before segmentation.
|
|
23
|
-
const rest = text.replace(/[\p{L}\p{N}]+(?:[_.:/@-][\p{L}\p{N}]+)+|[A-Za-z][A-Za-z0-9]*/gu, token => {
|
|
51
|
+
const rest = text.replace(/(?:\.{0,2}\/)?[\p{L}\p{N}]+(?:[_.:/@-][\p{L}\p{N}]+)+|[A-Za-z][A-Za-z0-9]*/gu, token => {
|
|
24
52
|
tokens.push(token);
|
|
25
53
|
return ' ';
|
|
26
54
|
});
|
|
@@ -44,7 +72,7 @@ export function extractRecallTerms(prompt) {
|
|
|
44
72
|
const seen = new Set();
|
|
45
73
|
const terms = tokens.filter(term => {
|
|
46
74
|
const key = term.toLocaleLowerCase();
|
|
47
|
-
if (seen.has(key) ||
|
|
75
|
+
if (seen.has(key) || isGenericTerm(term) || /^\d+$/u.test(key)
|
|
48
76
|
|| Array.from(key).length < 2 || key.length > 96) return false;
|
|
49
77
|
seen.add(key);
|
|
50
78
|
return true;
|
|
@@ -59,17 +87,21 @@ export function extractRecallTerms(prompt) {
|
|
|
59
87
|
* Returns explainable rejection reasons rather than weak positive matches.
|
|
60
88
|
*/
|
|
61
89
|
export function scoreRecallTurn(promptOrTerms, text, stats = {}) {
|
|
62
|
-
const terms = (Array.isArray(promptOrTerms) ? promptOrTerms : extractRecallTerms(promptOrTerms))
|
|
90
|
+
const terms = (Array.isArray(promptOrTerms) ? promptOrTerms : extractRecallTerms(promptOrTerms))
|
|
91
|
+
.filter(term => typeof term === 'string' && !isGenericTerm(term)).slice(0, MAX_TERMS);
|
|
63
92
|
const body = String(text || '').toLocaleLowerCase();
|
|
64
93
|
const matchedTerms = terms.filter(term => {
|
|
65
94
|
const key = term.toLocaleLowerCase();
|
|
66
|
-
if (/^[a-z0-9_]+$/u.test(key)) {
|
|
95
|
+
if (/^[a-z0-9_]+$/u.test(key) || isIdentifier(term)) {
|
|
67
96
|
const escaped = key.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
68
|
-
|
|
97
|
+
const boundary = isIdentifier(term) ? 'a-z0-9_.:/@-' : 'a-z0-9_';
|
|
98
|
+
const suffix = isIdentifier(term) ? '(?![a-z0-9_/@-]|[.:][a-z0-9_])' : `(?![${boundary}])`;
|
|
99
|
+
return new RegExp(`(?<![${boundary}])${escaped}${suffix}`, 'u').test(body);
|
|
69
100
|
}
|
|
70
101
|
return body.includes(key);
|
|
71
102
|
});
|
|
72
103
|
const identifierMatches = matchedTerms.filter(isIdentifier);
|
|
104
|
+
const exactReferenceMatches = matchedTerms.filter(isExactReference);
|
|
73
105
|
const independentTerms = matchedTerms.filter(term => !matchedTerms.some(other => (
|
|
74
106
|
other !== term && other.toLocaleLowerCase().includes(term.toLocaleLowerCase())
|
|
75
107
|
)));
|
|
@@ -79,16 +111,17 @@ export function scoreRecallTurn(promptOrTerms, text, stats = {}) {
|
|
|
79
111
|
const frequency = stats.termDocumentFrequency?.[term.toLocaleLowerCase()];
|
|
80
112
|
return sampleSize >= 8 && Number.isFinite(frequency) ? 1 - frequency / sampleSize : 1;
|
|
81
113
|
})) : 0;
|
|
114
|
+
const rawScore = Math.round((independentTerms.length * 2 + identifierMatches.length * 4
|
|
115
|
+
+ coverage * 2 + distinctiveness) * 100) / 100;
|
|
82
116
|
let reason = 'relevant';
|
|
83
117
|
if (!terms.length) reason = 'generic_prompt';
|
|
84
118
|
else if (!matchedTerms.length) reason = 'no_match';
|
|
85
119
|
else if (!identifierMatches.length && independentTerms.length < 2) reason = 'insufficient_keywords';
|
|
86
|
-
else if (
|
|
87
|
-
else if (!
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
return { score, matchedTerms, reason, coverage, distinctiveness, identifierMatches, sampleSize };
|
|
120
|
+
else if (coverage < MIN_COVERAGE) reason = 'low_coverage';
|
|
121
|
+
else if (!exactReferenceMatches.length && distinctiveness < MIN_DISTINCTIVENESS) reason = 'low_distinctiveness';
|
|
122
|
+
else if (rawScore < MIN_SCORE) reason = 'low_score';
|
|
123
|
+
const score = reason === 'relevant' ? rawScore : 0;
|
|
124
|
+
return { score, matchedTerms, reason, coverage, distinctiveness, identifierMatches, exactReferenceMatches, sampleSize };
|
|
92
125
|
}
|
|
93
126
|
|
|
94
127
|
export const RECALL_LIMITS = Object.freeze({
|
package/yeaft/engine.js
CHANGED
|
@@ -46,7 +46,7 @@ import { perfNowMs, recordAgentPerfTrace } from './perf-trace.js';
|
|
|
46
46
|
const MAIN_THREAD_ID = 'main';
|
|
47
47
|
import { pickEffort, parseEffortPrefix, snapshotEffortDecision } from './effort.js';
|
|
48
48
|
import { bindProviderState } from './llm/provider-state.js';
|
|
49
|
-
import { DEFAULT_CONTEXT_WINDOW, normalizeEffort, resolveContextWindow, resolveModel } from './models.js';
|
|
49
|
+
import { DEFAULT_CONTEXT_WINDOW, getModelInfo, normalizeEffort, parseModelRef, resolveContextWindow, resolveMaxOutputTokens, resolveModel } from './models.js';
|
|
50
50
|
import { lookupModelLimitSync } from './llm/models-dev.js';
|
|
51
51
|
import { attachRouterPlan, extractPriorPlan, stripMetaForWire } from './router/continuity.js';
|
|
52
52
|
import { resolveThinking } from './router/thinking.js';
|
|
@@ -1882,7 +1882,7 @@ export class Engine {
|
|
|
1882
1882
|
}
|
|
1883
1883
|
}
|
|
1884
1884
|
|
|
1885
|
-
async *#queryLifecycle({ prompt, promptParts = null, messages = [], signal, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null } = {}) {
|
|
1885
|
+
async *#queryLifecycle({ prompt, promptParts = null, messages = [], signal, turnConfig = null, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null } = {}) {
|
|
1886
1886
|
if (!prompt || typeof prompt !== 'string' || !prompt.trim()) {
|
|
1887
1887
|
const error = new Error('prompt is required and must be a non-empty string');
|
|
1888
1888
|
yield {
|
|
@@ -1974,7 +1974,7 @@ export class Engine {
|
|
|
1974
1974
|
try {
|
|
1975
1975
|
this.#currentThreadId = threadId || MAIN_THREAD_ID;
|
|
1976
1976
|
this.#currentCausalRootId = effectiveCausalRootId;
|
|
1977
|
-
yield* this.#runQuery({ prompt: effectivePrompt, promptParts: effectivePromptParts, messages, signal: runSignal, userEffort: explicitUserEffort, scenario, isSubAgent, parentEffortDecision, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds, projectInstruction, projectLabel, vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted, currentUserMessage, causalRootId: effectiveCausalRootId, getCurrentTodos, setCurrentTodos, askUser, threadId: this.#currentThreadId, vpTurnId, drainPendingUserMessages, prepareProviderRequest, startProviderRequest, finishProviderRequest, failProviderRequest, closePendingUserInput, collabToolPolicy: effectiveCollabToolPolicy, explicitSkillName: parsedSkill.skillName, retryLifecycle });
|
|
1977
|
+
yield* this.#runQuery({ prompt: effectivePrompt, promptParts: effectivePromptParts, messages, signal: runSignal, turnConfig: turnConfig ? { model: turnConfig.model, effort: turnConfig.effort, maxOutputTokens: turnConfig.maxOutputTokens } : null, userEffort: explicitUserEffort, scenario, isSubAgent, parentEffortDecision, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds, projectInstruction, projectLabel, vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted, currentUserMessage, causalRootId: effectiveCausalRootId, getCurrentTodos, setCurrentTodos, askUser, threadId: this.#currentThreadId, vpTurnId, drainPendingUserMessages, prepareProviderRequest, startProviderRequest, finishProviderRequest, failProviderRequest, closePendingUserInput, collabToolPolicy: effectiveCollabToolPolicy, explicitSkillName: parsedSkill.skillName, retryLifecycle });
|
|
1978
1978
|
} finally {
|
|
1979
1979
|
// Closing the async generator at a visible retry boundary means the
|
|
1980
1980
|
// continuation never reached a provider. Keep it out of history and
|
|
@@ -2029,7 +2029,7 @@ export class Engine {
|
|
|
2029
2029
|
* in a try/finally without indenting the whole loop.
|
|
2030
2030
|
* @private
|
|
2031
2031
|
*/
|
|
2032
|
-
async *#runQuery({ prompt, promptParts = null, messages, signal, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null, explicitSkillName = null, retryLifecycle }) {
|
|
2032
|
+
async *#runQuery({ prompt, promptParts = null, messages, signal, turnConfig = null, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null, explicitSkillName = null, retryLifecycle }) {
|
|
2033
2033
|
|
|
2034
2034
|
const effectiveCollabToolPolicy = collabToolPolicy === COLLAB_TOOL_POLICY.SINGLE_VP || collabToolPolicy === COLLAB_TOOL_POLICY.MULTI_VP
|
|
2035
2035
|
? collabToolPolicy
|
|
@@ -2103,13 +2103,18 @@ export class Engine {
|
|
|
2103
2103
|
...(internalTrigger ? { internal: true } : { userAuthored: true }),
|
|
2104
2104
|
}, { sessionId: runtimeSessionId });
|
|
2105
2105
|
}
|
|
2106
|
-
// Native Session history, including internal wakeups,
|
|
2107
|
-
//
|
|
2106
|
+
// Native Session history, including internal wakeups, comes from the
|
|
2107
|
+
// canonical message transcript. Dream memory loading is temporarily
|
|
2108
|
+
// disabled for every scenario (including Work Center and child agents)
|
|
2109
|
+
// while message-history recall is evaluated as its replacement. Keep the
|
|
2110
|
+
// downstream memory pipeline intact behind this single switch so it can be
|
|
2111
|
+
// restored without migrating or deleting persisted memory data.
|
|
2108
2112
|
const useMessageHistory = scenario !== 'work-item' && !!runtimeSessionId && !vpPersona?.subAgent;
|
|
2109
|
-
const useDreamMemory = scenario === 'work-item' || !!vpPersona?.subAgent
|
|
2110
|
-
|
|
2111
|
-
const
|
|
2112
|
-
const
|
|
2113
|
+
// const useDreamMemory = scenario === 'work-item' || !!vpPersona?.subAgent
|
|
2114
|
+
// || (!runtimeSessionId && !internalTrigger);
|
|
2115
|
+
const useDreamMemory = false;
|
|
2116
|
+
const recentTurnCap = Math.max(20, this.#config.yeaft?.recentTurnsLimit ?? 20);
|
|
2117
|
+
const relatedTurnCap = Math.min(5, this.#config.yeaft?.relatedTurnsLimit ?? 5);
|
|
2113
2118
|
let relatedHistoryTurns = [];
|
|
2114
2119
|
let historyRecallMeta = { source: 'messages', status: 'disabled' };
|
|
2115
2120
|
if (useMessageHistory && !internalTrigger && this.#conversationStore?.loadRecentBySession) {
|
|
@@ -2125,7 +2130,9 @@ export class Engine {
|
|
|
2125
2130
|
const beforeSeq = Number.isFinite(persistedQueryUser?.seq)
|
|
2126
2131
|
? persistedQueryUser.seq : parseSeqFromId(persistedQueryUser?.id);
|
|
2127
2132
|
if (Number.isFinite(beforeSeq)) {
|
|
2128
|
-
const
|
|
2133
|
+
const loadHistory = this.#conversationStore.loadProviderHistoryBySession
|
|
2134
|
+
|| this.#conversationStore.loadRecentBySession;
|
|
2135
|
+
const tail = await loadHistory.call(this.#conversationStore, runtimeSessionId, recentTurnCap, { beforeSeq });
|
|
2129
2136
|
messages = tail.filter(m => parseSeqFromId(m.id) < beforeSeq
|
|
2130
2137
|
&& (m.role !== 'tool' || !queryVpId || m.speakerVpId === queryVpId))
|
|
2131
2138
|
.map(m => {
|
|
@@ -2550,7 +2557,7 @@ export class Engine {
|
|
|
2550
2557
|
// `refreshConfig()` may publish a new Session model while a stream or a
|
|
2551
2558
|
// tool is running. Apply it only before the next provider request; the
|
|
2552
2559
|
// current request keeps the snapshot captured below.
|
|
2553
|
-
let currentModel = this.#config.model;
|
|
2560
|
+
let currentModel = turnConfig?.model || this.#config.model;
|
|
2554
2561
|
let primaryModelAtLastBoundary = currentModel;
|
|
2555
2562
|
let cumulativeInputTokens = 0;
|
|
2556
2563
|
let cumulativeOutputTokens = 0;
|
|
@@ -2595,7 +2602,7 @@ export class Engine {
|
|
|
2595
2602
|
// Keep a retry fallback selected by this query; replacing it here would
|
|
2596
2603
|
// turn an exhausted primary into an endless retry loop.
|
|
2597
2604
|
if (currentModel === primaryModelAtLastBoundary) {
|
|
2598
|
-
const refreshedPrimaryModel = this.#config.model;
|
|
2605
|
+
const refreshedPrimaryModel = turnConfig?.model || this.#config.model;
|
|
2599
2606
|
if (refreshedPrimaryModel !== primaryModelAtLastBoundary) {
|
|
2600
2607
|
currentModel = refreshedPrimaryModel;
|
|
2601
2608
|
primaryModelAtLastBoundary = refreshedPrimaryModel;
|
|
@@ -2607,6 +2614,21 @@ export class Engine {
|
|
|
2607
2614
|
// this request. Fallback retries intentionally retain their selected
|
|
2608
2615
|
// model, but still use the current policy and configured effort.
|
|
2609
2616
|
const requestConfig = { ...this.#config };
|
|
2617
|
+
// Overlay only the request snapshot. Never publish temporary settings via
|
|
2618
|
+
// refreshConfig or mutate the shared config / AdapterRouter catalog.
|
|
2619
|
+
if (turnConfig) {
|
|
2620
|
+
requestConfig.model = currentModel;
|
|
2621
|
+
const entry = requestConfig.availableModels?.find(model => model.ref === currentModel);
|
|
2622
|
+
requestConfig.modelInfo = getModelInfo(parseModelRef(currentModel).modelId, entry) || null;
|
|
2623
|
+
if (turnConfig.effort != null) requestConfig.modelEffort = turnConfig.effort;
|
|
2624
|
+
// Blank means the selected model's default cap, not the Session's
|
|
2625
|
+
// previous model budget. Re-resolve at every boundary (including fallback
|
|
2626
|
+
// retries / catalog refresh), without leaking the old global ceiling.
|
|
2627
|
+
const outputLimit = resolveMaxOutputTokens(parseModelRef(currentModel).modelId, {
|
|
2628
|
+
modelInfo: requestConfig.modelInfo,
|
|
2629
|
+
});
|
|
2630
|
+
requestConfig.maxOutputTokens = Math.min(turnConfig.maxOutputTokens ?? outputLimit, outputLimit);
|
|
2631
|
+
}
|
|
2610
2632
|
// Capture the matching provider catalog in the same synchronous boundary
|
|
2611
2633
|
// as config/model. Preflight may yield user/task events before the stream
|
|
2612
2634
|
// is built, but one request must never mix two refresh revisions.
|
|
@@ -2803,7 +2825,7 @@ export class Engine {
|
|
|
2803
2825
|
// effect at the next loop. A caller override or `/effort` prefix stays
|
|
2804
2826
|
// fixed for this query and still wins over live Session config.
|
|
2805
2827
|
const configuredEffort = normalizeEffort(requestConfig.modelEffort);
|
|
2806
|
-
const requestUserEffort = userEffort || configuredEffort || null;
|
|
2828
|
+
const requestUserEffort = normalizeEffort(turnConfig?.effort) || userEffort || configuredEffort || null;
|
|
2807
2829
|
let resolvedEffort = pickEffort({ scenario, toolLoopTurns, userEffort: requestUserEffort });
|
|
2808
2830
|
|
|
2809
2831
|
// DESIGN.md §9.16: thinking-mode precedence chain. When a VP
|
|
@@ -2864,8 +2886,8 @@ export class Engine {
|
|
|
2864
2886
|
const buckets = useMessageHistory ? buildHistoryBuckets(conversationMessages, {
|
|
2865
2887
|
prompt,
|
|
2866
2888
|
relatedTurns: relatedHistoryTurns,
|
|
2867
|
-
recentTurnCap: requestConfig.yeaft?.recentTurnsLimit ?? 20,
|
|
2868
|
-
relatedTurnCap: requestConfig.yeaft?.relatedTurnsLimit ??
|
|
2889
|
+
recentTurnCap: Math.max(20, requestConfig.yeaft?.recentTurnsLimit ?? 20),
|
|
2890
|
+
relatedTurnCap: Math.min(5, requestConfig.yeaft?.relatedTurnsLimit ?? 5),
|
|
2869
2891
|
messageTokenBudget: historyBudget,
|
|
2870
2892
|
currentTurnStartIndex: turnStartIdx,
|
|
2871
2893
|
language: requestConfig.language,
|