@yeaft/webchat-agent 1.0.508 → 1.0.510
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/connection/message-router.js +1 -1
- package/local-runtime/server/handlers/agent-sync.js +1 -0
- package/local-runtime/server/handlers/client-conversation.js +38 -2
- package/local-runtime/server/handlers/client-misc.js +1 -1
- package/local-runtime/version.json +1 -1
- package/local-runtime/web/app.bundle.js +212 -99
- package/local-runtime/web/app.bundle.js.gz +0 -0
- package/local-runtime/web/index.html +2 -2
- package/local-runtime/web/style.bundle.css +1 -1
- package/local-runtime/web/style.bundle.css.gz +0 -0
- package/package.json +1 -1
- package/yeaft/config-api.js +59 -6
- package/yeaft/config.js +3 -3
- package/yeaft/conversation/history-index-worker.js +3 -2
- package/yeaft/conversation/history-index.js +6 -4
- package/yeaft/conversation/persist.js +37 -0
- package/yeaft/conversation/recall-relevance.js +44 -11
- package/yeaft/engine.js +262 -56
- package/yeaft/history-window.js +128 -122
- package/yeaft/post-compact.js +76 -0
- package/yeaft/tool-folding/index.js +9 -15
- package/yeaft/tool-folding/t1-reflector.js +3 -2
- package/yeaft/web-bridge.js +130 -39
|
Binary file
|
package/package.json
CHANGED
package/yeaft/config-api.js
CHANGED
|
@@ -11,12 +11,60 @@
|
|
|
11
11
|
import { existsSync, readFileSync } from 'fs';
|
|
12
12
|
import { join } from 'path';
|
|
13
13
|
import { DEFAULT_YEAFT_DIR } from './init.js';
|
|
14
|
-
import { normalizeProviderModels, parseModelRef, serializeModelForPersistence } from './models.js';
|
|
14
|
+
import { getModelEffortOptions, normalizeEffort, normalizeProviderModels, parseModelRef, resolveMaxOutputTokens, serializeModelForPersistence } from './models.js';
|
|
15
|
+
import { inferProtocolFromModelId } from './llm/router.js';
|
|
15
16
|
import { clampYeaftField, normaliseTelemetrySection, normaliseYeaftSection } from './config.js';
|
|
16
17
|
import { normaliseBrowserRuntimeSection, validateBrowserRuntimeUpdate } from '../browser-runtime/config.js';
|
|
17
18
|
import { normalizePluginConfig } from './plugins.js';
|
|
18
19
|
import { mutateAgentConfig, readAgentConfigForWrite } from './config-store.js';
|
|
19
|
-
import { isGitHubCopilotProvider, serializeKnownProviderForPersistence } from './llm/known-providers.js';
|
|
20
|
+
import { isGitHubCopilotProvider, normalizeKnownProviderForRuntime, serializeKnownProviderForPersistence } from './llm/known-providers.js';
|
|
21
|
+
|
|
22
|
+
/** Agent-owned model catalog, with the same protocol/capability resolution as runtime. */
|
|
23
|
+
function quickSendModels(config) {
|
|
24
|
+
return (Array.isArray(config.providers) ? config.providers : []).flatMap(raw => {
|
|
25
|
+
const provider = normalizeKnownProviderForRuntime(raw);
|
|
26
|
+
return normalizeProviderModels(provider).map(model => ({
|
|
27
|
+
id: model.id,
|
|
28
|
+
ref: provider.name ? `${provider.name}/${model.id}` : model.id,
|
|
29
|
+
provider: provider.name,
|
|
30
|
+
label: model.id,
|
|
31
|
+
effortOptions: getModelEffortOptions(model.id, {
|
|
32
|
+
...model,
|
|
33
|
+
protocol: model.protocol || provider.protocol || inferProtocolFromModelId(model.id) || 'openai-responses',
|
|
34
|
+
}),
|
|
35
|
+
maxOutput: resolveMaxOutputTokens(model.id, { modelInfo: model }),
|
|
36
|
+
}));
|
|
37
|
+
});
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/** Validate inside mutateAgentConfig's lock, against the resulting provider catalog. */
|
|
41
|
+
function validateQuickSends(value, config) {
|
|
42
|
+
if (!Array.isArray(value) || value.length > 5) throw new Error('quickSends must be an array of at most 5 items');
|
|
43
|
+
const models = quickSendModels(config);
|
|
44
|
+
const ids = new Set();
|
|
45
|
+
return value.map(item => {
|
|
46
|
+
if (!item || typeof item !== 'object' || Array.isArray(item)) throw new Error('Each quick send must be an object');
|
|
47
|
+
const id = typeof item.id === 'string' ? item.id.trim() : '';
|
|
48
|
+
const name = typeof item.name === 'string' ? item.name.trim() : '';
|
|
49
|
+
if (!id || id.length > 128 || ids.has(id)) throw new Error('Quick send id must be unique and 1–128 characters');
|
|
50
|
+
if (!name || name.length > 80) throw new Error('Quick send name must be 1–80 characters');
|
|
51
|
+
ids.add(id);
|
|
52
|
+
const ref = typeof item.model === 'string' ? item.model.trim() : '';
|
|
53
|
+
const matches = models.filter(model => model.ref === ref);
|
|
54
|
+
const candidates = matches.length ? matches : models.filter(model => model.id === ref);
|
|
55
|
+
if (candidates.length !== 1) throw new Error(`Quick send model must exist and be unambiguous: ${ref}`);
|
|
56
|
+
const model = candidates[0];
|
|
57
|
+
const effort = item.effort ?? null;
|
|
58
|
+
if (effort !== null && (!normalizeEffort(effort) || !model.effortOptions.includes(effort))) {
|
|
59
|
+
throw new Error(`Quick send effort is not available for ${model.ref}`);
|
|
60
|
+
}
|
|
61
|
+
const maxOutputTokens = item.maxOutputTokens ?? null;
|
|
62
|
+
if (maxOutputTokens !== null && (!Number.isSafeInteger(maxOutputTokens) || maxOutputTokens <= 0 || maxOutputTokens > model.maxOutput)) {
|
|
63
|
+
throw new Error(`Quick send maxOutputTokens must be a positive integer no greater than ${model.maxOutput}`);
|
|
64
|
+
}
|
|
65
|
+
return { id, name, model: model.ref, effort, maxOutputTokens };
|
|
66
|
+
});
|
|
67
|
+
}
|
|
20
68
|
|
|
21
69
|
/**
|
|
22
70
|
* Read config.json before any public mutation. A missing file is a valid
|
|
@@ -44,7 +92,7 @@ function readLocalLlmConfig(dir) {
|
|
|
44
92
|
const configPath = join(root, 'config.json');
|
|
45
93
|
|
|
46
94
|
if (!existsSync(configPath)) {
|
|
47
|
-
return { providers: [], primaryModel: null, fastModel: null, language: 'en', needsSetup: true };
|
|
95
|
+
return { providers: [], primaryModel: null, fastModel: null, language: 'en', quickSends: [], availableModels: [], needsSetup: true };
|
|
48
96
|
}
|
|
49
97
|
|
|
50
98
|
const raw = readFileSync(configPath, 'utf8');
|
|
@@ -56,6 +104,8 @@ function readLocalLlmConfig(dir) {
|
|
|
56
104
|
fastModel: json.fastModel || null,
|
|
57
105
|
language: json.language || 'en',
|
|
58
106
|
debug: json.debug === true,
|
|
107
|
+
quickSends: Array.isArray(json.quickSends) ? json.quickSends : [],
|
|
108
|
+
availableModels: quickSendModels(json),
|
|
59
109
|
needsSetup: providers.length === 0 || providers.every(p => p.apiKey === 'proxy' || p.apiKey === '' || (!p.apiKey && !p.credentialProvider)),
|
|
60
110
|
};
|
|
61
111
|
}
|
|
@@ -169,6 +219,7 @@ export function updateLlmConfig(update, dir) {
|
|
|
169
219
|
if (update.providers !== undefined) normalizeManagedModelDefaults(existing);
|
|
170
220
|
if (update.language !== undefined) existing.language = update.language;
|
|
171
221
|
if (update.debug !== undefined) existing.debug = update.debug === true;
|
|
222
|
+
if (update.quickSends !== undefined) existing.quickSends = validateQuickSends(update.quickSends, existing);
|
|
172
223
|
|
|
173
224
|
const agentConfig = {
|
|
174
225
|
providers: Array.isArray(existing.providers) ? existing.providers : [],
|
|
@@ -176,6 +227,8 @@ export function updateLlmConfig(update, dir) {
|
|
|
176
227
|
fastModel: existing.fastModel || null,
|
|
177
228
|
language: existing.language || 'en',
|
|
178
229
|
debug: existing.debug === true,
|
|
230
|
+
quickSends: Array.isArray(existing.quickSends) ? existing.quickSends : [],
|
|
231
|
+
availableModels: quickSendModels(existing),
|
|
179
232
|
};
|
|
180
233
|
return {
|
|
181
234
|
...agentConfig,
|
|
@@ -216,7 +269,7 @@ export function getYeaftSettings(dir) {
|
|
|
216
269
|
* (LLM provider / model fields are untouched) and validates each field:
|
|
217
270
|
* `maxConcurrentThreads` must be 1..50, `autoArchiveIdleDays` must be
|
|
218
271
|
* 1..3650, `recentTurnsLimit` must be 1..500, `relatedTurnsLimit` must be
|
|
219
|
-
* 0..
|
|
272
|
+
* 0..5 (0 disables related recall). Dream limits are read-only
|
|
220
273
|
* runtime defaults here; invalid values are rejected
|
|
221
274
|
* outright so the UI sees an error rather than silently reverting — a
|
|
222
275
|
* silent revert would make "I set it to 100 and nothing happened"
|
|
@@ -256,8 +309,8 @@ export function updateYeaftSettings(update, dir) {
|
|
|
256
309
|
if (update.relatedTurnsLimit !== undefined) {
|
|
257
310
|
const value = clampYeaftField(update.relatedTurnsLimit, 'relatedTurnsLimit');
|
|
258
311
|
const n = Number(update.relatedTurnsLimit);
|
|
259
|
-
if (value === null || n < 0 || n >
|
|
260
|
-
return { error: 'relatedTurnsLimit must be between 0 and
|
|
312
|
+
if (value === null || n < 0 || n > 5) {
|
|
313
|
+
return { error: 'relatedTurnsLimit must be between 0 and 5' };
|
|
261
314
|
}
|
|
262
315
|
}
|
|
263
316
|
|
package/yeaft/config.js
CHANGED
|
@@ -55,8 +55,8 @@ const DEFAULTS = {
|
|
|
55
55
|
// through history pagination/search. Range: 1–500.
|
|
56
56
|
yeaftRecentTurnsLimit: 20,
|
|
57
57
|
// Same-Session related Q&A turns, selected by deterministic full-text rules.
|
|
58
|
-
// Range: 0–
|
|
59
|
-
yeaftRelatedTurnsLimit:
|
|
58
|
+
// Range: 0–5; 0 disables related recall without changing recent history.
|
|
59
|
+
yeaftRelatedTurnsLimit: 5,
|
|
60
60
|
// CLAUDE.md / AGENTS.md project-doc cap, in bytes. Mirrors Codex's
|
|
61
61
|
// `project_doc_max_bytes`. 0 disables the feature (no project-doc
|
|
62
62
|
// block is injected). Hand-edited values are NOT clamped — we let
|
|
@@ -306,7 +306,7 @@ export function clampYeaftField(v, field) {
|
|
|
306
306
|
let hi;
|
|
307
307
|
if (field === 'maxConcurrentThreads') { lo = 1; hi = 50; }
|
|
308
308
|
else if (field === 'recentTurnsLimit') { lo = 1; hi = 500; }
|
|
309
|
-
else if (field === 'relatedTurnsLimit') { lo = 0; hi =
|
|
309
|
+
else if (field === 'relatedTurnsLimit') { lo = 0; hi = 5; }
|
|
310
310
|
else { lo = 1; hi = 3650; } // autoArchiveIdleDays
|
|
311
311
|
return Math.min(hi, Math.max(lo, Math.floor(n)));
|
|
312
312
|
}
|
|
@@ -10,7 +10,7 @@ import {
|
|
|
10
10
|
normalizeLiteralSearch,
|
|
11
11
|
} from './visible-entry.js';
|
|
12
12
|
import { fingerprintConversationSources } from './history-index-state.js';
|
|
13
|
-
import { extractRecallTerms, scoreRecallTurn, RECALL_LIMITS } from './recall-relevance.js';
|
|
13
|
+
import { extractRecallTerms, scoreRecallTurn, normalizeRecallLimit, RECALL_LIMITS } from './recall-relevance.js';
|
|
14
14
|
|
|
15
15
|
const INDEX_SCHEMA_VERSION = 2;
|
|
16
16
|
const SHORT_BLOOM_BYTES = 256;
|
|
@@ -433,7 +433,7 @@ function recallTurns(request) {
|
|
|
433
433
|
const generation = Number(indexMeta.generation) || 0;
|
|
434
434
|
const terms = extractRecallTerms(request.prompt);
|
|
435
435
|
const cap = (value, fallback) => Math.min(fallback, Math.max(1, Math.floor(Number(value) || fallback)));
|
|
436
|
-
const limit =
|
|
436
|
+
const limit = normalizeRecallLimit(request.limit);
|
|
437
437
|
const maxTurnRows = cap(request.maxTurnRows, RECALL_LIMITS.maxTurnRows);
|
|
438
438
|
const maxTurnBytes = cap(request.maxTurnBytes, RECALL_LIMITS.maxTurnBytes);
|
|
439
439
|
const maxReadBytes = cap(request.maxReadBytes, RECALL_LIMITS.maxReadBytes);
|
|
@@ -445,6 +445,7 @@ function recallTurns(request) {
|
|
|
445
445
|
limits: { ...RECALL_LIMITS, maxTurnRows, maxTurnBytes, maxReadBytes, limit },
|
|
446
446
|
...indexStats(indexMeta),
|
|
447
447
|
};
|
|
448
|
+
if (limit === 0) return { turns: [], meta: { ...meta, status: 'disabled', reason: 'disabled' } };
|
|
448
449
|
const sourceBefore = currentSourceToken();
|
|
449
450
|
if (sourceBefore.fingerprint !== indexMeta.raw_source_fingerprint) {
|
|
450
451
|
return { turns: [], meta: { ...meta, status: 'not_ready', reason: 'stale_result' } };
|
|
@@ -2,7 +2,7 @@ import { Worker } from 'node:worker_threads';
|
|
|
2
2
|
import { existsSync, mkdirSync, readFileSync, rmSync } from 'node:fs';
|
|
3
3
|
import { dirname } from 'node:path';
|
|
4
4
|
import { writeAtomic } from '../storage/atomic.js';
|
|
5
|
-
import { extractRecallTerms, scoreRecallTurn } from './recall-relevance.js';
|
|
5
|
+
import { extractRecallTerms, scoreRecallTurn, normalizeRecallLimit } from './recall-relevance.js';
|
|
6
6
|
import {
|
|
7
7
|
conversationIndexDatabasePath,
|
|
8
8
|
conversationIndexManifestPath,
|
|
@@ -556,8 +556,8 @@ export async function searchConversationIndex(ownerRoot, sessionId, query, opts
|
|
|
556
556
|
/**
|
|
557
557
|
* Recall complete visible user/assistant turns from this owner-root Session.
|
|
558
558
|
* beforeSeq excludes that user turn and all later rows (exclusive seq fence).
|
|
559
|
-
* limit defaults to
|
|
560
|
-
* only lower hard worker caps. Canonical message IDs identify entries; their
|
|
559
|
+
* limit defaults to 5, capped at 5; 0 disables recall.
|
|
560
|
+
* maxTurnRows/maxTurnBytes/maxReadBytes may only lower hard worker caps. Canonical message IDs identify entries; their
|
|
561
561
|
* sourceMessageIds preserve all persisted identities, including aggregated VP
|
|
562
562
|
* replies. The first cold call yields not_ready and starts a background build;
|
|
563
563
|
* later calls wait up to 1500ms for readiness and the worker read, without
|
|
@@ -569,6 +569,8 @@ export async function searchConversationIndex(ownerRoot, sessionId, query, opts
|
|
|
569
569
|
*/
|
|
570
570
|
export async function recallConversationTurns(ownerRoot, sessionId, prompt, opts = {}) {
|
|
571
571
|
const terms = extractRecallTerms(prompt);
|
|
572
|
+
const limit = normalizeRecallLimit(opts.limit);
|
|
573
|
+
if (limit === 0) return { turns: [], meta: { status: 'disabled', reason: 'disabled', terms } };
|
|
572
574
|
const eligibility = scoreRecallTurn(terms, terms.join(' '));
|
|
573
575
|
if (!eligibility.score) {
|
|
574
576
|
return { turns: [], meta: { status: 'ready', reason: eligibility.reason, terms } };
|
|
@@ -577,7 +579,7 @@ export async function recallConversationTurns(ownerRoot, sessionId, prompt, opts
|
|
|
577
579
|
return await requestConversationHistoryIndex(ownerRoot, sessionId, 'recall-turns', {
|
|
578
580
|
prompt: typeof prompt === 'string' ? prompt.slice(0, 4096) : '',
|
|
579
581
|
beforeSeq: opts.beforeSeq,
|
|
580
|
-
limit
|
|
582
|
+
limit,
|
|
581
583
|
maxTurnRows: opts.maxTurnRows,
|
|
582
584
|
maxTurnBytes: opts.maxTurnBytes,
|
|
583
585
|
maxReadBytes: opts.maxReadBytes,
|
|
@@ -1564,6 +1564,43 @@ export class ConversationStore {
|
|
|
1564
1564
|
return pairSanitize(messages);
|
|
1565
1565
|
}
|
|
1566
1566
|
|
|
1567
|
+
/**
|
|
1568
|
+
* Provider-only chronological history. UI pages deliberately cap raw rows;
|
|
1569
|
+
* that cap is not a turn limit and must not truncate a tool-heavy query's
|
|
1570
|
+
* context. Yield between bounded scan batches; never mutate the transcript.
|
|
1571
|
+
* Stable user identities, not equal prompt text, define human turns.
|
|
1572
|
+
* @returns {Promise<object[]>} Complete past turns before the durable user fence.
|
|
1573
|
+
*/
|
|
1574
|
+
async loadProviderHistoryBySession(sessionId, turnsLimit = 20, { beforeSeq = Infinity } = {}) {
|
|
1575
|
+
if (!sessionId || !(turnsLimit > 0)) return [];
|
|
1576
|
+
const kept = [];
|
|
1577
|
+
const pending = [];
|
|
1578
|
+
const identities = new Set();
|
|
1579
|
+
let scanned = 0;
|
|
1580
|
+
let bytes = 0;
|
|
1581
|
+
for (const row of this.#iterateSessionRows(sessionId, { beforeSeq, desc: true })) {
|
|
1582
|
+
scanned += 1;
|
|
1583
|
+
bytes += Buffer.byteLength(JSON.stringify(row));
|
|
1584
|
+
if (scanned > 32768 || bytes > 64 * 1024 * 1024) {
|
|
1585
|
+
const error = new Error('Recent history scan limit reached before completing the requested turn window');
|
|
1586
|
+
error.code = 'HISTORY_RECENT_SCAN_LIMIT';
|
|
1587
|
+
throw error;
|
|
1588
|
+
}
|
|
1589
|
+
if (scanned % 64 === 0) await new Promise(resolve => setImmediate(resolve));
|
|
1590
|
+
if (!row || row.sessionId !== sessionId || isHiddenConversationRow(row)) continue;
|
|
1591
|
+
if (row.role === 'user') {
|
|
1592
|
+
const identity = row.clientMessageId ? `client:${row.clientMessageId}` : `message:${row.id}`;
|
|
1593
|
+
identities.add(identity);
|
|
1594
|
+
kept.push(...pending.splice(0), row);
|
|
1595
|
+
// The requested oldest user closes the reverse scan. Do not read the
|
|
1596
|
+
// previous turn's tool tail just to discover one more user boundary.
|
|
1597
|
+
if (identities.size >= turnsLimit) break;
|
|
1598
|
+
} else pending.push(row);
|
|
1599
|
+
}
|
|
1600
|
+
// Pending rows without their opening user are not a complete past turn.
|
|
1601
|
+
return pairSanitize(kept.reverse());
|
|
1602
|
+
}
|
|
1603
|
+
|
|
1567
1604
|
/**
|
|
1568
1605
|
* Load every hot message stamped with `sessionId`.
|
|
1569
1606
|
*
|
|
@@ -4,7 +4,35 @@ const MAX_PROMPT_CHARS = 4096;
|
|
|
4
4
|
const MAX_TERMS = 8;
|
|
5
5
|
const STOP_WORDS = new Set(`a an and are as at be been but by can could did do does for from had has have how i if in is it its me my of on or our please should so that the their them there these they this to was we were what when where which who why will with would you your about again before earlier previous remember recall history message messages conversation turn turns tell show find help need want use using make get know explain answer question code file project problem fix work task test tests implementation implement change thanks continue revisit discuss discussed discussion decide decided follow-up
|
|
6
6
|
的 了 是 在 和 与 或 我 你 他 她 它 我们 你们 他们 这个 那个 什么 怎么 如何 为什么 请 请问 帮 帮我 帮忙 可以 能 不能 是否 需要 想 要 再 还 又 也 就 都 把 将 给 对 从 到 上 下 中 里 有 没有 一下 一些 一个 这些 那些 之前 以前 上次 刚才 历史 记得 回忆 召回 消息 对话 问题 回答 内容 事情 继续 现在 今天 昨天 后来 然后 相关 具体 代码 文件 项目 实现 修改 功能 测试 方案 方法 工作 任务 谢谢 好的 好 看看 查找 搜索 查询 处理 解决 进行 使用 讨论 提到 记忆 总结 回顾 提醒`.split(/\s+/u));
|
|
7
|
+
for (const word of `system service config configuration settings build run error errors issue issues status result results request requests response responses check checks update updates version versions default option options limit limits page data user users model models tool tools server client input output changes details new old
|
|
8
|
+
系统 服务 配置 设置 构建 运行 错误 报错 状态 结果 请求 响应 检查 更新 版本 默认 选项 限制 页面 数据 用户 模型 工具 服务端 客户端 输入 输出 改动 详细 新 旧`.split(/\s+/u)) STOP_WORDS.add(word);
|
|
7
9
|
const segmenter = new Intl.Segmenter('zh', { granularity: 'word' });
|
|
10
|
+
const MIN_COVERAGE = 0.6;
|
|
11
|
+
const MIN_DISTINCTIVENESS = 0.5;
|
|
12
|
+
const MIN_SCORE = 6.5;
|
|
13
|
+
|
|
14
|
+
// Paths and concrete issue IDs are strong anchors even when a bounded sample
|
|
15
|
+
// contains many discussions of that exact reference. Mere camel/snake casing
|
|
16
|
+
// does not make an otherwise common word distinctive.
|
|
17
|
+
function isExactReference(term) {
|
|
18
|
+
return term.includes('/')
|
|
19
|
+
|| /\.[a-z][a-z0-9]{0,7}$/iu.test(term)
|
|
20
|
+
|| /^[a-z][a-z0-9]*(?:[-_:][a-z0-9]+)*[-_:]\d[\da-z_-]*$/iu.test(term)
|
|
21
|
+
|| /^(?:[a-z0-9]+[-_:])*[a-f0-9]{8,}$/iu.test(term);
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
function isGenericTerm(term) {
|
|
25
|
+
if (STOP_WORDS.has(term.toLocaleLowerCase())) return true;
|
|
26
|
+
const words = term.replace(/([a-z])([A-Z])/gu, '$1 $2').toLocaleLowerCase().split(/[\s_.:@-]+/u);
|
|
27
|
+
return words.every(word => STOP_WORDS.has(word));
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/** Automatic recall accepts zero and never exceeds five, even for raw callers. */
|
|
31
|
+
export function normalizeRecallLimit(value) {
|
|
32
|
+
const number = typeof value === 'number' || typeof value === 'string' && value.trim()
|
|
33
|
+
? Number(value) : NaN;
|
|
34
|
+
return Number.isFinite(number) ? Math.min(5, Math.max(0, Math.floor(number))) : 5;
|
|
35
|
+
}
|
|
8
36
|
|
|
9
37
|
function isIdentifier(term) {
|
|
10
38
|
return /[\p{L}\d][_.:/@-][\p{L}\d]/u.test(term)
|
|
@@ -20,7 +48,7 @@ export function extractRecallTerms(prompt) {
|
|
|
20
48
|
? prompt.slice(0, MAX_PROMPT_CHARS).replace(/@vp-[A-Za-z0-9_-]+\b/gu, ' ') : '';
|
|
21
49
|
const tokens = [];
|
|
22
50
|
// Keep paths, issue IDs, snake_case and camelCase intact before segmentation.
|
|
23
|
-
const rest = text.replace(/[\p{L}\p{N}]+(?:[_.:/@-][\p{L}\p{N}]+)+|[A-Za-z][A-Za-z0-9]*/gu, token => {
|
|
51
|
+
const rest = text.replace(/(?:\.{0,2}\/)?[\p{L}\p{N}]+(?:[_.:/@-][\p{L}\p{N}]+)+|[A-Za-z][A-Za-z0-9]*/gu, token => {
|
|
24
52
|
tokens.push(token);
|
|
25
53
|
return ' ';
|
|
26
54
|
});
|
|
@@ -44,7 +72,7 @@ export function extractRecallTerms(prompt) {
|
|
|
44
72
|
const seen = new Set();
|
|
45
73
|
const terms = tokens.filter(term => {
|
|
46
74
|
const key = term.toLocaleLowerCase();
|
|
47
|
-
if (seen.has(key) ||
|
|
75
|
+
if (seen.has(key) || isGenericTerm(term) || /^\d+$/u.test(key)
|
|
48
76
|
|| Array.from(key).length < 2 || key.length > 96) return false;
|
|
49
77
|
seen.add(key);
|
|
50
78
|
return true;
|
|
@@ -59,17 +87,21 @@ export function extractRecallTerms(prompt) {
|
|
|
59
87
|
* Returns explainable rejection reasons rather than weak positive matches.
|
|
60
88
|
*/
|
|
61
89
|
export function scoreRecallTurn(promptOrTerms, text, stats = {}) {
|
|
62
|
-
const terms = (Array.isArray(promptOrTerms) ? promptOrTerms : extractRecallTerms(promptOrTerms))
|
|
90
|
+
const terms = (Array.isArray(promptOrTerms) ? promptOrTerms : extractRecallTerms(promptOrTerms))
|
|
91
|
+
.filter(term => typeof term === 'string' && !isGenericTerm(term)).slice(0, MAX_TERMS);
|
|
63
92
|
const body = String(text || '').toLocaleLowerCase();
|
|
64
93
|
const matchedTerms = terms.filter(term => {
|
|
65
94
|
const key = term.toLocaleLowerCase();
|
|
66
|
-
if (/^[a-z0-9_]+$/u.test(key)) {
|
|
95
|
+
if (/^[a-z0-9_]+$/u.test(key) || isIdentifier(term)) {
|
|
67
96
|
const escaped = key.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
68
|
-
|
|
97
|
+
const boundary = isIdentifier(term) ? 'a-z0-9_.:/@-' : 'a-z0-9_';
|
|
98
|
+
const suffix = isIdentifier(term) ? '(?![a-z0-9_/@-]|[.:][a-z0-9_])' : `(?![${boundary}])`;
|
|
99
|
+
return new RegExp(`(?<![${boundary}])${escaped}${suffix}`, 'u').test(body);
|
|
69
100
|
}
|
|
70
101
|
return body.includes(key);
|
|
71
102
|
});
|
|
72
103
|
const identifierMatches = matchedTerms.filter(isIdentifier);
|
|
104
|
+
const exactReferenceMatches = matchedTerms.filter(isExactReference);
|
|
73
105
|
const independentTerms = matchedTerms.filter(term => !matchedTerms.some(other => (
|
|
74
106
|
other !== term && other.toLocaleLowerCase().includes(term.toLocaleLowerCase())
|
|
75
107
|
)));
|
|
@@ -79,16 +111,17 @@ export function scoreRecallTurn(promptOrTerms, text, stats = {}) {
|
|
|
79
111
|
const frequency = stats.termDocumentFrequency?.[term.toLocaleLowerCase()];
|
|
80
112
|
return sampleSize >= 8 && Number.isFinite(frequency) ? 1 - frequency / sampleSize : 1;
|
|
81
113
|
})) : 0;
|
|
114
|
+
const rawScore = Math.round((independentTerms.length * 2 + identifierMatches.length * 4
|
|
115
|
+
+ coverage * 2 + distinctiveness) * 100) / 100;
|
|
82
116
|
let reason = 'relevant';
|
|
83
117
|
if (!terms.length) reason = 'generic_prompt';
|
|
84
118
|
else if (!matchedTerms.length) reason = 'no_match';
|
|
85
119
|
else if (!identifierMatches.length && independentTerms.length < 2) reason = 'insufficient_keywords';
|
|
86
|
-
else if (
|
|
87
|
-
else if (!
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
return { score, matchedTerms, reason, coverage, distinctiveness, identifierMatches, sampleSize };
|
|
120
|
+
else if (coverage < MIN_COVERAGE) reason = 'low_coverage';
|
|
121
|
+
else if (!exactReferenceMatches.length && distinctiveness < MIN_DISTINCTIVENESS) reason = 'low_distinctiveness';
|
|
122
|
+
else if (rawScore < MIN_SCORE) reason = 'low_score';
|
|
123
|
+
const score = reason === 'relevant' ? rawScore : 0;
|
|
124
|
+
return { score, matchedTerms, reason, coverage, distinctiveness, identifierMatches, exactReferenceMatches, sampleSize };
|
|
92
125
|
}
|
|
93
126
|
|
|
94
127
|
export const RECALL_LIMITS = Object.freeze({
|