dsh-vibe-math 1.2.3 → 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +0 -2
- package/package.json +1 -1
- package/vibe-math-v3/vibe-math-v3.js +28 -12
package/README.md
CHANGED
|
@@ -39,8 +39,6 @@
|
|
|
39
39
|
|
|
40
40
|
**一句话流水线**:全部知识以 **Markdown 论文/研究报告式**存储与续写(`Problems/` 问题清单含依赖/后生问题来源动机、`Progress/` 研究日志按方向按轮续写、`Propos/` 命题库、`Methods/` 通用理论发明库、`Verified/` 绝对可信)→ 调度前调度器构造状态简报并调用**规划代理**,规划代理一次性安排接下来 N 步(spawn solver/verifier/explorer/method-keeper、interrupt、promote、wait),代码校验后执行(超出并发的动作排队跨 tick 消费;规划失败自动回退 v2 式启发式)→ 验证器独立审查→辩论→**近共识裁决**(同侧且均值 ≥0.85/≤0.15 取均值,修复 v2 flat 误判)→ 概率=1 收口并生成 `Verified/` 卡 → 求解器的 `methods_used`/`new_inventions` 上报由 **Method Keeper** 沉淀/完善方法库(可组成体系层级、跨项目复用)。
|
|
41
41
|
|
|
42
|
-
**一句话流水线**:全部知识以 **Markdown 论文/研究报告式**存储与续写(`Problems/` 问题清单含依赖/后生问题来源动机、`Progress/` 研究日志按方向按轮续写、`Propos/` 命题库、`Methods/` 通用理论发明库、`Verified/` 绝对可信)→ 调度前调度器构造状态简报并调用**规划代理**,规划代理一次性安排接下来 N 步(spawn solver/verifier/explorer/method-keeper、interrupt、promote、wait),代码校验后执行(超出并发的动作排队跨 tick 消费;规划失败自动回退 v2 式启发式)→ 验证器独立审查→辩论→**近共识裁决**(同侧且均值 ≥0.85/≤0.15 取均值,修复 v2 flat 误判)→ 概率=1 收口并生成 `Verified/` 卡 → 求解器的 `methods_used`/`new_inventions` 上报由 **Method Keeper** 沉淀/完善方法库(可组成体系层级、跨项目复用)。
|
|
43
|
-
|
|
44
42
|
---
|
|
45
43
|
|
|
46
44
|
## ✨ 功能特色
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-vibe-math",
|
|
3
3
|
"description": "Multi-agent mathematical problem-solving & verification frameworks for DeepSeek Harness — THREE agent presets in one install: vibe-math-v1 (classic pipeline), vibe-math-v2 (probability-driven: qs.json + Propos knowledge base + explorer→solver→review/debate verdict), and vibe-math-v3 (THIRD-generation, recommended: paper-style Markdown knowledge base with Problems/Progress/Propos/Methods/Verified + planner-agent scheduling that decides the next N actions + universal theory/method invention library + agents write their own Markdown directly via a per-file write lock). Installing this bundle auto-installs all three presets into the DSH preset root.",
|
|
4
|
-
"version": "1.2.
|
|
4
|
+
"version": "1.2.5",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "installer.js",
|
|
7
7
|
"exports": {
|
|
@@ -912,10 +912,10 @@ export function apply(ctx) {
|
|
|
912
912
|
return '\n3) FOLDERS:Problems/ 问题清单;Progress/ 研究日志(每问题一个聚合索引 <qid>.md + 每方向一个文件 <qid>/<dirId>.md,按方向按轮续写);Propos/ 命题库;Methods/ 理论发明库;Verified/ 绝对可信(只读);Reliable/ 可信参考文献(只读);Notes/ 自由笔记;Logs/ 审计;State/ 调度器私有——不要读也不要改。\n'
|
|
913
913
|
}
|
|
914
914
|
function kcMethodLibrary() {
|
|
915
|
-
return '\
|
|
915
|
+
return '\n5) METHOD LIBRARY RULES:开工前先查 Methods/(含全局 VibeMath/Methods/),有可复用方法/体系则引用其 ID;用后必须在 methods_used 上报(含效果与改进建议);本轮新发明/经验性总结必须在 new_inventions 上报(类型:理论体系|框架|工具|方法|思想|范式|技巧)——若与某张已有方法卡同类,在内容描述里注明"可并入 m-xxx"以便 Method Keeper 合并而非重复建卡。**重要区分**:methods_used 只能填**已存在方法卡的 ID**(形如 m-abc12345,来自 AVAILABLE METHODS 列表);你自己刚想出的新方法/新技巧不属于 methods_used,请如实填入 new_inventions(由 Method Keeper 蒸馏建卡);千万不要把方法名/标题文字当 id 填进 methods_used。\n'
|
|
916
916
|
}
|
|
917
917
|
function kcOutputQuality() {
|
|
918
|
-
return '\
|
|
918
|
+
return '\n4) OUTPUT QUALITY RULES:完整性、不断章取义——任何输出的问题/命题/结论都要给出完整陈述并补全所依赖的对象/环境/背景定义;引用必须给出处(文件路径 + ID + 锚点/节),事实只引 Verified/;若结论依赖临时假设 p,必须显式写「若 <p 完整陈述> 成立,则:…」。你的机器回复是一个 JSON 对象(```json 围栏内),JSON 之外不要再输出其他文本——任何要写进 md 的内容都通过文件工具写入,不要当作聊天气泡输出。'
|
|
919
919
|
}
|
|
920
920
|
// 写文件共用规则(仅供真正写 md 的代理:solver / method-keeper):写锁、上报、回退;不含具体归属文件(按角色注入)
|
|
921
921
|
function kcWriteRules() {
|
|
@@ -994,7 +994,7 @@ export function apply(ctx) {
|
|
|
994
994
|
knowledgeContextText('explorer') +
|
|
995
995
|
methodsIndexText() +
|
|
996
996
|
capabilitiesText('explorer') +
|
|
997
|
-
'\nDo a first-stage METACOGNITIVE BRAINSTORM: decompose constraints, test boundary/extreme cases, map to similar known problems. First check the AVAILABLE METHODS list — if a listed method/system underlies a direction you will propose, reference its id in methods_used (
|
|
997
|
+
'\nDo a first-stage METACOGNITIVE BRAINSTORM: decompose constraints, test boundary/extreme cases, map to similar known problems. First check the AVAILABLE METHODS list — if a listed method/system underlies a direction you will propose, reference its id in methods_used (the method card will log this direction as building on it; you are planning to leverage it, not claiming you already applied it). ' +
|
|
998
998
|
'Then propose 3-6 DIVERSE, mutually distinct solution directions (e.g. analytic method, constructive proof, contradiction, numeric approximation + limit passage, categorical abstraction, ...). ' +
|
|
999
999
|
'Record each direction with its core assumption and an initial feasibility estimate. Every direction must be self-contained: title / method / core_assumption written completely, defining every object they mention — no 断章取义.\n\n' +
|
|
1000
1000
|
'feasibility ∈ [0,1] = your estimate of the probability this direction leads to a full solution. Respond with ONLY a single JSON object in a ```json code fence (no prose outside it). Register the directions as metadata; the scheduler writes them into the research log:\n' +
|
|
@@ -1039,11 +1039,13 @@ export function apply(ctx) {
|
|
|
1039
1039
|
head += knowledgeContextText('solver')
|
|
1040
1040
|
head += methodsIndexText()
|
|
1041
1041
|
head += capabilitiesText('solver')
|
|
1042
|
-
head += '\nStart from the last recorded node of direction ' + dir.id + ' (inherit progress, or branch a sub-route under it). Consult AVAILABLE METHODS first — reuse a listed method/system when it fits (report it in methods_used)
|
|
1042
|
+
head += '\nStart from the last recorded node of direction ' + dir.id + ' (inherit progress, or branch a sub-route under it). Consult AVAILABLE METHODS first — reuse a listed method/system when it fits (report it in methods_used).\n' +
|
|
1043
|
+
'PRIMARY GOAL: drive toward a COMPLETE solution of the problem along this direction. The single most valuable thing you can deliver is the full proof/solution; intermediate lemmas, sub-routes, lessons and inventions are by-products to record as you go, NOT the main deliverable — do not spread your effort across them at the expense of the proof itself. If the complete solution is not attainable this round, report honestly and still push as far as the core argument as you can.\n' +
|
|
1044
|
+
'Each round you should report (whenever produced):\n' +
|
|
1043
1045
|
'- new lemmas / intermediate conclusions WITH full proofs (they become Propos/ proposition cards);\n' +
|
|
1044
1046
|
'- each concrete sub-route tried, its progress overview, an EXPLICIT feasibility signal (e.g. "unremovable singularity", "conflicts with known theorem X"), and any blocker;\n' +
|
|
1045
1047
|
'- lessons learned from failed attempts;\n' +
|
|
1046
|
-
'-
|
|
1048
|
+
'- survival ∈ (0,1) = your updated confidence that this direction can still be pushed to a full proof (not the confidence the current partial work is right);\n' +
|
|
1047
1049
|
'- ANY new theory/tool/method/idea you invented or summarized this round in new_inventions (类型:理论体系|框架|工具|方法|思想|范式|技巧) — the Method Keeper will distill it into the theory library.'
|
|
1048
1050
|
head += '\nIf you encounter an EXTREMELY complex auxiliary conjecture/sub-problem q_sub: list it in "sub_questions" as a PROBLEM-class object with its COMPLETE statement (every object/definition/notation fully defined — 不断章取义), together with p_{q-tmp}: a PROPOSITION-class TEMPORARY ASSUMPTION answering q_sub. TEMPORARILY ASSUME p_{q-tmp} holds and continue the main line — every later proposition/conclusion depending on it MUST be stated as "若 <p_{q-tmp} 的完整陈述> 成立,则:..." (complete definitions).\n'
|
|
1049
1051
|
head += '\nIMPORTANT — PROBABILITY RULES FOR NEW RESULTS: any 概率 / prob / solution_prob / survival you output for NEW results must be strictly BETWEEN 0 and 1 (they await independent verifier confirmation). NEVER mark your own fresh lemma or solution as 1 or 0 — that is the verifiers\' job. Only facts already recorded in Verified/ count as certain.\n'
|
|
@@ -1444,20 +1446,28 @@ export function apply(ctx) {
|
|
|
1444
1446
|
const actions = (parsed && Array.isArray(parsed.plan)) ? parsed.plan : []
|
|
1445
1447
|
const summary = (parsed && parsed.summary) ? String(parsed.summary) : ''
|
|
1446
1448
|
const planId = meta.planId || ('plan-' + shortId())
|
|
1447
|
-
if (
|
|
1449
|
+
if (!parsed) {
|
|
1450
|
+
// 输出不可解析 = 真正的规划器失败 → 计失败,3 次禁用(退回启发式)
|
|
1448
1451
|
plannerFails += 1
|
|
1449
|
-
logActivity('plan', 'planner ' + planId + ' returned
|
|
1452
|
+
logActivity('plan', 'planner ' + planId + ' returned UNPARSEABLE output (' + String(output || '').slice(0, 200) + ')')
|
|
1450
1453
|
if (plannerFails >= (Number(params.plannerMaxFails) || 3)) { params.plannerEnabled = false; logActivity('plan', 'planner disabled after ' + plannerFails + ' consecutive failures — heuristic mode') }
|
|
1451
1454
|
await fallbackScheduler()
|
|
1452
1455
|
return
|
|
1453
1456
|
}
|
|
1457
|
+
if (actions.length === 0) {
|
|
1458
|
+
// 空计划:多为"规划到处理之间工作已被完成/解决"的状态竞争,不是规划器失败——不累计、不禁用
|
|
1459
|
+
logActivity('plan', 'planner ' + planId + ' returned empty plan (no actionable work)')
|
|
1460
|
+
await fallbackScheduler()
|
|
1461
|
+
return
|
|
1462
|
+
}
|
|
1454
1463
|
plannerFails = 0
|
|
1455
1464
|
const validated = await validatePlan(actions)
|
|
1456
1465
|
lastPlanSummary = { at: now(), planId: planId, summary: summary, actions: validated.length, outcomes: [] }
|
|
1457
1466
|
await writeJson('Logs/Plans/' + planId + '.json', { at: now(), planId: planId, summary: summary, raw: actions, validated: validated, project: currentProject })
|
|
1458
1467
|
logActivity('plan', 'planner ' + planId + ' → ' + validated.length + '/' + actions.length + ' valid action(s): ' + validated.map(function (a) { return a.action + (a.role ? ':' + a.role : '') + (a.target ? ':' + a.target : '') }).join(', '))
|
|
1459
1468
|
if (validated.length === 0) {
|
|
1460
|
-
|
|
1469
|
+
// 计划被校验全部剔除(多为规划后状态已变/动作冗余,属状态竞争)——不累计 plannerFails、不禁用规划器
|
|
1470
|
+
logActivity('plan', 'planner ' + planId + ' plan all filtered by validation (state changed) — no op')
|
|
1461
1471
|
await fallbackScheduler()
|
|
1462
1472
|
return
|
|
1463
1473
|
}
|
|
@@ -2272,9 +2282,10 @@ export function apply(ctx) {
|
|
|
2272
2282
|
}
|
|
2273
2283
|
async function checkTermination() {
|
|
2274
2284
|
const unsolved = allProblems().filter(function (q) { return !(q.状态 === '已解决' || q.优先级 === 'never') })
|
|
2275
|
-
//
|
|
2276
|
-
//
|
|
2277
|
-
|
|
2285
|
+
// 终止前必须无遗留验证对象(未验证命题/证明/解法);否则会在命题/解法仍未验证时提前停机。
|
|
2286
|
+
// 注意:待沉淀发明(pendingInventions)不应阻塞终止——它是"批量蒸馏"(积够 methodKeepEvery 才触发),
|
|
2287
|
+
// 少量残留不会再有 method-keeper 触发,若纳入会令 scheduler 永不终止(闲置空转)。
|
|
2288
|
+
const leftoverVerify = (await buildVerifyCandidates()).length > 0
|
|
2278
2289
|
if (unsolved.length === 0 && !leftoverVerify && Object.keys(agentRegistry).length === 0 && Object.keys(tasks).length === 0 && planQueue.length === 0) {
|
|
2279
2290
|
scheduler.running = false
|
|
2280
2291
|
await releaseProjectLock()
|
|
@@ -2409,7 +2420,12 @@ export function apply(ctx) {
|
|
|
2409
2420
|
const dir = dirs.find(function (d) { return d.id === dirId })
|
|
2410
2421
|
if (q && dir) {
|
|
2411
2422
|
if (typeof meta.survival === 'number') dir.survival = clamp01(meta.survival)
|
|
2412
|
-
if (meta.status)
|
|
2423
|
+
if (meta.status) {
|
|
2424
|
+
// 只有 success/dead-end 是方向的调度终态;'continue' 表示方向仍可续轮,必须保持 active,
|
|
2425
|
+
// 否则 validatePlan / hasSchedulableWork / fallbackScheduler 只认 'active' 会漏调度 → 方向永久卡死。
|
|
2426
|
+
const st = String(meta.status)
|
|
2427
|
+
if (st === 'success' || st === 'dead-end') dir.status = st
|
|
2428
|
+
}
|
|
2413
2429
|
if (meta.dead_end_reason) dir.dead_end_reason = String(meta.dead_end_reason)
|
|
2414
2430
|
if (meta.round) dir.round = Number(meta.round)
|
|
2415
2431
|
// 轮次上限:达到 solverMaxRounds 且仍未成功/死路 → 强制死路(新协议路径没有 followup 自迭代,必须靠此收口,与旧路径一致)
|