dsh-vibe-math 1.3.9 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md
CHANGED
|
@@ -63,6 +63,8 @@
|
|
|
63
63
|
> 🔧 **v1.3.8 基于真实测试的诊断修复**:**① 共识验证真正"全体一致"**——`finalizeVerify` 现在要求**全体在册常驻都投了票**才可判"一致",否则辩论或保留为未定论(实测曾出现 2/4 投票却被判"全体一致为真",已修);**② 会议必须全体发言**——`continueMeetingRound` 按"是否已发言"收口,`allSolved`(stop)要求全员发言+全票 true,杜绝缺席成员被带偏;**③ 背景只在首轮讲一次**——`contextBrief` 只在 `brainstormPrompt`(首轮)注入完整版,后续 normal/meeting/verify/CHECKPOINT 用极简当前状态,不再每轮重复长背景(省上下文);**④ 验证目标按提出者精确定位**(`targetOwner` + `findSourceRel` 优先提出者库,避免同名 id 撞车);**⑤ 主代理放权**——persona 明确"让常驻自组织(hands-off)",不注议程/优先级/分工/验证决定,只 read status/report,用户明确要求或明显僵死时才 message/meeting 且只促成不决定。详见 §21。
|
|
64
64
|
>
|
|
65
65
|
> 🔧 **v1.3.9 进一步按哲学打磨**:**① 会议议程来源**确认是非 bug(r-1 的 `propose_meeting` 发起);**② 压缩后重申核心规则**——每次压缩触发时在提示开头重申短核心规则(治"压缩遗忘规则"),其余轮仍极简;**③ 验证 `verdict` 明确为 0–1 正确概率**——1=绝对为真/判真、0=绝对为假/判假、0.5=不确定,"全体一致判真(verdict=1)/判假(verdict=0)才算数",未全票留库附平均正确概率(兼容旧字符串);**④ 创建项目不立即启动**——新增 `vibe_v4_configure {project?, problem?, params?}` 只建/配项目与参数(持久化 `State/settings.json`)不唤醒常驻,随后 `vibe_v4_start` 才启动;支持设置**项目名**。详见 §22。
|
|
66
|
+
>
|
|
67
|
+
> 🔧 **v1.4.0 verdict 改为纯概率数值 + 全面审计修复**:**① `verdict` 是纯 0–1 正确概率(程度),不再二分类**——仅当**全体一致给 1(真)或全体一致给 0(假)**才按真/假写入 Verified/,否则只作为概率数值保留、附全组平均正确概率(0.97 不再算"真",属更严格的绝对一致口径);**② 审计修复**:`resume` 补 `loadSettings()`(跨进程不再重置参数)、`configure` 直接写出问题卡、补上缺失的 `/v4 set` 分支、真实 `/compact` 成功后置 `needCompact` 以**重申核心规则**。详见 §23。
|
|
66
68
|
|
|
67
69
|
---
|
|
68
70
|
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-vibe-math",
|
|
3
3
|
"description": "Multi-agent mathematical problem-solving & verification frameworks for DeepSeek Harness — FOUR agent presets in one install: vibe-math-v1 (classic pipeline), vibe-math-v2 (probability-driven: qs.json + Propos knowledge base + explorer→solver→review/debate verdict), vibe-math-v3 (THIRD-generation, recommended: paper-style Markdown knowledge base with Problems/Progress/Propos/Methods/Verified + planner-agent scheduling that decides the next N actions + universal theory/method invention library + agents write their own Markdown directly via a per-file write lock), and vibe-math-v4 (FOURTH-generation: persistent self-organizing resident subagents that message & meet to decide all tasks, verify only by unanimous consensus, /compact at a context threshold, and stop only when all agree the problem is solved). Installing this bundle auto-installs all four presets into the DSH preset root.",
|
|
4
|
-
"version": "1.
|
|
4
|
+
"version": "1.4.0",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "installer.js",
|
|
7
7
|
"exports": {
|
|
@@ -194,11 +194,11 @@ export function apply(ctx) {
|
|
|
194
194
|
+'{"input":"<your real contribution to this discussion>","propose_task":"<task title or null>","task_desc":"...","claim_task":"<task id or null>","propose_verify":"<id or null>","voteSolved":true}'
|
|
195
195
|
}
|
|
196
196
|
function verifyPrompt(r, vs){
|
|
197
|
-
const others=Object.entries(vs.verdicts).map(([k,v])=>'- '+k+':
|
|
197
|
+
const others=Object.entries(vs.verdicts).map(([k,v])=>'- '+k+': 正确概率 '+String(v.prob!=null?Number(v.prob).toFixed(2):0.5)+' → '+v.reason).join('\n')
|
|
198
198
|
return (params.residentPersona?params.residentPersona+'\n':'')
|
|
199
199
|
+'Resident '+r.rId+' — 团队验证。 The group is verifying object '+vs.targetId+'('+vs.targetType+',提出者 '+vs.targetOwner+')。\n'
|
|
200
|
-
+'
|
|
201
|
-
+'
|
|
200
|
+
+'请给出你对「该对象为真」的**正确概率 `verdict`**,仅一个 0–1 数值:**1 = 绝对为真,0 = 绝对为假,0.5 = 完全不确定,其余为介于其间的程度**(不要给 TRUE/FALSE,就给一个数值)。\n'
|
|
201
|
+
+'判定规则:仅当**全体常驻一致给 1(都认为是真)或一致给 0(都认为是假)**,才按「真/假」写入 Verified/;否则**只作为概率数值(一种程度)保留在库中**,附全组平均正确概率,不写成真/假。\n'
|
|
202
202
|
+'请给出你**诚实独立的判断**'
|
|
203
203
|
+(vs.stage==='debate'?',并参考他人意见:\n':'。\n')
|
|
204
204
|
+(vs.stage==='debate'&&others?('### 他人意见(已转发给你)\n'+others+'\n'):'')
|
|
@@ -345,8 +345,9 @@ export function apply(ctx) {
|
|
|
345
345
|
const vs=verifyState; const expected=Array.from(residents.keys()).length
|
|
346
346
|
const allVoted = expected>0 && Object.keys(vs.verdicts).length>=expected
|
|
347
347
|
const vals=Object.values(vs.verdicts)
|
|
348
|
-
|
|
349
|
-
const
|
|
348
|
+
// verdict is a PURE 0-1 probability; only ALL=1 (true) or ALL=0 (false) is a binary verdict.
|
|
349
|
+
const allTrue = allVoted && vals.every(x=>Number(x.prob)===1)
|
|
350
|
+
const allFalse = allVoted && vals.every(x=>Number(x.prob)===0)
|
|
350
351
|
if(allTrue||allFalse){ await closeVerify(vs,allTrue); return }
|
|
351
352
|
if(vs.round+1<params.verdictMaxRounds){ vs.stage='debate'; vs.round+=1; vs.asked=[]; logActivity('verify',vs.targetId+' round '+vs.round+' → debate'); await saveAll(); await scheduleNext(); return }
|
|
352
353
|
const avg=vals.length? vals.reduce((a,x)=>a+(x.prob!=null?x.prob:0.5),0)/vals.length : 0.5
|
|
@@ -363,7 +364,7 @@ export function apply(ctx) {
|
|
|
363
364
|
}
|
|
364
365
|
async function writeDebateDoc(vs,done,val){
|
|
365
366
|
const lines=['# 验证辩论|'+vs.targetId+'('+vs.targetType+')|'+fmtTime(),'',(done?('**结论**:'+(val===1?'全体一致为真':'全体一致为假')):('**未达成全体一致**,平均概率 '+val.toFixed(2))),'','## 各常驻意见']
|
|
366
|
-
for(const [k,v] of Object.entries(vs.verdicts)){ lines.push('### '+k+'
|
|
367
|
+
for(const [k,v] of Object.entries(vs.verdicts)){ lines.push('### '+k+'|正确概率 '+(v.prob!=null?Number(v.prob).toFixed(2):'0.50')); lines.push(v.reason||''); lines.push('') }
|
|
367
368
|
await writeText('Shared/debates/'+vs.targetId+'.md', lines.join('\n'))
|
|
368
369
|
}
|
|
369
370
|
async function writeVerifiedCard(vs,isTrue){
|
|
@@ -432,7 +433,9 @@ export function apply(ctx) {
|
|
|
432
433
|
const signal = makeSignal(params.activityTimeoutMs||60000)
|
|
433
434
|
const result = await compaction.compactIfNeeded(agent, 'pressure', signal)
|
|
434
435
|
if(result && (result.shadowedSeqs||[]).length>0){
|
|
435
|
-
|
|
436
|
+
// the resident's real session was compacted → its context is now a summary.
|
|
437
|
+
// Flag needCompact so the NEXT wake re-anchors the core rules (they may have been blurred).
|
|
438
|
+
r.roundsSinceCompact=0; r.needCompact=true; r.contextPct=Math.min(r.contextPct||15,25)
|
|
436
439
|
logActivity('compact', r.rId+' real /compact (shadowed '+result.shadowedSeqs.length+' items, ~'+String(result.shadowedTokenCount||0)+' tokens)')
|
|
437
440
|
}
|
|
438
441
|
} catch(e){ /* real compaction unavailable/failed; the soft directive already covers it */ }
|
|
@@ -502,8 +505,8 @@ export function apply(ctx) {
|
|
|
502
505
|
else if(/^TRUE$/i.test(String(v.verdict))){ p=1 }
|
|
503
506
|
else if(/^FALSE$/i.test(String(v.verdict))){ p=0 }
|
|
504
507
|
else { p=clamp01(Number(v.confidence)) }
|
|
505
|
-
|
|
506
|
-
verifyState.verdicts[r.rId]={
|
|
508
|
+
// verdict is a PURE 0-1 probability (a degree); no binary TRUE/FALSE classification.
|
|
509
|
+
verifyState.verdicts[r.rId]={prob:p,confidence:p,reason:String(v.reason||parsed.summary||'')}
|
|
507
510
|
await saveAll(); await continueVerifyRound(); return
|
|
508
511
|
}
|
|
509
512
|
// normal turn
|
|
@@ -542,7 +545,7 @@ export function apply(ctx) {
|
|
|
542
545
|
await saveAll(); return {ok:true,message:'v4 started: '+params.residentCount+' resident(s) brainstorming',project:currentProject}
|
|
543
546
|
}
|
|
544
547
|
async function resume(){
|
|
545
|
-
currentProject=await readCurrentProject(); await ensureDirs(); await loadAll()
|
|
548
|
+
currentProject=await readCurrentProject(); await ensureDirs(); await loadAll(); await loadSettings()
|
|
546
549
|
if(phase==='idle' && !running && residents.size===0) return {ok:false,message:'nothing to resume'}
|
|
547
550
|
// If the persisted State came from a DIFFERENT process (crash/restart), the saved
|
|
548
551
|
// childIds are stale; clear them so residents re-spawn (their libraries persist on
|
|
@@ -573,7 +576,10 @@ export function apply(ctx) {
|
|
|
573
576
|
if(cfg && cfg.project && String(cfg.project).trim()) currentProject=String(cfg.project).trim()
|
|
574
577
|
if(cfg && cfg.problem) problemText=String(cfg.problem)
|
|
575
578
|
if(cfg && cfg.params && typeof cfg.params==='object') for(const k of Object.keys(cfg.params)) if(k in params) params[k]=cfg.params[k]
|
|
576
|
-
await writeCurrentProject(); await ensureDirs(); await saveSettings()
|
|
579
|
+
await writeCurrentProject(); await ensureDirs(); await saveSettings()
|
|
580
|
+
// create the problem card so the project is complete BEFORE the run starts
|
|
581
|
+
if(problemText){ const pid=slugify(problemText.slice(0,40))||'problem'; await writeText('Problems/'+pid+'.md','# 问题|'+pid+'\n- ID: '+pid+'\n- 类型: 问题\n- 状态: 求解中\n- 优先级: 1\n- 依赖: []\n\n## 陈述\n'+problemText+'\n') }
|
|
582
|
+
await saveAll()
|
|
577
583
|
return {ok:true,project:currentProject,problem:problemText?problemText.slice(0,60):'',params:Object.keys(params).map(k=>k+'='+params[k]).join(', ')}
|
|
578
584
|
}
|
|
579
585
|
async function initAbort(){ clearHeartbeat(); running=false; phase='idle'; autoDone=false; for(const [,r] of residents){ if(r.childId){ try{ subagents.interrupt(r.childId,{kind:'ancestor',agent:rootAgent}) }catch(e){} } r.childId=''; r.lastActiveAt=0; r.roundsSinceCompact=0 } await saveAll(); return {ok:true,message:'aborted'} }
|
|
@@ -656,6 +662,7 @@ export function apply(ctx) {
|
|
|
656
662
|
else if(cmd==='members') r={ok:true,residents:s.listResidents()}
|
|
657
663
|
else if(cmd==='add') r=await s.addMember(rest.join(' '))
|
|
658
664
|
else if(cmd==='remove') r=await s.removeMember(rest[0]||'')
|
|
665
|
+
else if(cmd==='set'){ const upd={}; for(const tok of rest){ const eq=tok.indexOf('='); if(eq>0){ const k=tok.slice(0,eq); const rv=tok.slice(eq+1); const n=Number(rv); upd[k]=Number.isFinite(n)?n:rv } } r=s.setParams(upd) }
|
|
659
666
|
else r={ok:false,usage:'configure|start|resume|pause|abort|status|report|message|meeting|members|add|remove|set'}
|
|
660
667
|
return {kind:'success',text:JSON.stringify(r,null,2)}
|
|
661
668
|
},
|
|
@@ -462,5 +462,24 @@ VibeMath/Projects/<project>/
|
|
|
462
462
|
### 4. 创建项目不再立即启动(先配置后启动)+ settings 文件 + 项目名
|
|
463
463
|
新增 **`vibe_v4_configure {project?, problem?, params?}`**:**只创建/配置项目(名称、问题、参数),不唤醒任何常驻**;参数持久化到 `State/settings.json`。之后 **`vibe_v4_start {problem?, residentCount?, seedDirections?}`** 才真正启动(若已配置问题可省略)。这样"先设好参数再启动",不再一创建就着急跑。`/v4 configure` 子命令 + 主代理 persona 的 `Main controls` 已同步(`vibe_v4_set` 也持久化到 settings 文件);支持设置**项目名**(`configure.project`)。
|
|
464
464
|
|
|
465
|
+
---
|
|
466
|
+
|
|
467
|
+
## 23. verdict 改为纯概率数值 + 全面审计修复(v1.4.0)
|
|
468
|
+
|
|
469
|
+
### 1. `verdict` 是**纯 0–1 概率数值**(不再二分类)
|
|
470
|
+
按用户要求,`verdict` 现在**只是一个 0–1 的正确概率**(程度),框架**不再把它分段映射成 TRUE/FALSE/不确定**。判定规则:
|
|
471
|
+
- **仅当全体常驻一致给 `1`(都认为是真)** → 按"真"写入 Verified/;
|
|
472
|
+
- **仅当全体常驻一致给 `0`(都认为是假)** → 按"假"写入 Verified/;
|
|
473
|
+
- **否则**:只作为**概率数值(一种程度)保留在库中**,附全组平均正确概率(`- 概率:` 写该平均值),**不写成真/假**。
|
|
474
|
+
|
|
475
|
+
相应地:验证提示改为"请给出对该对象为真的正确概率 verdict(一个 0–1 数值,不要给 TRUE/FALSE)";解析/存储/辩论录/源卡回写全用**数值概率**。向后兼容旧 `"TRUE"/"FALSE"`(解析为 1/0)。**注意**:这也意味着"0.97(很高但非 1)"不再算"真",会保留为概率 0.97——这是更严格的"绝对一致"口径。
|
|
476
|
+
|
|
477
|
+
### 2. 全面审计修复
|
|
478
|
+
- **`resume` 补上 `loadSettings()`**:否则跨进程 resume 后参数会退回默认(settings 不加载)。
|
|
479
|
+
- **`configure` 现在直接写出问题卡**(`Problems/<id>.md`),使"创建项目"在启动前就完整;`start` 仍会幂等重写。
|
|
480
|
+
- **补上缺失的 `/v4 set` 分支**:此前 usage 列了 `set` 但命令处理器没实现,`/v4 set` 会掉到 usage;现支持 `key=value` 解析并 `setParams`(也持久化到 settings)。
|
|
481
|
+
- **真实 `/compact` 后重申规则**:`realCompact` 成功压缩常驻真实会话后,置 `needCompact=true`,使下一次唤醒**重申核心规则**(与软压缩一致),避免真实压缩后常驻淡忘规则。
|
|
482
|
+
|
|
483
|
+
|
|
465
484
|
|
|
466
485
|
|