@furongjun1999/dsh-memory 0.4.8 → 0.4.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +54 -26
- package/codebuddy/CODEBUDDY.md +196 -195
- package/codebuddy/README.md +13 -1
- package/codebuddy/mcp.json +9 -0
- package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -0
- package/docs/README.md +1 -1
- package/docs/discipline/harnesses.yaml +18 -7
- package/docs/discipline/templates/full.md.tmpl +4 -3
- package/docs/experiments/linkref_backfill/candidates_20260917.json +726 -0
- package/docs/experiments/linkref_backfill/candidates_internal_20260917.json +602 -0
- package/docs/experiments/linkref_backfill/candidates_internal_v2.json +603 -0
- package/docs/experiments/linkref_backfill/candidates_secret_20260917.json +884 -0
- package/docs/experiments/linkref_backfill/candidates_secret_v2.json +789 -0
- package/docs/hive//345/244/232/347/253/257harness/351/200/232/344/277/241/345/245/221/347/272/246_v0.1.md +42 -0
- package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -0
- package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -0
- package/docs/hive//347/234/237/345/256/236/345/272/223/347/253/257/345/210/260/347/253/257/345/256/236/346/265/213_S1b/344/270/216/345/217/254/345/233/236/346/235/203/350/241/241_v0.1.md +57 -0
- package/docs/hive//347/234/237/345/256/236/345/272/223/347/253/257/345/210/260/347/253/257/345/256/236/346/265/213_S7/345/200/222/346/216/222/345/200/231/351/200/211/345/261/202_v0.1.md +126 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -0
- package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -0
- package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.5.md +631 -625
- package/docs/mdcg//344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/345/245/221/347/272/246_v0.1.md +82 -0
- package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -0
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +47 -45
- package/docs/mdcg//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/{206_v0.4.md → 206_v0.5.md} +92 -4
- package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -0
- package/docs/mdcg//347/216/257/344/272/214_/347/231/275/347/256/261/345/241/253/345/205/205/346/265/201/346/260/264/347/272/277_v0.1.md +31 -0
- package/docs/mdcg//350/267/250/347/253/257/351/252/214/350/257/201/344/270/216/345/220/214/346/255/245/345/215/217/350/256/256_v0.1.md +93 -0
- package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +2 -2
- package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +3 -3
- package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +2 -2
- package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +434 -433
- package/docs//347/201/265/346/236/242/350/207/252/346/210/221/346/224/271/350/277/233/345/267/245/344/275/234/350/256/241/345/210/222_/345/244/226/351/203/250/347/240/224/347/251/266/347/263/273/345/210/227/345/220/270/346/224/266_v1_20260919.md +286 -0
- package/dsh/README.md +33 -0
- package/dsh/cordis.yml.example +13 -7
- package/dsh/hive-mcp-probe.mjs +94 -0
- package/dsh/hive-mcp.example.yml +62 -0
- package/dsh/update-lingshu.bat +11 -0
- package/dsh/update-lingshu.ps1 +337 -0
- package/lib/hooks.d.ts +3 -0
- package/lib/hooks.js +17 -23
- package/lib/index.d.ts +4 -2
- package/lib/index.js +29 -6
- package/lib/lib/datapath.d.ts +76 -1
- package/lib/lib/datapath.js +199 -13
- package/lib/lib/mdcg_client.d.ts +6 -3
- package/lib/lib/mdcg_client.js +6 -5
- package/lib/lib/mutual.js +4 -4
- package/lib/lib/token_store.js +4 -5
- package/md_cg/audit.py +17 -2
- package/md_cg/autonomy.py +86 -15
- package/md_cg/backfill.py +36 -1
- package/md_cg/backfill_bigdomain.py +34 -0
- package/md_cg/bench6_arms.py +28 -1
- package/md_cg/bench6_common.py +10 -1
- package/md_cg/bench6_competitors.py +6 -1
- package/md_cg/bench_axis_domain.py +9 -1
- package/md_cg/bench_blind_comp.py +7 -1
- package/md_cg/bench_en_atoms_public.py +9 -0
- package/md_cg/bench_governance.py +348 -0
- package/md_cg/bench_lme_zh.py +16 -1
- package/md_cg/bench_locomo.py +2 -1
- package/md_cg/bench_locomo_zh.py +16 -1
- package/md_cg/bench_locomo_zh_public.py +4 -1
- package/md_cg/bench_longmem.py +2 -1
- package/md_cg/bench_membench.py +27 -1
- package/md_cg/bench_p0.py +4 -1
- package/md_cg/bench_progressive.py +13 -1
- package/md_cg/bench_role_views.py +238 -0
- package/md_cg/bench_task_ab.py +8 -1
- package/md_cg/bench_task_ab_llm.py +13 -1
- package/md_cg/bench_unified_en.py +6 -1
- package/md_cg/bench_zh_mad.py +20 -1
- package/md_cg/blindspot_tickets.py +123 -0
- package/md_cg/branches.py +12 -1
- package/md_cg/build_postings.py +73 -0
- package/md_cg/ccgc.py +67 -2
- package/md_cg/census.py +5 -1
- package/md_cg/chain.py +24 -3
- package/md_cg/codeindex.py +134 -17
- package/md_cg/coldverify.py +265 -0
- package/md_cg/comment_gate.py +338 -0
- package/md_cg/cond_compose.py +190 -0
- package/md_cg/cond_facts.py +155 -0
- package/md_cg/cond_template.json +107 -0
- package/md_cg/condition_anchor.py +143 -0
- package/md_cg/conformance.py +69 -4
- package/md_cg/consistency.py +24 -1
- package/md_cg/consolidate.py +53 -2
- package/md_cg/corpus.py +4 -0
- package/md_cg/crosscheck.py +42 -2
- package/md_cg/crypto.py +35 -1
- package/md_cg/d_meta.py +310 -0
- package/md_cg/datapath.py +201 -26
- package/md_cg/docindex.py +122 -1
- package/md_cg/eval_common.py +29 -1
- package/md_cg/evidence.py +27 -1
- package/md_cg/evolution.py +21 -1
- package/md_cg/export.py +11 -1
- package/md_cg/forgetting.py +23 -1
- package/md_cg/fsutil.py +18 -1
- package/md_cg/hotcache.py +214 -0
- package/md_cg/hyperedge.py +251 -0
- package/md_cg/identity.py +18 -1
- package/md_cg/insight.py +17 -1
- package/md_cg/lexicon/build_cedict_en_zh.py +9 -0
- package/md_cg/lexicon/build_standard_en.py +171 -168
- package/md_cg/lexicon/expand_en_zh.py +6 -0
- package/md_cg/lifecycle.py +12 -1
- package/md_cg/linkref.py +281 -0
- package/md_cg/links.py +29 -1
- package/md_cg/mcp_server.py +362 -43
- package/md_cg/md_whitebox.py +53 -1
- package/md_cg/mdcg.py +1003 -27
- package/md_cg/mdcos.py +558 -36
- package/md_cg/metacognition.py +37 -2
- package/md_cg/migrate.py +4 -0
- package/md_cg/migrate_aeis.py +221 -213
- package/md_cg/migrate_roleplay.py +8 -0
- package/md_cg/migrate_wisdom_graph.py +14 -1
- package/md_cg/mreview/__main__.py +3 -0
- package/md_cg/mreview/bundle.py +8 -0
- package/md_cg/mreview/candidates.py +9 -0
- package/md_cg/mreview/govern.py +21 -1
- package/md_cg/mreview/locate.py +34 -0
- package/md_cg/mreview/pipeline.py +29 -1
- package/md_cg/mreview/ruleset.py +16 -1
- package/md_cg/nodefile.py +233 -3
- package/md_cg/pooling.py +23 -1
- package/md_cg/postings.py +298 -0
- package/md_cg/predict.py +89 -9
- package/md_cg/progressive.py +3 -0
- package/md_cg/protect.py +14 -1
- package/md_cg/protocol.py +372 -0
- package/md_cg/provenance.py +262 -0
- package/md_cg/reach.py +453 -0
- package/md_cg/refindex.py +47 -2
- package/md_cg/refine.py +20 -1
- package/md_cg/roleviews.py +89 -0
- package/md_cg/routing.py +76 -0
- package/md_cg/scrub.py +63 -2
- package/md_cg/security.py +26 -1
- package/md_cg/self_state.py +64 -1
- package/md_cg/selfreport.py +151 -0
- package/md_cg/semantic/canonical.py +5 -0
- package/md_cg/semantic/en_normalizer.py +364 -355
- package/md_cg/semantic/zh_en_atoms.py +139 -136
- package/md_cg/signer.py +41 -1
- package/md_cg/sources.py +583 -547
- package/md_cg/statushdr.py +179 -0
- package/md_cg/stg.py +59 -18
- package/md_cg/subgraph.py +23 -0
- package/md_cg/sustain.py +56 -1
- package/md_cg/tasks.py +26 -2
- package/md_cg/test_autonomy.py +26 -0
- package/md_cg/test_bench_governance.py +102 -0
- package/md_cg/test_blindspot_tickets.py +166 -0
- package/md_cg/test_ccgc.py +10 -0
- package/md_cg/test_codeindex.py +338 -0
- package/md_cg/test_comment_gate.py +187 -0
- package/md_cg/test_cond_compose_anchors.py +76 -0
- package/md_cg/test_condition_anchor.py +82 -0
- package/md_cg/test_d_meta.py +412 -0
- package/md_cg/test_datapath_root.py +188 -0
- package/md_cg/test_gain_gate.py +47 -1
- package/md_cg/test_hot_cold.py +187 -0
- package/md_cg/test_hyperedge.py +245 -0
- package/md_cg/test_linkref.py +306 -0
- package/md_cg/test_md_access_parity.py +15 -3
- package/md_cg/test_mr_m1.py +108 -18
- package/md_cg/test_mr_m3.py +8 -1
- package/md_cg/test_p26_refindex.py +49 -20
- package/md_cg/test_p27_docindex.py +236 -2
- package/md_cg/test_p2_mcp.py +1 -1
- package/md_cg/test_p31_insight.py +24 -0
- package/md_cg/test_p44_md_whitebox.py +14 -1
- package/md_cg/test_protocol.py +243 -0
- package/md_cg/test_reach.py +378 -0
- package/md_cg/test_reach_keys.py +201 -0
- package/md_cg/test_reach_meta_exits.py +145 -0
- package/md_cg/test_read_clip.py +8 -4
- package/md_cg/test_retr_s1.py +340 -0
- package/md_cg/test_retr_s1b.py +209 -0
- package/md_cg/test_retr_s3.py +194 -0
- package/md_cg/test_retr_s4.py +163 -0
- package/md_cg/test_retr_s5.py +200 -0
- package/md_cg/test_retr_s6.py +157 -0
- package/md_cg/test_retr_s7.py +385 -0
- package/md_cg/test_retr_s8_time.py +316 -0
- package/md_cg/test_retr_s9_edges.py +286 -0
- package/md_cg/test_retr_s9_entity_ctx.py +175 -0
- package/md_cg/test_review_conformance.py +59 -2
- package/md_cg/test_role_views.py +354 -0
- package/md_cg/test_trust.py +361 -0
- package/md_cg/test_units_poll.py +71 -0
- package/md_cg/test_v14_fixes.py +397 -0
- package/md_cg/test_validity_filter.py +280 -0
- package/md_cg/test_wisdom_md_store.py +7 -3
- package/md_cg/test_writepipe.py +5 -1
- package/md_cg/theory.py +16 -1
- package/md_cg/tokens.py +40 -8
- package/md_cg/tool_face.py +13 -2
- package/md_cg/trust.py +943 -0
- package/md_cg/twophase.py +12 -1
- package/md_cg/units.py +129 -10
- package/md_cg/vision_evidence.py +24 -1
- package/md_cg/weights.py +24 -1
- package/md_cg/whitebox.py +32 -1
- package/md_cg/whitebox_kb/data/verify_cache.json +21210 -365
- package/md_cg/whitebox_kb/data/verify_savings.jsonl +5078 -0
- package/md_cg/whitebox_kb/wisdom/audit_log/chain_heat.json +10 -10
- package/md_cg/whitebox_kb/wisdom/code_compose.py +113 -6
- package/md_cg/whitebox_kb/wisdom/code_solidified.json +1 -1
- package/md_cg/whitebox_kb/wisdom/verifier.py +340 -55
- package/md_cg/whitebox_kb/wisdom/wisdom-book-cloud.db +0 -0
- package/md_cg/writelimit.py +18 -4
- package/md_cg/writepipe.py +178 -7
- package/package.json +2 -2
- package/skills/skills/designer-perspective/scripts/__pycache__/designer.cpython-310.pyc +0 -0
- package/skills/skills/designer-perspective/scripts/designer.py +17 -1
- package/skills/skills/designer-perspective/tests/selftest.py +3 -1
- package/src/hooks.ts +17 -21
- package/src/index.ts +33 -6
- package/src/lib/datapath.ts +211 -13
- package/src/lib/mdcg_client.ts +12 -8
- package/src/lib/mutual.ts +411 -411
- package/src/lib/token_store.ts +4 -5
- package/zcode/AGENTS.md +196 -195
- package/zcode/README.md +4 -0
- package/md_cg/whitebox_kb/wisdom/wisdom-book-cloud.db-shm +0 -0
- package/md_cg/whitebox_kb/wisdom/wisdom-book-cloud.db-wal +0 -0
package/md_cg/consolidate.py
CHANGED
|
@@ -111,6 +111,7 @@ _CCG_LINE_RE = re.compile(
|
|
|
111
111
|
r"^\s*#\s*(功能名|生效条件|子功能|执行|验证方式|不适用条件)\s*[::]")
|
|
112
112
|
|
|
113
113
|
|
|
114
|
+
# 生效条件:给定 content,返回剔除所有匹配 _CCG_LINE_RE 的行后以换行连接的非声明正文;content 为 None 时按空串处理。
|
|
114
115
|
def body_text(content: str) -> str:
|
|
115
116
|
"""剥掉 CCG 声明行后的正文——验证只认正文,不认已写下的声明。"""
|
|
116
117
|
return "\n".join(l for l in (content or "").split("\n")
|
|
@@ -171,6 +172,7 @@ VERIFY_PROMPT = (
|
|
|
171
172
|
|
|
172
173
|
# ---- LLM 侧(黑箱只在离线工序,产出候选)--------------------------------
|
|
173
174
|
|
|
175
|
+
# 生效条件:给定 role,按显式参数、角色环境变量、通用兜底依次解析并返回 (model, base, key);key 不落 DEEPSEEK_API_KEY 除非 role 为 REFLECT_ROLE。
|
|
174
176
|
def role_config(role: str, model: str = None, base: str = None,
|
|
175
177
|
key: str = None) -> tuple:
|
|
176
178
|
"""解析某角色的 (model, base, key):显式参数 > 角色环境变量 > 通用兜底。
|
|
@@ -197,6 +199,7 @@ def role_config(role: str, model: str = None, base: str = None,
|
|
|
197
199
|
return model, base, key
|
|
198
200
|
|
|
199
201
|
|
|
202
|
+
# 生效条件:给定 prompt 且 role 解析或通用兜底得到非空 key 时,向 base 的 /chat/completions 发 POST 并返回首个 choice 的 message.content;key 为空则抛 RuntimeError。
|
|
200
203
|
def http_llm(prompt: str, model: str = None, base: str = None, key: str = None,
|
|
201
204
|
role: str = None, timeout: int = 120, max_tokens: int = 1200) -> str:
|
|
202
205
|
"""标准库 HTTP 调 LLM(OpenAI 兼容 /chat/completions)。零第三方依赖。
|
|
@@ -234,6 +237,7 @@ def http_llm(prompt: str, model: str = None, base: str = None, key: str = None,
|
|
|
234
237
|
return data["choices"][0]["message"]["content"]
|
|
235
238
|
|
|
236
239
|
|
|
240
|
+
# 生效条件:给定 role,若 role_config 得到非空 key 则 GET base/models 并返回含 ok/model_available/models 的字典;无 key 或请求异常则返回 ok=False 及错误信息。
|
|
237
241
|
def probe_models(role: str, timeout: int = 20) -> dict:
|
|
238
242
|
"""零 token 探测:列出该角色网关的可用模型 id(GET /models)。"""
|
|
239
243
|
model, base, key = role_config(role)
|
|
@@ -258,6 +262,7 @@ def probe_models(role: str, timeout: int = 20) -> dict:
|
|
|
258
262
|
return res
|
|
259
263
|
|
|
260
264
|
|
|
265
|
+
# 生效条件:raw 为 None 或 strip 后不含 "{"(i<0)、或末个 "}" 的位置 j<=i 时返回 None;否则对 s 从首个 "{" 到末个 "}" 的切片 json.loads,成功则返回解析结果,抛 ValueError 时返回 None。
|
|
261
266
|
def _extract_json_obj(raw: str):
|
|
262
267
|
"""从 LLM 输出里抠出第一个 JSON 对象(容忍代码围栏 / 前后废话)。"""
|
|
263
268
|
s = (raw or "").strip()
|
|
@@ -270,6 +275,7 @@ def _extract_json_obj(raw: str):
|
|
|
270
275
|
return None
|
|
271
276
|
|
|
272
277
|
|
|
278
|
+
# 生效条件:v 为 None 返回 [];否则按 v 是 str 取 [v]、是 list/tuple 取逐项、其他取 [str(v)],逐项 strip 并去两端包裹标点后跳过空串及长度超 MAX_TERM_LEN 的项,未出现过的才 append,每次 append 后若 len(out) >= limit 即 break 返回 out(故 limit<=0 且存在有效项时仍返回 1 项)。
|
|
273
279
|
def _as_terms(v, limit: int = MAX_TERMS):
|
|
274
280
|
"""把 LLM 给的值规范成去重、限长的短语列表。"""
|
|
275
281
|
if v is None:
|
|
@@ -292,6 +298,7 @@ def _as_terms(v, limit: int = MAX_TERMS):
|
|
|
292
298
|
return out
|
|
293
299
|
|
|
294
300
|
|
|
301
|
+
# 生效条件:给定 raw,若 _extract_json_obj 解析出 dict,则按 CCG_FIELDS 与 FIELD_ALIASES 提取非空字段并规范为列表或单值返回字典;否则返回 {}。
|
|
295
302
|
def parse_candidate(raw: str) -> dict:
|
|
296
303
|
"""LLM 原始输出 → {字段: 列表/字符串};解析失败返回 {}。"""
|
|
297
304
|
obj = _extract_json_obj(raw)
|
|
@@ -317,6 +324,7 @@ def parse_candidate(raw: str) -> dict:
|
|
|
317
324
|
return out
|
|
318
325
|
|
|
319
326
|
|
|
327
|
+
# 生效条件:给定 raw,若解析出 dict,则按 CCG_FIELDS 提取 keep/drop/has_keep/reason 结构返回字典;否则返回 {}。
|
|
320
328
|
def parse_verdict(raw: str) -> dict:
|
|
321
329
|
"""验证单元输出 → {字段: {keep, drop, has_keep, reason}};解析失败返回 {}。"""
|
|
322
330
|
obj = _extract_json_obj(raw)
|
|
@@ -342,6 +350,7 @@ def parse_verdict(raw: str) -> dict:
|
|
|
342
350
|
return out
|
|
343
351
|
|
|
344
352
|
|
|
353
|
+
# 生效条件:给定 kept 与 verdict,若 verdict 为空则返回 (dict(kept), {});否则按 drop 与 has_keep 收窄候选并返回 (收窄后候选, 被剔除明细)。
|
|
345
354
|
def narrow_by_verdict(kept: dict, verdict: dict):
|
|
346
355
|
"""按验证单元裁决收窄候选——**只能否决,不能新增**。
|
|
347
356
|
|
|
@@ -377,6 +386,7 @@ def narrow_by_verdict(kept: dict, verdict: dict):
|
|
|
377
386
|
|
|
378
387
|
# ---- 确定性验证(零 LLM)-------------------------------------------------
|
|
379
388
|
|
|
389
|
+
# 生效条件:给定 term 与 body,若 term 的 bigram 序列非空则返回命中 bigram 数除以总 bigram 数,否则返回 0.0。
|
|
380
390
|
def grounding_score(term: str, body: str) -> float:
|
|
381
391
|
"""候选短语在正文里的字符级支撑度 = 命中 bigram 数 / 总 bigram 数。"""
|
|
382
392
|
bg = bigrams(term or "")
|
|
@@ -386,6 +396,7 @@ def grounding_score(term: str, body: str) -> float:
|
|
|
386
396
|
return hit / len(bg)
|
|
387
397
|
|
|
388
398
|
|
|
399
|
+
# 生效条件:给定 cand 与 body,按 thresholds 更新 DEFAULT_GROUNDING 后逐字段过滤候选,返回 (达标 kept, detail);不达标者丢弃。
|
|
389
400
|
def grounding_filter(cand: dict, body: str, thresholds: dict = None):
|
|
390
401
|
"""逐字段过滤候选:返回 (kept, detail)。不达标者丢弃(对应「不猜测」)。"""
|
|
391
402
|
th = dict(DEFAULT_GROUNDING)
|
|
@@ -405,6 +416,7 @@ def grounding_filter(cand: dict, body: str, thresholds: dict = None):
|
|
|
405
416
|
return kept, detail
|
|
406
417
|
|
|
407
418
|
|
|
419
|
+
# 生效条件:给定 pos_terms、neg_terms、body,返回含 pos_recall、neg_separated、no_conflict、ok 的回放判定字典。
|
|
408
420
|
def replay_check(pos_terms, neg_terms, body: str) -> dict:
|
|
409
421
|
"""回放生产判定:正例召回 + 负例剔除 + 无自相矛盾。
|
|
410
422
|
|
|
@@ -438,10 +450,12 @@ def replay_check(pos_terms, neg_terms, body: str) -> dict:
|
|
|
438
450
|
|
|
439
451
|
# ---- 写盘(固化)---------------------------------------------------------
|
|
440
452
|
|
|
453
|
+
# 生效条件:给定 content 与 field,当 content 含 "# field:" 或 "# field:" 时返回 True,否则 False。
|
|
441
454
|
def _has_ccg_line(content: str, field: str) -> bool:
|
|
442
455
|
return f"# {field}:" in (content or "") or f"# {field}:" in (content or "")
|
|
443
456
|
|
|
444
457
|
|
|
458
|
+
# 生效条件:给定 fm 与 content,对每个 CCG_FIELDS,若 frontmatter.comment 值非空或正文含对应 CCG 行则记入,返回已有字段字典。
|
|
445
459
|
def existing_fields(fm: dict, content: str) -> dict:
|
|
446
460
|
"""节点当前已有的四要素:正文 CCG 行 或 frontmatter.comment 任一存在即算有。"""
|
|
447
461
|
comment = (fm.get("state_attributes") or {}).get("comment") or {}
|
|
@@ -455,6 +469,7 @@ def existing_fields(fm: dict, content: str) -> dict:
|
|
|
455
469
|
return out
|
|
456
470
|
|
|
457
471
|
|
|
472
|
+
# 生效条件:给定 content、field、value,若已有 "# field:" 行则替换并返回新正文;否则插在 "# 功能名" 之后,若无则该行前置。
|
|
458
473
|
def _upsert_ccg_line(content: str, field: str, value: str) -> str:
|
|
459
474
|
"""在正文里写入/替换 `# <字段>:<值>`,优先插在「# 功能名」之后。"""
|
|
460
475
|
lines = (content or "").split("\n")
|
|
@@ -474,12 +489,14 @@ def _upsert_ccg_line(content: str, field: str, value: str) -> str:
|
|
|
474
489
|
return newline + "\n" + (content or "")
|
|
475
490
|
|
|
476
491
|
|
|
492
|
+
# 生效条件:给定 kept 字段字典,返回一句话规律字符串,列出缺失字段名并声明补齐后可路由。
|
|
477
493
|
def _evo_pattern(kept: dict) -> str:
|
|
478
494
|
"""规律(一句话):这一类节点反复缺的正是这批条件。"""
|
|
479
495
|
names = "、".join(kept.keys())
|
|
480
496
|
return f"缺「{names}」的节点条件不可判;补齐后四要素完整、可路由"
|
|
481
497
|
|
|
482
498
|
|
|
499
|
+
# 生效条件:给定 prov 字典,拼接 reflect/verify 模型、grounding、replay、verification_basis 中存在的证据项并返回。
|
|
483
500
|
def _evo_evidence(prov: dict) -> str:
|
|
484
501
|
"""证据:本次固化凭什么成立(模型 / 闸门 / 回放)。"""
|
|
485
502
|
parts = []
|
|
@@ -499,6 +516,7 @@ def _evo_evidence(prov: dict) -> str:
|
|
|
499
516
|
return " · ".join(parts)
|
|
500
517
|
|
|
501
518
|
|
|
519
|
+
# 生效条件:给定 cg、e、fm、content、kept、prov,将 kept 字段写入正文 CCG 行与 frontmatter.comment,不适用条件同步 non_applicable_conditions,并写 llm_consolidation 与演化记录,返回 None。
|
|
502
520
|
def _apply_node(cg, e, fm: dict, content: str, kept: dict, prov: dict,
|
|
503
521
|
basis: str = "", basis_enum: str = BASIS_ENUM_DEFAULT):
|
|
504
522
|
"""把通过验证的字段固化进 md:正文 CCG 行 + frontmatter.comment + 负条件 + provenance。
|
|
@@ -543,6 +561,7 @@ def _apply_node(cg, e, fm: dict, content: str, kept: dict, prov: dict,
|
|
|
543
561
|
before=before, after=evolution.state_of(cg, nid) or {})
|
|
544
562
|
|
|
545
563
|
|
|
564
|
+
# 生效条件:给定 root,扫描正排层节点并执行反思→白箱闸门→验证→固化,返回报表 rep;require_verify=True 且无 verify_fn 时全部 DEFER。
|
|
546
565
|
def consolidate(root: str, layer: str = None, limit: int = None, apply: bool = False,
|
|
547
566
|
overwrite: bool = False, llm_fn=None, reflect_fn=None,
|
|
548
567
|
verify_fn=None, reflect_model: str = "", verify_model: str = "",
|
|
@@ -569,6 +588,7 @@ def consolidate(root: str, layer: str = None, limit: int = None, apply: bool = F
|
|
|
569
588
|
"per_field": {f: 0 for f in CCG_FIELDS}, "verify_dropped": 0,
|
|
570
589
|
"verification_basis_missing": 0, "samples": []}
|
|
571
590
|
|
|
591
|
+
# 生效条件:以 reason 为键写入闭包 rep["reasons"],计数按 rep["reasons"].get(reason, 0) + 1 递增(键缺失从 0 起算),无返回值。
|
|
572
592
|
def _bump(reason):
|
|
573
593
|
rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1
|
|
574
594
|
|
|
@@ -695,6 +715,7 @@ def consolidate(root: str, layer: str = None, limit: int = None, apply: bool = F
|
|
|
695
715
|
return rep
|
|
696
716
|
|
|
697
717
|
|
|
718
|
+
# 生效条件:给定 root 与 basis,对缺 "# 验证方式" 行的节点补写验证方式并在需要时写入 basis_enum,返回统计 rep。
|
|
698
719
|
def fill_verification_basis(root: str, basis: str, layer: str = None,
|
|
699
720
|
limit: int = None, apply: bool = False,
|
|
700
721
|
basis_enum: str = BASIS_ENUM_DEFAULT) -> dict:
|
|
@@ -755,6 +776,7 @@ MAINTAIN_LOG = "_maintain.jsonl"
|
|
|
755
776
|
CCG_REQUIRED = ("生效条件", "子功能", "执行", "不适用条件")
|
|
756
777
|
|
|
757
778
|
|
|
779
|
+
# 生效条件:给定 cg、nid、e、fm、content、target_layer,把节点写入目标层(必要时按 routing 分桶)并删除旧路径,返回新相对路径与 bucket。
|
|
758
780
|
def _relocate_layer(cg, nid, e, fm, content, target_layer):
|
|
759
781
|
"""把节点正文迁到目标层的正确目录(含分桶),删除旧文件。返回新相对路径。"""
|
|
760
782
|
d = os.path.join(cg.root, target_layer)
|
|
@@ -773,6 +795,7 @@ def _relocate_layer(cg, nid, e, fm, content, target_layer):
|
|
|
773
795
|
"bucket": bucket}
|
|
774
796
|
|
|
775
797
|
|
|
798
|
+
# 生效条件:给定 root,把 source_layer 中命中次数不小于 min_merge 或 importance 不小于 min_importance 且条件完整的节点提升到 target_layer,返回统计 rep。
|
|
776
799
|
def promote_memories(root, source_layer="contextual", target_layer="knowledge",
|
|
777
800
|
min_merge=2, min_importance=0.6, require_conditions=True,
|
|
778
801
|
limit=None, apply=False, actor="maintain") -> dict:
|
|
@@ -844,6 +867,7 @@ def promote_memories(root, source_layer="contextual", target_layer="knowledge",
|
|
|
844
867
|
return rep
|
|
845
868
|
|
|
846
869
|
|
|
870
|
+
# 生效条件:给定 root,按 _maintain.jsonl 中 action=promote 记录(可再按 node_ids/batch 过滤)把节点迁回原层,成功返回 ok=True/reverted/ids,无记录返回 ok=False/error=no_records。
|
|
847
871
|
def rollback_promotion(root, node_ids=None, batch=None, actor="maintain") -> dict:
|
|
848
872
|
"""回滚情境提升:把 promoted_from 层迁回,并记一条演化条目。"""
|
|
849
873
|
cg = MdCGOS(root)
|
|
@@ -902,11 +926,13 @@ def rollback_promotion(root, node_ids=None, batch=None, actor="maintain") -> dic
|
|
|
902
926
|
CONTEXTUALIZE_REASON_DEFAULT = "情境性内容归位(批次流水账 / 感知产物)"
|
|
903
927
|
|
|
904
928
|
|
|
929
|
+
# 生效条件:给定 e,返回 e.id 字符串,若缺 id 则回落到 basename(e.path) 去掉 .md。
|
|
905
930
|
def _entry_id(e) -> str:
|
|
906
931
|
"""索引条目取 id:优先 `id` 字段,回落到文件名(索引不保证带 id)。"""
|
|
907
932
|
return str(e.get("id") or os.path.basename(e.get("path") or "")[:-3])
|
|
908
933
|
|
|
909
934
|
|
|
935
|
+
# 生效条件:给定 root 与 base,若 base 不在维护日志已用批次中则返回 base,否则返回 base.n 且 n 为最小未用序号。
|
|
910
936
|
def _unique_batch(root, base) -> str:
|
|
911
937
|
"""批次号去重:**同一秒内的两次调用不得共用批次号**。
|
|
912
938
|
|
|
@@ -921,6 +947,7 @@ def _unique_batch(root, base) -> str:
|
|
|
921
947
|
return f"{base}.{n}"
|
|
922
948
|
|
|
923
949
|
|
|
950
|
+
# 生效条件:给定 root 且 prefixes 或 node_ids 至少一个非空,把 source_layer 中匹配的节点迁到 target_layer,返回统计 rep;两者皆空则抛 ValueError。
|
|
924
951
|
def contextualize_prefixes(root, prefixes=None, node_ids=None,
|
|
925
952
|
source_layer="knowledge", target_layer="contextual",
|
|
926
953
|
reason="", limit=None, apply=False,
|
|
@@ -938,6 +965,7 @@ def contextualize_prefixes(root, prefixes=None, node_ids=None,
|
|
|
938
965
|
"(如 ['note_','imgpart_']),拒绝对整层无差别改写")
|
|
939
966
|
cg = MdCGOS(root)
|
|
940
967
|
|
|
968
|
+
# 生效条件:e 经 _entry_id 得到 nid 后,若闭包 ids 不为 None 则返回 nid in ids 的真假,若 ids 为 None 则返回 nid.startswith(pref) 的真假。
|
|
941
969
|
def _hit(e) -> bool:
|
|
942
970
|
nid = _entry_id(e)
|
|
943
971
|
return nid in ids if ids is not None else nid.startswith(pref)
|
|
@@ -999,6 +1027,7 @@ def contextualize_prefixes(root, prefixes=None, node_ids=None,
|
|
|
999
1027
|
return rep
|
|
1000
1028
|
|
|
1001
1029
|
|
|
1030
|
+
# 生效条件:给定 root,按 _maintain.jsonl 中 action=contextualize 记录(可再按 node_ids/batch 过滤)把节点迁回原层,成功返回 ok=True/reverted/ids,无记录返回 ok=False/error=no_records。
|
|
1002
1031
|
def rollback_contextualize(root, node_ids=None, batch=None, actor="maintain") -> dict:
|
|
1003
1032
|
"""回滚层归位:按 `_maintain.jsonl` 的 contextualize 记录把节点迁回原层。"""
|
|
1004
1033
|
cg = MdCGOS(root)
|
|
@@ -1040,6 +1069,7 @@ def rollback_contextualize(root, node_ids=None, batch=None, actor="maintain") ->
|
|
|
1040
1069
|
return {"ok": True, "reverted": reverted, "ids": ids}
|
|
1041
1070
|
|
|
1042
1071
|
|
|
1072
|
+
# 生效条件:对 _read_maintain(root) 中 action 为 "contextualize" 或 "contextualize_rollback" 的记录,取 recs[-(int(limit) or 50):] 作为 records 返回 {'ok': True, ...}——仅当 int(limit) 成功且为 0 时回落 50,limit 为 None/""/[] 等无法 int() 的值会先抛 TypeError/ValueError。
|
|
1043
1073
|
def contextualize_history(root, limit=50):
|
|
1044
1074
|
"""层归位的批次记录(只读)。"""
|
|
1045
1075
|
recs = [r for r in _read_maintain(root)
|
|
@@ -1047,6 +1077,7 @@ def contextualize_history(root, limit=50):
|
|
|
1047
1077
|
return {"ok": True, "records": recs[-(int(limit) or 50):]}
|
|
1048
1078
|
|
|
1049
1079
|
|
|
1080
|
+
# 生效条件:给定 root,读取 root 下 MAINTAIN_LOG 的 JSONL 并返回记录列表。
|
|
1050
1081
|
def _read_maintain(root):
|
|
1051
1082
|
from .fsutil import read_jsonl
|
|
1052
1083
|
return list(read_jsonl(os.path.join(root, MAINTAIN_LOG)))
|
|
@@ -1076,10 +1107,15 @@ CONCEPT_MEMBER_REL = "instance_of" # member → concept(inferred)
|
|
|
1076
1107
|
CONCEPT_PREFIX = "concept_"
|
|
1077
1108
|
CONCEPT_IMPORTANCE = 0.5
|
|
1078
1109
|
CONCEPT_TAGS = ("concept", "induced")
|
|
1110
|
+
# 巩固留痕字段(2026-09-19 阶段一):**字段名真源在 md_cg/nodefile.py**,
|
|
1111
|
+
# 本处只做短别名引用(非复制),与 `nodefile.VALID_FROM_FIELD` 的登记纪律同构。
|
|
1112
|
+
CONSOLIDATED_AT_FIELD = nodefile.CONSOLIDATED_AT_FIELD
|
|
1113
|
+
CONSOLIDATED_INTO_FIELD = nodefile.CONSOLIDATED_INTO_FIELD
|
|
1079
1114
|
# 归纳候选排除:受保护节点,以及洞察/场景/前馈/概念等派生物(避免自我进食)
|
|
1080
1115
|
INDUCE_SKIP_TAGS = ("insight", "scene", "reconstructed", "gap_hint", "concept")
|
|
1081
1116
|
|
|
1082
1117
|
|
|
1118
|
+
# 生效条件:给定 members,返回 CONCEPT_PREFIX 拼接排序后成员串的 SHA1 前 10 位。
|
|
1083
1119
|
def _concept_id(members):
|
|
1084
1120
|
"""概念节点 id:由成员清单派生,保证「同成员 ⇒ 同 id」的幂等性。"""
|
|
1085
1121
|
h = hashlib.sha1("|".join(sorted(str(m) for m in members))
|
|
@@ -1087,12 +1123,14 @@ def _concept_id(members):
|
|
|
1087
1123
|
return CONCEPT_PREFIX + h[:10]
|
|
1088
1124
|
|
|
1089
1125
|
|
|
1126
|
+
# 生效条件:给定 a 与 b,若任一为空集则返回 0.0,否则返回交集大小除以并集大小。
|
|
1090
1127
|
def _jaccard(a, b):
|
|
1091
1128
|
if not a or not b:
|
|
1092
1129
|
return 0.0
|
|
1093
1130
|
return len(a & b) / float(len(a | b))
|
|
1094
1131
|
|
|
1095
1132
|
|
|
1133
|
+
# 生效条件:给定 term_sets,返回出现次数不小于 max(2, ceil(min_share * len(term_sets))) 的词面排序列表;空输入返回 []。
|
|
1096
1134
|
def _common_terms(term_sets, min_share=0.6):
|
|
1097
1135
|
"""出现在 ≥ min_share 比例成员中的词面(共同条件);少于 2 个成员共享不算。"""
|
|
1098
1136
|
if not term_sets:
|
|
@@ -1105,6 +1143,7 @@ def _common_terms(term_sets, min_share=0.6):
|
|
|
1105
1143
|
return sorted(t for t, c in cnt.items() if c >= need)
|
|
1106
1144
|
|
|
1107
1145
|
|
|
1146
|
+
# 生效条件:给定 term_sets,按集合排序去重拼接后返回前 limit(默认 INDUCE_MAX_TERMS)个词面。
|
|
1108
1147
|
def _union_terms(term_sets, limit=INDUCE_MAX_TERMS):
|
|
1109
1148
|
seen = []
|
|
1110
1149
|
for s in term_sets:
|
|
@@ -1114,6 +1153,7 @@ def _union_terms(term_sets, limit=INDUCE_MAX_TERMS):
|
|
|
1114
1153
|
return seen[:limit]
|
|
1115
1154
|
|
|
1116
1155
|
|
|
1156
|
+
# 生效条件:给定 cg、cid、members、reason、actor、batch,为概念节点与成员节点写对称 inferred 边(已存在则跳过),返回含 concept 与 members 的字典。
|
|
1117
1157
|
def _link_concept(cg, cid, members, reason, actor, batch):
|
|
1118
1158
|
"""写概念↔成员对称 inferred 边(幂等:已存在则不重复写)。"""
|
|
1119
1159
|
nodes = (getattr(cg, "index", None) or {}).get("nodes") or {}
|
|
@@ -1150,6 +1190,12 @@ def _link_concept(cg, cid, members, reason, actor, batch):
|
|
|
1150
1190
|
"reason": reason, "created_at": time.time(),
|
|
1151
1191
|
"confidence": 0.5, "verified": 0, "evidence": "inferred"})
|
|
1152
1192
|
fm["edges"] = edges
|
|
1193
|
+
# 巩固留痕(2026-09-19 阶段一):`consolidated_into` 为**规范名**,
|
|
1194
|
+
# `induced_concept` 保留为历史别名(既有读取面零破坏);`consolidated_at`
|
|
1195
|
+
# 补齐**成员侧**巩固时刻——此前只有概念侧 `induced_at`,成员侧无从判定
|
|
1196
|
+
# 「何时被并进去」,故「合并后前身可定位」只在概念侧半成立。
|
|
1197
|
+
fm[CONSOLIDATED_INTO_FIELD] = cid
|
|
1198
|
+
fm[CONSOLIDATED_AT_FIELD] = time.time()
|
|
1153
1199
|
fm["induced_concept"] = cid
|
|
1154
1200
|
ent = nodes.get(m) or {}
|
|
1155
1201
|
cg._write_node(m, os.path.join(cg.root, ent.get("path") or f"{m}.md"),
|
|
@@ -1160,6 +1206,7 @@ def _link_concept(cg, cid, members, reason, actor, batch):
|
|
|
1160
1206
|
return out
|
|
1161
1207
|
|
|
1162
1208
|
|
|
1209
|
+
# 生效条件:给定 members、common_pos、neg_union,返回标注 inferred 的概念节点正文,含功能名、生效条件、子功能、执行、验证方式、不适用条件。
|
|
1163
1210
|
def _concept_payload(members, common_pos, neg_union):
|
|
1164
1211
|
"""概念节点正文:把成员的共性条件抽象为可追溯的知识条目(显式标注 inferred)。"""
|
|
1165
1212
|
label = "、".join(common_pos[:INDUCE_MAX_TERMS])
|
|
@@ -1176,6 +1223,7 @@ def _concept_payload(members, common_pos, neg_union):
|
|
|
1176
1223
|
)
|
|
1177
1224
|
|
|
1178
1225
|
|
|
1226
|
+
# 生效条件:给定 cg_or_root,从 source_layer 聚类归纳为 target_layer 概念节点,apply=True 才写盘并返回统计 rep。
|
|
1179
1227
|
def induce_memories(cg_or_root, source_layer="contextual", target_layer="knowledge",
|
|
1180
1228
|
min_cluster=INDUCE_MIN_CLUSTER, min_jaccard=INDUCE_MIN_JACCARD,
|
|
1181
1229
|
max_nodes=INDUCE_MAX_NODES, require_conditions=True,
|
|
@@ -1271,9 +1319,11 @@ def induce_memories(cg_or_root, source_layer="contextual", target_layer="knowled
|
|
|
1271
1319
|
continue
|
|
1272
1320
|
content = _concept_payload(p["members"], p["common_conditions"],
|
|
1273
1321
|
p["non_applicable"])
|
|
1322
|
+
_consolidated_at = time.time() # 概念形成时刻 = 巩固时刻(单一取值,禁两处取时)
|
|
1274
1323
|
cg.add(cid, content, layer=target_layer, tags=list(CONCEPT_TAGS),
|
|
1275
1324
|
importance=CONCEPT_IMPORTANCE, verification_basis="other",
|
|
1276
|
-
induced_from=list(p["members"]), induced_at=
|
|
1325
|
+
induced_from=list(p["members"]), induced_at=_consolidated_at,
|
|
1326
|
+
consolidated_at=_consolidated_at,
|
|
1277
1327
|
induction={"method": "bigram_jaccard", "min_jaccard": float(min_jaccard),
|
|
1278
1328
|
"common_conditions": p["common_conditions"], "batch": batch,
|
|
1279
1329
|
"actor": actor, "evidence": "inferred"},
|
|
@@ -1294,6 +1344,7 @@ def induce_memories(cg_or_root, source_layer="contextual", target_layer="knowled
|
|
|
1294
1344
|
|
|
1295
1345
|
# ---- CLI ----------------------------------------------------------------
|
|
1296
1346
|
|
|
1347
|
+
# 生效条件:不适用(无必需形参与模块级常量)
|
|
1297
1348
|
def _cli(argv=None) -> int:
|
|
1298
1349
|
ap = argparse.ArgumentParser(
|
|
1299
1350
|
description="md_cg 离线固化:反思单元(LLM)产出候选 → 白箱闸门 → "
|
|
@@ -1386,4 +1437,4 @@ def _cli(argv=None) -> int:
|
|
|
1386
1437
|
|
|
1387
1438
|
|
|
1388
1439
|
if __name__ == "__main__":
|
|
1389
|
-
raise SystemExit(_cli())
|
|
1440
|
+
raise SystemExit(_cli())
|
package/md_cg/corpus.py
CHANGED
|
@@ -40,6 +40,7 @@ MARKS = ("# 功能名:", "# 生效条件:", "# 子功能:", "# 执行:",
|
|
|
40
40
|
EXPECTED_NODES = sum(len(points) for _dom, points in DOMAINS) * len(ASPECTS)
|
|
41
41
|
|
|
42
42
|
|
|
43
|
+
# 生效条件:给定 dom、point、aspect 且 marks 取默认 True(真值)时,返回带「# 功能名/生效条件/子功能/执行/验证方式/不适用条件」MARKS 行的完整正文;marks 为假值时返回不含 MARKS 行、仅叙述「{point} 是 {dom} 领域的知识点」的正文。
|
|
43
44
|
def body(dom, point, aspect, marks=True):
|
|
44
45
|
"""节点正文。marks=True 给完整 CCG 五要素,否则给纯叙述正文(无 MARKS 行)。"""
|
|
45
46
|
if marks:
|
|
@@ -56,10 +57,12 @@ def body(dom, point, aspect, marks=True):
|
|
|
56
57
|
"常用于条件化检索与召回对照。\n")
|
|
57
58
|
|
|
58
59
|
|
|
60
|
+
# 生效条件:给定 di、pi、ai 时返回 f-string 拼成的字符串 `kp_{di:02d}_{pi:02d}_{ai}`(di 与 pi 之间由下划线分隔,di、pi 两位补零后接 ai)。
|
|
59
61
|
def node_id(di, pi, ai):
|
|
60
62
|
return f"kp_{di:02d}_{pi:02d}_{ai}"
|
|
61
63
|
|
|
62
64
|
|
|
65
|
+
# 生效条件:给定 cg 及默认 marks=True、layer="knowledge"、verification_basis="data" 时,按模块级 DOMAINS 与 ASPECTS 逐组合调用 cg.add(节点 id 取自 node_id,importance=0.4+0.1*ai,tags 为 domain:{dom},condition_space 含 observation_position/observation_tool/existence_constraint),每写一个 n 加一,最后 cg.flush() 并返回 n。
|
|
63
66
|
def seed(cg, marks=True, layer="knowledge", verification_basis="data"):
|
|
64
67
|
"""把整套语料写进 cg,返回写入节点数。幂等:节点 id 固定(同 id 原子覆盖)。"""
|
|
65
68
|
n = 0
|
|
@@ -81,6 +84,7 @@ def seed(cg, marks=True, layer="knowledge", verification_basis="data"):
|
|
|
81
84
|
return n
|
|
82
85
|
|
|
83
86
|
|
|
87
|
+
# 生效条件:给定 root 且 clean 为真、os.path.isdir(root) 成立时先 shutil.rmtree(root),随后无论 clean 取值都以 exist_ok=True 调用 os.makedirs(root)。
|
|
84
88
|
def reset_root(root, clean=False):
|
|
85
89
|
"""确保测试根存在:保证「重跑 ≡ 首跑」,且上一轮残留节点不会被当成先验。
|
|
86
90
|
|