@furongjun1999/dsh-memory 0.5.0 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +552 -465
- package/docs/README.md +1 -0
- package/docs/eval/bench_lingshu_self/bench_self.py +140 -0
- package/docs/eval/bench_lingshu_self/self_bench_result.json +404 -0
- package/docs/eval/bench_lingshu_self//347/201/265/346/236/242/350/207/252/345/272/223/347/253/257/345/210/260/347/253/257/346/243/200/347/264/242/345/256/236/346/265/213_v1.0.md +38 -0
- package/docs/eval//344/270/215/345/217/257/351/235/240/346/200/247/350/220/275/345/234/260_P0_v1.0.md +185 -0
- package/docs/eval//346/225/205/351/232/234/346/263/250/345/205/245/345/256/236/346/265/213_v1.0.md +422 -0
- package/docs/eval//347/253/257/345/210/260/347/253/257LoCoMoQA/345/220/214/345/217/243/345/276/204/345/257/271/347/205/247_v1.0.md +100 -0
- package/docs/eval//347/253/257/345/210/260/347/253/257/345/271/262/346/211/260/346/261/240/350/257/204/346/265/213_/347/241/256/345/256/232/346/200/247/350/243/201/345/206/263vsLLM_judge_v1.1.md +197 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v1.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v10.md +210 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v11.md +227 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v12.md +203 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v13.md +233 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v14.md +191 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v15.md +213 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v16.md +214 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v17.md +199 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v2.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v3.md +152 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v4.md +128 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v5.md +114 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v6.md +192 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v7.md +187 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v8.md +207 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v9.md +203 -0
- package/docs/hive//345/256/211/345/205/250/345/256/241/350/256/241/345/256/236/351/224/232_v0.1.md +202 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +19 -3
- package/docs/images/lingshu-moonlight-covenant-poster-preview.jpg +0 -0
- package/docs/images/lingshu-moonlight-covenant-poster.png +0 -0
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/theory//344/270/215/345/217/257/351/235/240/345/256/232/347/220/206/344/270/216/345/244/261/346/225/210/344/274/230/345/205/210/346/241/206/346/236/266_v0.3.md +365 -0
- package/docs//344/270/215/345/217/257/351/235/240/346/200/247/347/220/206/350/256/272_v0.1.md +343 -0
- package/lib/bridge.d.ts +9 -0
- package/lib/bridge.js +35 -0
- package/lib/index.js +7 -1
- package/lib/lib/roleplay_web.js +530 -443
- package/lib/lib/token_store.d.ts +7 -1
- package/lib/lib/token_store.js +12 -3
- package/md_cg/audit.py +12 -1
- package/md_cg/backfill.py +16 -15
- package/md_cg/backfill_bucket_zh.py +35 -0
- package/md_cg/bench_e2e_judge.py +532 -0
- package/md_cg/bench_e2e_locomo_qa.py +368 -0
- package/md_cg/bench_e2e_qa.py +256 -0
- package/md_cg/branches.py +18 -2
- package/md_cg/ccgc.py +3 -2
- package/md_cg/chain.py +19 -4
- package/md_cg/consolidate.py +7 -6
- package/md_cg/crosscheck.py +4 -3
- package/md_cg/crypto.py +439 -437
- package/md_cg/datapath.py +395 -335
- package/md_cg/evidence.py +582 -580
- package/md_cg/export.py +3 -1
- package/md_cg/forgetting.py +2 -2
- package/md_cg/fsutil.py +48 -0
- package/md_cg/hotcache.py +255 -238
- package/md_cg/interop.py +161 -22
- package/md_cg/judgment_manifest.py +177 -0
- package/md_cg/links.py +655 -622
- package/md_cg/mcp_server.py +280 -29
- package/md_cg/mdcg.py +616 -125
- package/md_cg/mdcos.py +430 -66
- package/md_cg/mreview/govern.py +5 -4
- package/md_cg/postings.py +4 -2
- package/md_cg/readcache.py +76 -18
- package/md_cg/reconcile.py +228 -0
- package/md_cg/review_cli.py +170 -0
- package/md_cg/routing.py +28 -0
- package/md_cg/run_tests.py +211 -0
- package/md_cg/scrub.py +862 -852
- package/md_cg/security.py +385 -275
- package/md_cg/selfreport.py +3 -2
- package/md_cg/signer.py +565 -562
- package/md_cg/sources.py +3 -2
- package/md_cg/stg.py +6 -0
- package/md_cg/sustain.py +1168 -1138
- package/md_cg/test_access_hints.py +147 -0
- package/md_cg/test_branch_discard_tombstone.py +136 -0
- package/md_cg/test_branches.py +259 -249
- package/md_cg/test_chain_read_isolate.py +168 -0
- package/md_cg/test_datapath_device_name.py +203 -0
- package/md_cg/test_emit_negtail_cache.py +156 -0
- package/md_cg/test_en_pipeline.py +186 -166
- package/md_cg/test_govern_directread.py +421 -0
- package/md_cg/test_i32_hotcache_env_key.py +218 -0
- package/md_cg/test_identity_attribution.py +228 -147
- package/md_cg/test_index_durability.py +238 -224
- package/md_cg/test_interop.py +4 -2
- package/md_cg/test_interop_judgment.py +228 -0
- package/md_cg/test_issue39_utf8_stdio.py +273 -0
- package/md_cg/test_links_concurrent_write.py +188 -0
- package/md_cg/test_merge_upsert.py +168 -0
- package/md_cg/test_n123_derive_expiry_chain.py +205 -0
- package/md_cg/test_n130_verify_falsified_protect.py +185 -0
- package/md_cg/test_n131_merge_gate.py +205 -0
- package/md_cg/test_p1x_ref_root.py +160 -0
- package/md_cg/test_p27_docindex.py +774 -765
- package/md_cg/test_p2_mcp.py +3 -0
- package/md_cg/test_p32_backfill.py +304 -298
- package/md_cg/test_p39_verify_flow.py +90 -50
- package/md_cg/test_p47_session_view.py +60 -25
- package/md_cg/test_propose_tail_index.py +157 -0
- package/md_cg/test_read_scope_b27.py +277 -0
- package/md_cg/test_readcache_default_on.py +168 -0
- package/md_cg/test_readcache_precise_inval.py +270 -0
- package/md_cg/test_readcache_prodpath.py +55 -7
- package/md_cg/test_reconcile_v0.py +294 -0
- package/md_cg/test_retr_s1.py +6 -2
- package/md_cg/test_retr_s1b.py +67 -0
- package/md_cg/test_retr_s7.py +8 -0
- package/md_cg/test_retr_s9_entity_ctx.py +181 -175
- package/md_cg/test_retr_score_once.py +208 -0
- package/md_cg/test_review_onepass.py +170 -0
- package/md_cg/test_rrf_graph_seed_cache.py +154 -0
- package/md_cg/test_security_audit.py +155 -0
- package/md_cg/test_security_audit_b26.py +161 -0
- package/md_cg/test_security_audit_v21.py +250 -0
- package/md_cg/test_semantic_canonical.py +255 -241
- package/md_cg/test_session_isolation.py +168 -0
- package/md_cg/test_snapshot_autoclose.py +187 -0
- package/md_cg/test_tail_watermark_race.py +208 -0
- package/md_cg/test_tenant_env_override_warn.py +139 -0
- package/md_cg/test_tenant_registry_corrupt_warn.py +151 -0
- package/md_cg/test_v14_fixes.py +415 -397
- package/md_cg/test_verify_dirty_reconcile.py +157 -0
- package/md_cg/theory.py +276 -273
- package/md_cg/tokens.py +734 -677
- package/md_cg/units.py +3 -2
- package/md_cg/vision_evidence.py +4 -3
- package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
- package/md_cg/whitebox_kb/wisdom/multilang_ir.py +128 -124
- package/md_cg/writepipe.py +554 -550
- package/package.json +6 -2
- package/src/bridge.ts +434 -401
- package/src/index.ts +526 -518
- package/src/lib/roleplay_web.ts +1019 -932
- package/src/lib/token_store.ts +202 -192
|
@@ -1,765 +1,774 @@
|
|
|
1
|
-
# -*- coding: utf-8 -*-
|
|
2
|
-
"""条件文档图端到端测试(P27 · 认知图的文档面)。
|
|
3
|
-
|
|
4
|
-
R2 改造的验收(对照 docs/mdcg/认知图_索引与工程规范化_计划_v0.1.md §六 R2、§七):
|
|
5
|
-
|
|
6
|
-
① 切块纪律:只切 level<=3;直接正文 <200 字且无子节的小节**合并进父节**
|
|
7
|
-
(不单独建节点,其文字进父节摘要)。既不过细,也不丢内容。
|
|
8
|
-
② md 解析硬约束:
|
|
9
|
-
· **围栏代码块内的 `#` 不是标题**(docs/ 里全是 python/shell 片段,不跟踪
|
|
10
|
-
围栏就会切出假标题、把一节切碎);
|
|
11
|
-
· **开头 YAML frontmatter 不被当正文索引**;**正文里的 `---` 不污染
|
|
12
|
-
frontmatter**(nodefile 既有纪律,这里验证不回归)。
|
|
13
|
-
③ 渲染即 CCG:`docindex.render` 必须产 CCG 6 行,否则文档节点会像改造前的
|
|
14
|
-
codeindex 一样「存得进、判不了、检索不到」(恒定 BLINDSPOT)。
|
|
15
|
-
④ doc_ref + op=ref:只存标题与摘要、**不存全文**;正文按 doc_ref 回读,
|
|
16
|
-
与 code 节点共用同一 `region_hash` → hash_match 可检漂移、可重跑恢复。
|
|
17
|
-
⑤ §1.3-3 裁定落地:layer 默认 knowledge;密级默认 internal **显式写入**
|
|
18
|
-
frontmatter,路径段命中私有提示再保守降为 private;调用方可显式覆盖。
|
|
19
|
-
⑥ 真实 docs/:章节可定位(行号与源文件一致),能检索到并按 CCG 判 ACCEPT。
|
|
20
|
-
⑦ 不静默:truncated / skipped_suffixes / skipped_dirs 显式上报;排除为**追加**
|
|
21
|
-
(只增不减,内置 .git/.venv/node_modules 不可被关闭);只读契约(源 mtime 不变);
|
|
22
|
-
幂等(重跑节点数不变);`index_doc` 进 ALL_OPS 且与工具 schema 一致。
|
|
23
|
-
⑧ fence 往返列六件套的形状与边界:BINDING_FIELDS / binding_of / binding_key
|
|
24
|
-
(**不含行位**)/ binding_slug / validate_binding / binding_drift / locate。
|
|
25
|
-
为什么必须钉死:列形状漂移会让对账器把坏列当好消息(漏报);键若悄悄带上
|
|
26
|
-
行位,真源上方插一行就会让整库失配——两者都是静默退化,只有断言看得见。
|
|
27
|
-
⑨ 逐字节回放回归:对**真实索引出的卡**断言 `render(条目) + "\n" == 卡片正文`。
|
|
28
|
-
⑧ 只证「列可定位」;只有逐字节相等才证明卡片是真源条目的**派生物**而非另一份
|
|
29
|
-
副本(渲染漂移或人工改写时列全然不变,对账器看不出来)。外部历史库另段抽样
|
|
30
|
-
(设 MDCG_ROOT 则跑,未设如实 SKIP 不虚报通过)。
|
|
31
|
-
|
|
32
|
-
运行:python -m md_cg.test_p27_docindex
|
|
33
|
-
设 MDCG_ROOT=<认知图库根>
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
import
|
|
38
|
-
|
|
39
|
-
import
|
|
40
|
-
import
|
|
41
|
-
import
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
from .
|
|
46
|
-
from .
|
|
47
|
-
from .
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
#
|
|
77
|
-
#
|
|
78
|
-
#
|
|
79
|
-
#
|
|
80
|
-
#
|
|
81
|
-
#
|
|
82
|
-
#
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
if
|
|
93
|
-
continue
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
picks
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
#
|
|
107
|
-
#
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
except
|
|
115
|
-
cache[ckey] = (
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
"
|
|
148
|
-
"
|
|
149
|
-
"",
|
|
150
|
-
"
|
|
151
|
-
"",
|
|
152
|
-
|
|
153
|
-
"",
|
|
154
|
-
|
|
155
|
-
"",
|
|
156
|
-
"
|
|
157
|
-
"",
|
|
158
|
-
"
|
|
159
|
-
"
|
|
160
|
-
"|
|
|
161
|
-
"",
|
|
162
|
-
"
|
|
163
|
-
"",
|
|
164
|
-
"
|
|
165
|
-
"",
|
|
166
|
-
"
|
|
167
|
-
"",
|
|
168
|
-
"
|
|
169
|
-
"
|
|
170
|
-
"
|
|
171
|
-
"
|
|
172
|
-
"
|
|
173
|
-
"
|
|
174
|
-
"",
|
|
175
|
-
"
|
|
176
|
-
"",
|
|
177
|
-
"
|
|
178
|
-
"",
|
|
179
|
-
"
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
"
|
|
187
|
-
"",
|
|
188
|
-
"
|
|
189
|
-
"",
|
|
190
|
-
"
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
print("=" * 68)
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
with open(os.path.join(fx, "
|
|
215
|
-
f.write(
|
|
216
|
-
with open(os.path.join(fx, "
|
|
217
|
-
f.write(
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
unknown =
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
check("
|
|
246
|
-
|
|
247
|
-
check("
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
find(items,
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
s9[
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
check("
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
#
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
check("
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
check("
|
|
309
|
-
|
|
310
|
-
check("
|
|
311
|
-
|
|
312
|
-
check("
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
check("guide
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
check("
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
g_fm.get("condition_space")
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
dr.get("
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
check("
|
|
405
|
-
(tr.get("
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
#
|
|
421
|
-
|
|
422
|
-
#
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
#
|
|
435
|
-
#
|
|
436
|
-
#
|
|
437
|
-
#
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
check("
|
|
466
|
-
s9r
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
#
|
|
516
|
-
#
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
#
|
|
604
|
-
#
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
check("binding_of
|
|
628
|
-
docindex.binding_of(
|
|
629
|
-
|
|
630
|
-
|
|
631
|
-
and docindex.binding_of(
|
|
632
|
-
|
|
633
|
-
docindex.
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
docindex.validate_binding(
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
("
|
|
644
|
-
"
|
|
645
|
-
("类型错
|
|
646
|
-
|
|
647
|
-
("
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
docindex.binding_key(b0))
|
|
660
|
-
|
|
661
|
-
docindex.
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
docindex.binding_drift(b0,
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
docindex.
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
docindex.
|
|
676
|
-
|
|
677
|
-
docindex.binding_slug(b0)
|
|
678
|
-
|
|
679
|
-
docindex.
|
|
680
|
-
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
docindex.locate(li,
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
docindex.locate(li,
|
|
687
|
-
|
|
688
|
-
|
|
689
|
-
and docindex.locate(li,
|
|
690
|
-
and docindex.locate(
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
#
|
|
695
|
-
#
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
rp[
|
|
704
|
-
check("
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
rp["
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
print(f"
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
739
|
-
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
|
|
743
|
-
hp[
|
|
744
|
-
f"
|
|
745
|
-
check("【12b
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
f"
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""条件文档图端到端测试(P27 · 认知图的文档面)。
|
|
3
|
+
|
|
4
|
+
R2 改造的验收(对照 docs/mdcg/认知图_索引与工程规范化_计划_v0.1.md §六 R2、§七):
|
|
5
|
+
|
|
6
|
+
① 切块纪律:只切 level<=3;直接正文 <200 字且无子节的小节**合并进父节**
|
|
7
|
+
(不单独建节点,其文字进父节摘要)。既不过细,也不丢内容。
|
|
8
|
+
② md 解析硬约束:
|
|
9
|
+
· **围栏代码块内的 `#` 不是标题**(docs/ 里全是 python/shell 片段,不跟踪
|
|
10
|
+
围栏就会切出假标题、把一节切碎);
|
|
11
|
+
· **开头 YAML frontmatter 不被当正文索引**;**正文里的 `---` 不污染
|
|
12
|
+
frontmatter**(nodefile 既有纪律,这里验证不回归)。
|
|
13
|
+
③ 渲染即 CCG:`docindex.render` 必须产 CCG 6 行,否则文档节点会像改造前的
|
|
14
|
+
codeindex 一样「存得进、判不了、检索不到」(恒定 BLINDSPOT)。
|
|
15
|
+
④ doc_ref + op=ref:只存标题与摘要、**不存全文**;正文按 doc_ref 回读,
|
|
16
|
+
与 code 节点共用同一 `region_hash` → hash_match 可检漂移、可重跑恢复。
|
|
17
|
+
⑤ §1.3-3 裁定落地:layer 默认 knowledge;密级默认 internal **显式写入**
|
|
18
|
+
frontmatter,路径段命中私有提示再保守降为 private;调用方可显式覆盖。
|
|
19
|
+
⑥ 真实 docs/:章节可定位(行号与源文件一致),能检索到并按 CCG 判 ACCEPT。
|
|
20
|
+
⑦ 不静默:truncated / skipped_suffixes / skipped_dirs 显式上报;排除为**追加**
|
|
21
|
+
(只增不减,内置 .git/.venv/node_modules 不可被关闭);只读契约(源 mtime 不变);
|
|
22
|
+
幂等(重跑节点数不变);`index_doc` 进 ALL_OPS 且与工具 schema 一致。
|
|
23
|
+
⑧ fence 往返列六件套的形状与边界:BINDING_FIELDS / binding_of / binding_key
|
|
24
|
+
(**不含行位**)/ binding_slug / validate_binding / binding_drift / locate。
|
|
25
|
+
为什么必须钉死:列形状漂移会让对账器把坏列当好消息(漏报);键若悄悄带上
|
|
26
|
+
行位,真源上方插一行就会让整库失配——两者都是静默退化,只有断言看得见。
|
|
27
|
+
⑨ 逐字节回放回归:对**真实索引出的卡**断言 `render(条目) + "\n" == 卡片正文`。
|
|
28
|
+
⑧ 只证「列可定位」;只有逐字节相等才证明卡片是真源条目的**派生物**而非另一份
|
|
29
|
+
副本(渲染漂移或人工改写时列全然不变,对账器看不出来)。外部历史库另段抽样
|
|
30
|
+
(设 MDCG_ROOT 则跑,未设如实 SKIP 不虚报通过)。
|
|
31
|
+
|
|
32
|
+
运行:python -m md_cg.test_p27_docindex
|
|
33
|
+
设 MDCG_TEST_LIVE_ROOT=1 且 MDCG_ROOT=<认知图库根> 时才追加跑【12b】
|
|
34
|
+
外部历史库逐字节回放抽样(该段会以新身份打开外部库并 provision DEK,
|
|
35
|
+
故默认不跑:跑测试不该动生产库)
|
|
36
|
+
"""
|
|
37
|
+
from __future__ import annotations
|
|
38
|
+
|
|
39
|
+
import os
|
|
40
|
+
import random
|
|
41
|
+
import shutil
|
|
42
|
+
import sys
|
|
43
|
+
import tempfile
|
|
44
|
+
|
|
45
|
+
from . import codeindex, corpus, docindex, nodefile, refindex, routing, tokens
|
|
46
|
+
from . import mcp_server
|
|
47
|
+
from .mdcos import MdCGOS, MdCGSecure # 生产路径:forget 属 OS 层,基础层 MdCG 无删除原语
|
|
48
|
+
from .mcp_server import call_tool
|
|
49
|
+
from .security import Principal
|
|
50
|
+
|
|
51
|
+
PASS = FAIL = 0
|
|
52
|
+
FAILS = []
|
|
53
|
+
|
|
54
|
+
_BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
55
|
+
ROOT = os.path.join(_BASE, "_md_cg_p27")
|
|
56
|
+
DOCS = os.path.join(_BASE, "docs")
|
|
57
|
+
PLAN_DOC = "mdcg/认知图_MD目录方案_v0.1.md"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def check(name, cond, detail=""):
|
|
61
|
+
global PASS, FAIL
|
|
62
|
+
if cond:
|
|
63
|
+
PASS += 1
|
|
64
|
+
print(f" [PASS] {name}" + (f" · {detail}" if detail else ""))
|
|
65
|
+
else:
|
|
66
|
+
FAIL += 1
|
|
67
|
+
FAILS.append(name)
|
|
68
|
+
print(f" [FAIL] {name} · {detail}")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _norm(p):
|
|
72
|
+
"""路径归一(大小写 + 分隔符):root 归属比对用,避免同一目录被判成两个。"""
|
|
73
|
+
return os.path.normcase(os.path.normpath(str(p or "")))
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
# 逐字节回放:把「往返列」从「形状可校验」推进到「内容可无损重建」。
|
|
77
|
+
# 生效条件:cg 已索引完成的库;索引条目 tags 含 "doc" 且 frontmatter 有非空 doc_ref。
|
|
78
|
+
# 判据:真源同键条目 `render(条目)` 补一个换行 == 卡片正文——`nodefile.dumps` 对不以
|
|
79
|
+
# 换行结尾的正文补 '\n',故卡片正文恒比 render 多 1 字节(是常态,不是缺陷;判等
|
|
80
|
+
# 一律按 `render + "\n"`,避免把口径差当回归)。
|
|
81
|
+
# 为什么非此不可:列形状对(validate_binding)只保证「可定位」;只有逐字节相等才
|
|
82
|
+
# 证明卡片是真源条目的**派生物**而非另一份副本。渲染漂移或人工改写卡片正文时,
|
|
83
|
+
# 列全都不变(路径/行位/hash 都是真源侧的值),对账器看不出来——本函数才看得见。
|
|
84
|
+
# 悬空(真源已移出)/ 未解析如实计数:那是数据面事实,不冒充渲染失败。
|
|
85
|
+
def _replay_cards(cg, roots=None, sample=0, seed=7):
|
|
86
|
+
nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
|
|
87
|
+
picks = []
|
|
88
|
+
for nid in sorted(nodes):
|
|
89
|
+
if "doc" not in ((nodes.get(nid) or {}).get("tags") or []):
|
|
90
|
+
continue
|
|
91
|
+
kind, ref = refindex.ref_of(cg.get(nid))
|
|
92
|
+
if kind != "doc_ref" or not ref:
|
|
93
|
+
continue
|
|
94
|
+
if roots is not None and _norm(ref.get("root")) not in roots:
|
|
95
|
+
continue
|
|
96
|
+
picks.append((nid, ref))
|
|
97
|
+
if sample and len(picks) > sample:
|
|
98
|
+
picks = sorted(random.Random(seed).sample(picks, sample))
|
|
99
|
+
out = {"sampled": len(picks), "ok": 0, "dangling": 0, "gone": 0,
|
|
100
|
+
"unreadable": 0, "mismatch": []}
|
|
101
|
+
cache = {}
|
|
102
|
+
for nid, ref in picks:
|
|
103
|
+
fp = (str(ref.get("root") or ""), str(ref.get("path") or ""))
|
|
104
|
+
ckey = (_norm(fp[0]), fp[1])
|
|
105
|
+
if ckey not in cache:
|
|
106
|
+
# 把「源文件已移出」(数据面事实,非缺陷)与其它 IO/抽取异常(可能是真
|
|
107
|
+
# 问题:权限、编码、源被改成无标题文档)分开计数——混为一谈会把真实
|
|
108
|
+
# 故障伪装成历史遗留,静默吞掉回归。三态:True 读得 / False 不存在 /
|
|
109
|
+
# None 读不了。
|
|
110
|
+
try:
|
|
111
|
+
with open(os.path.join(fp[0], fp[1]), "r", encoding="utf-8",
|
|
112
|
+
errors="replace") as f:
|
|
113
|
+
cache[ckey] = (True, docindex.extract(f.read(), fp[1]))
|
|
114
|
+
except FileNotFoundError:
|
|
115
|
+
cache[ckey] = (False, None)
|
|
116
|
+
except (OSError, ValueError):
|
|
117
|
+
cache[ckey] = (None, None)
|
|
118
|
+
hit, items = cache[ckey]
|
|
119
|
+
if hit is not True:
|
|
120
|
+
out["gone" if hit is False else "unreadable"] += 1
|
|
121
|
+
continue
|
|
122
|
+
node = cg.get(nid) or {}
|
|
123
|
+
want = docindex.binding_key(docindex.binding_of(node))
|
|
124
|
+
# 键(path#heading_path)在同名标题重复时可能命中多条:先按行位锁定同名代,
|
|
125
|
+
# 行位也漂了才回落首条(此时本来就判「不符」,不掩盖结论)。
|
|
126
|
+
cands = [i for i in items if docindex.binding_key(i) == want]
|
|
127
|
+
if not cands:
|
|
128
|
+
out["dangling"] += 1
|
|
129
|
+
continue
|
|
130
|
+
item = next((i for i in cands if i.get("lineno") == ref.get("lineno")),
|
|
131
|
+
cands[0])
|
|
132
|
+
got = docindex.render(item) + "\n"
|
|
133
|
+
content = node.get("content") or ""
|
|
134
|
+
if got == content:
|
|
135
|
+
out["ok"] += 1
|
|
136
|
+
continue
|
|
137
|
+
k = next((i for i in range(min(len(got), len(content)))
|
|
138
|
+
if got[i] != content[i]), min(len(got), len(content)))
|
|
139
|
+
out["mismatch"].append(
|
|
140
|
+
f"{nid}@{k} 回放={got[k:k + 24]!r} 卡片={content[k:k + 24]!r}")
|
|
141
|
+
return out
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
LONG = "这是总览段落," + "用以验证章节切块在长正文下的表现," * 12 + "结束。"
|
|
145
|
+
|
|
146
|
+
GUIDE_LINES = [
|
|
147
|
+
"---",
|
|
148
|
+
"title: 测试指南",
|
|
149
|
+
"tags: [a, b]",
|
|
150
|
+
"---",
|
|
151
|
+
"",
|
|
152
|
+
"# 总览",
|
|
153
|
+
"",
|
|
154
|
+
LONG,
|
|
155
|
+
"",
|
|
156
|
+
"## 9. 分阶段实施",
|
|
157
|
+
"",
|
|
158
|
+
"表格与正文说明。",
|
|
159
|
+
"",
|
|
160
|
+
"| 阶段 | 内容 |",
|
|
161
|
+
"|---|---|",
|
|
162
|
+
"| R1 | 索引 |",
|
|
163
|
+
"",
|
|
164
|
+
"### 9.1 小节点",
|
|
165
|
+
"",
|
|
166
|
+
"一行小注。",
|
|
167
|
+
"",
|
|
168
|
+
"## 代码示例",
|
|
169
|
+
"",
|
|
170
|
+
"```python",
|
|
171
|
+
"# 这不是标题",
|
|
172
|
+
"## 也不是标题",
|
|
173
|
+
"def f():",
|
|
174
|
+
" return 1",
|
|
175
|
+
"```",
|
|
176
|
+
"",
|
|
177
|
+
"---",
|
|
178
|
+
"",
|
|
179
|
+
"### 末尾小节",
|
|
180
|
+
"",
|
|
181
|
+
"收尾正文。",
|
|
182
|
+
]
|
|
183
|
+
GUIDE = "\n".join(GUIDE_LINES)
|
|
184
|
+
|
|
185
|
+
SECRET = "\n".join([
|
|
186
|
+
"# 私有配置",
|
|
187
|
+
"",
|
|
188
|
+
"这是私有目录下的文档。",
|
|
189
|
+
"",
|
|
190
|
+
"## 子节",
|
|
191
|
+
"",
|
|
192
|
+
"内容。",
|
|
193
|
+
])
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def find(items, heading):
|
|
197
|
+
return next((i for i in items if i["heading"] == heading), None)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def index_doc(cg, path, **extra):
|
|
201
|
+
a = {"op": "index_doc", "path": path}
|
|
202
|
+
a.update(extra)
|
|
203
|
+
return call_tool(cg, "cg", a)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def main():
|
|
207
|
+
print("=" * 68)
|
|
208
|
+
print("md 认知图 P27 验收 · 文档索引(index_doc / doc_ref / op=ref)")
|
|
209
|
+
print("=" * 68)
|
|
210
|
+
|
|
211
|
+
tmp = tempfile.mkdtemp(prefix="mdcg_docidx_")
|
|
212
|
+
fx = os.path.join(tmp, "fx")
|
|
213
|
+
os.makedirs(os.path.join(fx, "private"))
|
|
214
|
+
with open(os.path.join(fx, "guide.md"), "w", encoding="utf-8") as f:
|
|
215
|
+
f.write(GUIDE)
|
|
216
|
+
with open(os.path.join(fx, "private", "secret.md"), "w", encoding="utf-8") as f:
|
|
217
|
+
f.write(SECRET)
|
|
218
|
+
with open(os.path.join(fx, "notes.txt"), "w", encoding="utf-8") as f:
|
|
219
|
+
f.write("不是文档")
|
|
220
|
+
|
|
221
|
+
corpus.reset_root(ROOT)
|
|
222
|
+
cg = MdCGOS(ROOT)
|
|
223
|
+
guide_path = os.path.join(fx, "guide.md")
|
|
224
|
+
|
|
225
|
+
try:
|
|
226
|
+
# ===================================================== ① 注册表 + 切块
|
|
227
|
+
print("\n【1】注册一致性 / 切块纪律(level<=3 + 小节点合并)")
|
|
228
|
+
cg_tool = next(t for t in mcp_server.KERNEL_TOOLS if t["name"] == "cg")
|
|
229
|
+
declared = set(cg_tool["inputSchema"]["properties"]["op"]["description"].split("|"))
|
|
230
|
+
check("工具 op 枚举 == tokens.ALL_OPS(§七-11 防漏改)",
|
|
231
|
+
declared == set(tokens.ALL_OPS),
|
|
232
|
+
f"only-schema={sorted(declared - set(tokens.ALL_OPS))} "
|
|
233
|
+
f"only-ops={sorted(set(tokens.ALL_OPS) - declared)}")
|
|
234
|
+
check("index_doc 已进 ALL_OPS", "index_doc" in tokens.ALL_OPS)
|
|
235
|
+
|
|
236
|
+
try:
|
|
237
|
+
docindex.extract("x", "foo.txt")
|
|
238
|
+
unknown = False
|
|
239
|
+
except ValueError as exc:
|
|
240
|
+
unknown = "无文档提取器" in str(exc)
|
|
241
|
+
check("无文档提取器后缀显式报错", unknown)
|
|
242
|
+
|
|
243
|
+
items = docindex.extract(GUIDE, "guide.md")
|
|
244
|
+
heads = {i["heading"] for i in items}
|
|
245
|
+
check("切块只产出 3 个节点(小节点已合并)", len(items) == 3,
|
|
246
|
+
f"n={len(items)} heads={sorted(heads)}")
|
|
247
|
+
check("围栏内的 # 不是标题(假标题零泄漏)",
|
|
248
|
+
not any("不是标题" in h for h in heads), str(sorted(heads)))
|
|
249
|
+
check("总览区间覆盖至文末(未被假标题截断)",
|
|
250
|
+
find(items, "总览")["end"] == len(GUIDE_LINES),
|
|
251
|
+
f"end={find(items, '总览')['end']}")
|
|
252
|
+
check("YAML frontmatter 未被当正文(总览起始行=6)",
|
|
253
|
+
find(items, "总览")["lineno"] == 6,
|
|
254
|
+
f"lineno={find(items, '总览')['lineno']}")
|
|
255
|
+
s9 = find(items, "9. 分阶段实施")
|
|
256
|
+
check("9. 章节区间正确(10-21)",
|
|
257
|
+
s9["lineno"] == 10 and s9["end"] == 21,
|
|
258
|
+
f"{s9['lineno']}-{s9['end']}")
|
|
259
|
+
check("heading_path 反映嵌套",
|
|
260
|
+
s9["heading_path"] == ["总览", "9. 分阶段实施"],
|
|
261
|
+
str(s9["heading_path"]))
|
|
262
|
+
check("anchor 生成正常", s9["anchor"] == "9-分阶段实施", s9["anchor"])
|
|
263
|
+
check("小节点正文并入父节摘要(不丢内容)",
|
|
264
|
+
"一行小注" in s9["summary_parts"], s9["summary_parts"][:60])
|
|
265
|
+
code = find(items, "代码示例")
|
|
266
|
+
check("文末小节点正文并入其父节摘要",
|
|
267
|
+
"收尾正文" in code["summary_parts"], code["summary_parts"][:60])
|
|
268
|
+
check("只切 level<=3(条目层级均 <=3)",
|
|
269
|
+
all(i["level"] <= 3 for i in items),
|
|
270
|
+
str([i["level"] for i in items]))
|
|
271
|
+
|
|
272
|
+
# ===================================================== ② 渲染即 CCG
|
|
273
|
+
print("\n【2】render 产出 CCG(否则恒定 BLINDSPOT)")
|
|
274
|
+
rendered = docindex.render(s9)
|
|
275
|
+
cpl = nodefile.ccg_completeness(rendered)
|
|
276
|
+
check("CCG 5 要素齐全", cpl["complete"] is True, str(cpl["required_present"]))
|
|
277
|
+
check("CCG 6 行全在", cpl["all_present"] is True)
|
|
278
|
+
# 「不存全文」的准确含义:不逐字复制整节,且摘要栏有长度上限
|
|
279
|
+
# (表格/代码若落在前 200 字内被摘要带上是有意设计——否则检索不到关键词)。
|
|
280
|
+
raw_section = "\n".join(GUIDE_LINES[9:21])
|
|
281
|
+
exec_line = next(ln for ln in rendered.split("\n") if ln.startswith("# 执行:"))
|
|
282
|
+
check("不存全文(不逐字复制整节 + 摘要栏封顶)",
|
|
283
|
+
raw_section not in rendered and len(exec_line) <= 200 + len("# 执行:"),
|
|
284
|
+
f"exec_len={len(exec_line)}")
|
|
285
|
+
|
|
286
|
+
# 生效条件必须与 frontmatter 同源:四槽合成,单槽不是生效条件。
|
|
287
|
+
# 改造前正文写「文档=X;检索…时」(第三种方言),frontmatter 只写单槽
|
|
288
|
+
# observation_position → condition_space_text(require_full=True) 恒为 ""。
|
|
289
|
+
cs = docindex.condition_space(s9)
|
|
290
|
+
check("condition_space 四槽齐备",
|
|
291
|
+
set(cs) == set(nodefile.CONDITION_SLOTS_REQUIRED), str(sorted(cs)))
|
|
292
|
+
synth = nodefile.condition_space_text(cs)
|
|
293
|
+
check("四槽合成出非空生效条件(单槽冒充已废止)", bool(synth), synth)
|
|
294
|
+
check("正文生效条件 = 四槽合成结果(正文与 frontmatter 同源)",
|
|
295
|
+
f"# 生效条件:{synth}" in rendered, synth)
|
|
296
|
+
check("时间槽用全时窗哨兵(不把索引时刻伪造成条件)",
|
|
297
|
+
nodefile.is_full_time_window(cs.get("time_window")),
|
|
298
|
+
str(cs.get("time_window")))
|
|
299
|
+
|
|
300
|
+
# ===================================================== ③ 索引 + 密级
|
|
301
|
+
print("\n【3】index_doc:落盘 / 密级 / 层(§1.3-3 裁定)")
|
|
302
|
+
mt_before = {}
|
|
303
|
+
for d, _s, fs in os.walk(fx):
|
|
304
|
+
for fn in fs:
|
|
305
|
+
mt_before[os.path.join(d, fn)] = os.path.getmtime(os.path.join(d, fn))
|
|
306
|
+
out = index_doc(cg, fx)
|
|
307
|
+
check("op=index_doc ok", out.get("ok"), str(out)[:120])
|
|
308
|
+
check("indexed == 4(guide 3 + secret 1)", out.get("indexed") == 4,
|
|
309
|
+
str(out.get("indexed")))
|
|
310
|
+
check("error_count == 0", out.get("error_count") == 0, str(out.get("errors")))
|
|
311
|
+
check("files 只计 .md(2)", out.get("files") == 2, str(out.get("files")))
|
|
312
|
+
check("skipped_suffixes 报出 .txt", ".txt" in out["skipped_suffixes"],
|
|
313
|
+
str(out["skipped_suffixes"]))
|
|
314
|
+
check("layer 默认 knowledge", out.get("layer") == "knowledge")
|
|
315
|
+
check("密级默认 internal、私有目录降 private",
|
|
316
|
+
out["sensitivity"].get("internal") == 3
|
|
317
|
+
and out["sensitivity"].get("private") == 1,
|
|
318
|
+
str(out["sensitivity"]))
|
|
319
|
+
|
|
320
|
+
g_id = docindex.node_id(s9)
|
|
321
|
+
g_fm = (cg.get(g_id) or {}).get("frontmatter") or {}
|
|
322
|
+
check("guide 节点密级 frontmatter 显式 internal",
|
|
323
|
+
g_fm.get("sensitivity") == "internal", str(g_fm.get("sensitivity")))
|
|
324
|
+
check("guide 节点 layer=knowledge", g_fm.get("layer") == "knowledge")
|
|
325
|
+
check("verification_basis=data(文档以原文为准)",
|
|
326
|
+
g_fm.get("verification_basis") == "data")
|
|
327
|
+
check("frontmatter.condition_space 四槽齐备(单槽冒充已废止)",
|
|
328
|
+
set(g_fm.get("condition_space") or {}) ==
|
|
329
|
+
set(nodefile.CONDITION_SLOTS_REQUIRED),
|
|
330
|
+
str(sorted(g_fm.get("condition_space") or {})))
|
|
331
|
+
check("frontmatter 条件空间与正文生效条件同源(同一纯函数)",
|
|
332
|
+
g_fm.get("condition_space") == docindex.condition_space(s9),
|
|
333
|
+
str(g_fm.get("condition_space")))
|
|
334
|
+
check("路由域由 domain: 标签显式承担(分桶结果与改造前逐字相同)",
|
|
335
|
+
routing.route_key(g_fm.get("condition_space"), g_fm.get("tags"))
|
|
336
|
+
== routing.normalize_domain("guide.md"),
|
|
337
|
+
str(routing.route_key(g_fm.get("condition_space"), g_fm.get("tags"))))
|
|
338
|
+
sec = docindex.extract(SECRET, "private/secret.md")[0]
|
|
339
|
+
s_fm = (cg.get(docindex.node_id(sec)) or {}).get("frontmatter") or {}
|
|
340
|
+
check("私有目录文档密级=private(保守降级)",
|
|
341
|
+
s_fm.get("sensitivity") == "private", str(s_fm.get("sensitivity")))
|
|
342
|
+
|
|
343
|
+
mt_after = {}
|
|
344
|
+
for d, _s, fs in os.walk(fx):
|
|
345
|
+
for fn in fs:
|
|
346
|
+
mt_after[os.path.join(d, fn)] = os.path.getmtime(os.path.join(d, fn))
|
|
347
|
+
check("只读契约:索引不触碰源文件", mt_before == mt_after)
|
|
348
|
+
|
|
349
|
+
# ===================================================== ④ doc_ref
|
|
350
|
+
print("\n【4】frontmatter.doc_ref(指回原文,不存全文)")
|
|
351
|
+
dr = g_fm.get("doc_ref") or {}
|
|
352
|
+
check("doc_ref 字段齐全",
|
|
353
|
+
all(k in dr for k in ("path", "heading", "heading_path", "level",
|
|
354
|
+
"lineno", "end", "anchor", "lang", "precise",
|
|
355
|
+
"hash", "root")), str(sorted(dr)))
|
|
356
|
+
check("doc_ref 指向相对路径与源行号",
|
|
357
|
+
dr.get("path") == "guide.md" and dr.get("lineno") == 10
|
|
358
|
+
and dr.get("end") == 21, str({k: dr.get(k) for k in ("path", "lineno", "end")}))
|
|
359
|
+
check("正文含 --- 未污染 frontmatter(doc_ref 可正常读回)",
|
|
360
|
+
dr.get("heading_path") == ["总览", "9. 分阶段实施"],
|
|
361
|
+
str(dr.get("heading_path")))
|
|
362
|
+
|
|
363
|
+
# ===================================================== ⑤ 检索资格
|
|
364
|
+
print("\n【5】检索资格(文档节点不得是 BLINDSPOT)")
|
|
365
|
+
cg.flush()
|
|
366
|
+
rd = call_tool(cg, "cg", {"op": "read", "query": "分阶段实施", "k": 10})
|
|
367
|
+
hits = [r for r in rd.get("results", []) if r["node"]["id"] == g_id]
|
|
368
|
+
check("文档节点可被检索到", bool(hits), f"hits={len(hits)}")
|
|
369
|
+
check("CCG 完整 → state=DEFER(v2:情境未确认条件不冒充接受,非 BLINDSPOT)",
|
|
370
|
+
bool(hits) and hits[0]["state"] == "DEFER",
|
|
371
|
+
str(hits[0]["state"] if hits else None))
|
|
372
|
+
|
|
373
|
+
# ===================================================== ⑥ op=ref
|
|
374
|
+
print("\n【6】op=ref 回读文档区间 + 漂移")
|
|
375
|
+
rr = call_tool(cg, "cg", {"op": "ref", "node_id": g_id})
|
|
376
|
+
check("ref 回读成功且识别为 doc_ref",
|
|
377
|
+
rr.get("ok") and rr.get("ref_kind") == "doc_ref",
|
|
378
|
+
str(rr.get("ref_kind")))
|
|
379
|
+
check("回读文本 = 源章节(含标题与表格行)",
|
|
380
|
+
"# 9. 分阶段实施" in (rr.get("text") or "")
|
|
381
|
+
and "| R1 | 索引 |" in (rr.get("text") or ""))
|
|
382
|
+
check("hash_match=True(索引侧与回读侧同算法)",
|
|
383
|
+
rr.get("hash_match") is True, str(rr.get("hash")))
|
|
384
|
+
with open(guide_path, "w", encoding="utf-8") as f:
|
|
385
|
+
f.write(GUIDE.replace("| R1 | 索引 |", "| R9 | 已改 |"))
|
|
386
|
+
rr2 = call_tool(cg, "cg", {"op": "ref", "node_id": g_id})
|
|
387
|
+
check("文档改动 → stale 可检出", rr2.get("hash_match") is False
|
|
388
|
+
and rr2.get("stale") is True, str(rr2.get("hash")))
|
|
389
|
+
index_doc(cg, fx)
|
|
390
|
+
rr3 = call_tool(cg, "cg", {"op": "ref", "node_id": g_id})
|
|
391
|
+
check("重跑 index_doc → 漂移消除", rr3.get("hash_match") is True)
|
|
392
|
+
with open(guide_path, "w", encoding="utf-8") as f:
|
|
393
|
+
f.write(GUIDE)
|
|
394
|
+
|
|
395
|
+
# ===================================================== ⑦ 覆盖/截断
|
|
396
|
+
print("\n【7】显式覆盖与截断上报")
|
|
397
|
+
ov = index_doc(cg, fx, sensitivity="public")
|
|
398
|
+
check("sensitivity 参数可覆盖默认",
|
|
399
|
+
ov["sensitivity"].get("public") == 4, str(ov["sensitivity"]))
|
|
400
|
+
g_fm2 = (cg.get(g_id) or {}).get("frontmatter") or {}
|
|
401
|
+
check("覆盖后 frontmatter 密级=public", g_fm2.get("sensitivity") == "public")
|
|
402
|
+
index_doc(cg, fx) # 复位默认密级
|
|
403
|
+
tr = index_doc(cg, fx, max_files=1)
|
|
404
|
+
check("触 max_files → truncated=True", tr.get("truncated") is True,
|
|
405
|
+
str(tr.get("truncated_reason")))
|
|
406
|
+
check("截断时 note 明确警告", "截断" in (tr.get("note") or ""),
|
|
407
|
+
(tr.get("note") or "")[:50])
|
|
408
|
+
index_doc(cg, fx)
|
|
409
|
+
|
|
410
|
+
# ===================================================== ⑧ 幂等
|
|
411
|
+
print("\n【8】幂等(重跑 ≡ 首跑)")
|
|
412
|
+
first = index_doc(cg, fx)
|
|
413
|
+
n1 = sum(1 for e in cg.index["nodes"].values() if e["layer"] == "knowledge")
|
|
414
|
+
second = index_doc(cg, fx)
|
|
415
|
+
n2 = sum(1 for e in cg.index["nodes"].values() if e["layer"] == "knowledge")
|
|
416
|
+
check("重跑节点数不变(按 id 原子覆盖,不清目录)", n1 == n2, f"{n1} vs {n2}")
|
|
417
|
+
check("重跑 id 集合稳定(幂等)",
|
|
418
|
+
set(first["ids"]) == set(second["ids"]), str(first["ids"]))
|
|
419
|
+
|
|
420
|
+
# ===================================================== ⑨ 真实 docs/
|
|
421
|
+
print("\n【9】真实 docs/:章节可定位、行号与源一致、可检索")
|
|
422
|
+
# docs/experiments/ 是 .gitignore 整目录忽略的实验产物(实测 2358 个 md,
|
|
423
|
+
# 属索引噪声而非文档事实源),会把 max_files=500 撑爆:显式 skip_dirs 排除。
|
|
424
|
+
# 用「排除 + 回报」而不是「调大上限」——排除结果落在 skipped_dirs 里,不静默。
|
|
425
|
+
SKIP_NOISE = ("experiments",)
|
|
426
|
+
|
|
427
|
+
def _count_md(base, skip_names):
|
|
428
|
+
n = 0
|
|
429
|
+
for d, dirs, fs in os.walk(base):
|
|
430
|
+
dirs[:] = [x for x in dirs if x not in skip_names]
|
|
431
|
+
n += sum(1 for fn in fs if fn.lower().endswith(".md"))
|
|
432
|
+
return n
|
|
433
|
+
|
|
434
|
+
# 测试不得依赖仓库外状态:docs/experiments/ 被 .gitignore 整目录忽略,
|
|
435
|
+
# 外部 clone 后物理不存在,而本段断言的正是「真实扫描中 skip_dirs 命中
|
|
436
|
+
# 要被回报」。不存在时自备最小探针目录(跑完即清);已存在时零动作
|
|
437
|
+
# (不碰本机历史产物)。探针为无标题正文——extract 返回空条目,不产生
|
|
438
|
+
# doc 节点;且探针在 experiments 内,_count_md 与被测扫描两侧口径一致,
|
|
439
|
+
# 对「全部 md 被索引」断言零影响。
|
|
440
|
+
probe_dir = os.path.join(DOCS, "experiments")
|
|
441
|
+
probe_created = not os.path.isdir(probe_dir)
|
|
442
|
+
if probe_created:
|
|
443
|
+
os.makedirs(probe_dir, exist_ok=True)
|
|
444
|
+
with open(os.path.join(probe_dir, "_probe_no_heading.md"),
|
|
445
|
+
"w", encoding="utf-8") as f:
|
|
446
|
+
f.write("探针正文:无标题不产节点,仅让 experiments 物理存在。\n")
|
|
447
|
+
try:
|
|
448
|
+
expect_files = _count_md(DOCS, SKIP_NOISE)
|
|
449
|
+
real = index_doc(cg, DOCS, skip_dirs=list(SKIP_NOISE))
|
|
450
|
+
finally:
|
|
451
|
+
if probe_created:
|
|
452
|
+
shutil.rmtree(probe_dir)
|
|
453
|
+
check("docs/ 全部 md 被索引(无静默跳过)",
|
|
454
|
+
real.get("error_count") == 0 and real.get("files") == expect_files
|
|
455
|
+
and real.get("truncated") is False,
|
|
456
|
+
f"files={real.get('files')}/{expect_files} errs={real.get('error_count')}")
|
|
457
|
+
check("skip_dirs 实际排掉的目录被回报(排除不静默)",
|
|
458
|
+
any("experiments" in p for p in (real.get("skipped_dirs") or [])),
|
|
459
|
+
str(real.get("skipped_dirs"))[:80])
|
|
460
|
+
r_items = docindex.extract(
|
|
461
|
+
open(os.path.join(DOCS, PLAN_DOC), encoding="utf-8").read(), PLAN_DOC)
|
|
462
|
+
s9r = next((i for i in r_items if i["heading"].startswith("9. ")), None)
|
|
463
|
+
real_lines = open(os.path.join(DOCS, PLAN_DOC), encoding="utf-8").read().split("\n")
|
|
464
|
+
hline = 1 + next(k for k, ln in enumerate(real_lines) if ln.startswith("## 9."))
|
|
465
|
+
check("§9 章节被切出(真实文档)", s9r is not None,
|
|
466
|
+
s9r["heading"] if s9r else "缺失")
|
|
467
|
+
check("行号与源文件一致(可定位)",
|
|
468
|
+
s9r is not None and s9r["lineno"] == hline,
|
|
469
|
+
f"{s9r['lineno'] if s9r else None} == {hline}")
|
|
470
|
+
r_id = docindex.node_id(s9r)
|
|
471
|
+
rd2 = call_tool(cg, "cg", {"op": "read", "query": "分阶段实施(稳健)", "k": 10})
|
|
472
|
+
h2 = [r for r in rd2.get("results", []) if r["node"]["id"] == r_id]
|
|
473
|
+
check("§9 表可被检索到且可判(state=DEFER,非 BLINDSPOT)",
|
|
474
|
+
bool(h2) and h2[0]["state"] == "DEFER",
|
|
475
|
+
str(h2[0]["state"] if h2 else None))
|
|
476
|
+
rr4 = call_tool(cg, "cg", {"op": "ref", "node_id": r_id})
|
|
477
|
+
check("回读到 §9 表原文(给出行号区间)",
|
|
478
|
+
rr4.get("ok") and "分阶段实施" in (rr4.get("text") or ""),
|
|
479
|
+
f"L{s9r['lineno']}-L{s9r['end']}" if s9r else "")
|
|
480
|
+
|
|
481
|
+
# ===================================================== ⑨b skip_dirs
|
|
482
|
+
print("\n【9b】skip_dirs:追加排除(只增不减)+ 结果可审计")
|
|
483
|
+
hit0, rules0 = codeindex.skip_matcher(None)
|
|
484
|
+
check("skip_dirs 缺省时判定器为空(默认行为与改造前逐字一致)",
|
|
485
|
+
hit0 is None and rules0 == [], f"{hit0} {rules0}")
|
|
486
|
+
sb = os.path.join(tmp, "skipdirs")
|
|
487
|
+
for rel in ("docs/keep/keep.md", "docs/experiments/probe/p.md",
|
|
488
|
+
"docs/experiments/x.md", "sub/experiments/y.md",
|
|
489
|
+
"node_modules/pkg/n.md"):
|
|
490
|
+
fp = os.path.join(sb, rel.replace("/", os.sep))
|
|
491
|
+
os.makedirs(os.path.dirname(fp), exist_ok=True)
|
|
492
|
+
with open(fp, "w", encoding="utf-8") as f:
|
|
493
|
+
f.write("# 标题\n\n" + LONG + "\n")
|
|
494
|
+
_i, _e, st_plain = docindex.index_dir(sb)
|
|
495
|
+
check("不给 skip_dirs:内置排除照旧(node_modules 不计入)",
|
|
496
|
+
st_plain["files"] == 4 and st_plain["skipped_dirs"] == [],
|
|
497
|
+
f"files={st_plain['files']} skip={st_plain['skipped_dirs']}")
|
|
498
|
+
_i, _e, st_name = docindex.index_dir(sb, skip_dirs=["experiments"])
|
|
499
|
+
check("不含 / 的规则按目录名匹配(各层级同名目录都排)"
|
|
500
|
+
" · 且内置排除不可被关闭(node_modules 仍被排)",
|
|
501
|
+
st_name["files"] == 1
|
|
502
|
+
and sorted(st_name["skipped_dirs"]) == ["docs/experiments",
|
|
503
|
+
"sub/experiments"],
|
|
504
|
+
f"files={st_name['files']} skip={sorted(st_name['skipped_dirs'])}")
|
|
505
|
+
_i, _e, st_path = docindex.index_dir(sb, skip_dirs=["docs/experiments"])
|
|
506
|
+
check("含 / 的规则按相对路径匹配(只排这一处)",
|
|
507
|
+
st_path["files"] == 2 and st_path["skipped_dirs"] == ["docs/experiments"],
|
|
508
|
+
f"files={st_path['files']} skip={st_path['skipped_dirs']}")
|
|
509
|
+
_i, _e, st_norm = docindex.index_dir(sb, skip_dirs=["docs\\experiments\\"])
|
|
510
|
+
check("规则先归一(反斜杠/首尾斜杠)再匹配,不静默失配",
|
|
511
|
+
st_norm["files"] == 2
|
|
512
|
+
and st_norm["skipped_dirs"] == ["docs/experiments"],
|
|
513
|
+
f"files={st_norm['files']} skip={st_norm['skipped_dirs']}")
|
|
514
|
+
|
|
515
|
+
# ===================================================== ⑩ 索引对账
|
|
516
|
+
# 缺口:node_id 含 heading_path,改标题 → 整篇 id 重算;add_items 只做
|
|
517
|
+
# 同 id 幂等 upsert,旧代节点无人清退 → 新旧并存、同一文档召回两份。
|
|
518
|
+
# 水位 reconcile 只剪水位条目、不剪节点,所以必须在索引后显式清。
|
|
519
|
+
print("\n【10】节点级对账:改标题 → 清退过期代(消除重复召回)")
|
|
520
|
+
rn = os.path.join(fx, "rename.md")
|
|
521
|
+
b1 = "# 标题甲\n\n" + LONG + "\n"
|
|
522
|
+
with open(rn, "w", encoding="utf-8") as f:
|
|
523
|
+
f.write(b1)
|
|
524
|
+
r1 = index_doc(cg, fx)
|
|
525
|
+
ids1 = {docindex.node_id(i) for i in docindex.extract(b1, "rename.md")}
|
|
526
|
+
check("首轮:节点在库", all(cg.get(n) for n in ids1), f"n={len(ids1)}")
|
|
527
|
+
check("首轮无孤儿(清退计数 0)",
|
|
528
|
+
(r1.get("pruned") or {}).get("count") == 0, str(r1.get("pruned")))
|
|
529
|
+
|
|
530
|
+
b2 = "# 标题乙(已改名)\n\n" + LONG + "\n"
|
|
531
|
+
with open(rn, "w", encoding="utf-8") as f:
|
|
532
|
+
f.write(b2)
|
|
533
|
+
r2 = index_doc(cg, fx)
|
|
534
|
+
ids2 = {docindex.node_id(i) for i in docindex.extract(b2, "rename.md")}
|
|
535
|
+
check("改标题后新代上台", all(cg.get(n) for n in ids2), f"n={len(ids2)}")
|
|
536
|
+
check("旧代被清退(新旧不并存)",
|
|
537
|
+
not any(cg.get(n) for n in ids1),
|
|
538
|
+
f"残留={sorted(n for n in ids1 if cg.get(n))}")
|
|
539
|
+
check("清退数量上报", (r2.get("pruned") or {}).get("count") == len(ids1 - ids2),
|
|
540
|
+
str(r2.get("pruned")))
|
|
541
|
+
|
|
542
|
+
b3 = "# 标题丙\n\n" + LONG + "\n"
|
|
543
|
+
with open(rn, "w", encoding="utf-8") as f:
|
|
544
|
+
f.write(b3)
|
|
545
|
+
r3 = index_doc(cg, fx, prune_dry_run=True)
|
|
546
|
+
check("prune_dry_run 只列不清(节点仍在库)",
|
|
547
|
+
(r3.get("pruned") or {}).get("dry_run") is True
|
|
548
|
+
and any(cg.get(n) for n in ids2), str(r3.get("pruned")))
|
|
549
|
+
index_doc(cg, fx)
|
|
550
|
+
check("随后实际清退 → 旧代消失", not any(cg.get(n) for n in ids2))
|
|
551
|
+
|
|
552
|
+
b4 = "# 标题丁\n\n" + LONG + "\n"
|
|
553
|
+
with open(rn, "w", encoding="utf-8") as f:
|
|
554
|
+
f.write(b4)
|
|
555
|
+
ids4 = {docindex.node_id(i) for i in docindex.extract(b4, "rename.md")}
|
|
556
|
+
r5 = index_doc(cg, fx, prune=False)
|
|
557
|
+
check("prune=false 可关闭(不清退)",
|
|
558
|
+
r5.get("pruned") is None and all(cg.get(n) for n in ids4),
|
|
559
|
+
str(r5.get("pruned")))
|
|
560
|
+
r6 = index_doc(cg, fx, max_files=1)
|
|
561
|
+
check("截断时不清退(没扫完 ≠ 剩下都过期)",
|
|
562
|
+
r6.get("pruned") is None, f"truncated={r6.get('truncated')}")
|
|
563
|
+
|
|
564
|
+
print("\n【10b】悬空清退:源文件删除 → op=ref action=prune 处置")
|
|
565
|
+
ck1 = call_tool(cg, "cg", {"op": "ref", "action": "check", "max_nodes": 5000})
|
|
566
|
+
pre = len(ck1.get("dangling") or [])
|
|
567
|
+
os.remove(rn)
|
|
568
|
+
ck2 = call_tool(cg, "cg", {"op": "ref", "action": "check", "max_nodes": 5000})
|
|
569
|
+
check("删源后 check 报悬空", len(ck2.get("dangling") or []) > pre,
|
|
570
|
+
f"{pre}→{len(ck2.get('dangling') or [])}")
|
|
571
|
+
dry = call_tool(cg, "cg", {"op": "ref", "action": "prune", "dry_run": True})
|
|
572
|
+
check("prune_dry_run 列出待清退但不删",
|
|
573
|
+
dry.get("dry_run") is True and dry.get("count", 0) >= 1
|
|
574
|
+
and any(cg.get(n) for n in ids4), str(dry)[:130])
|
|
575
|
+
gone = call_tool(cg, "cg", {"op": "ref", "action": "prune"})
|
|
576
|
+
check("prune 清退悬空节点",
|
|
577
|
+
gone.get("count", 0) >= 1 and not any(cg.get(n) for n in ids4),
|
|
578
|
+
str(gone)[:130])
|
|
579
|
+
ck3 = call_tool(cg, "cg", {"op": "ref", "action": "check", "max_nodes": 5000})
|
|
580
|
+
check("清退后悬空归零", not (ck3.get("dangling") or []),
|
|
581
|
+
str(ck3.get("dangling"))[:130])
|
|
582
|
+
|
|
583
|
+
# 跨进程一致性:新实例重放 _index_log 后,已清退节点不得复活成幽灵条目
|
|
584
|
+
# (索引有条目、文件不存在)——只有把「删除」也写进增量日志才成立。
|
|
585
|
+
cg2 = MdCGOS(ROOT)
|
|
586
|
+
revived = sorted(n for n in ids4 if n in (cg2.index.get("nodes") or {}))
|
|
587
|
+
check("清退后重开不复活幽灵条目", not revived, f"复活={revived}")
|
|
588
|
+
|
|
589
|
+
# 存量幽灵(历史「删除只摘内存索引」遗留):索引有条目、节点文件不存在
|
|
590
|
+
# 也必须清得掉,否则 dangling 永远够不着零。
|
|
591
|
+
gid = "doc_ghost_probe0001"
|
|
592
|
+
cg.index["nodes"][gid] = {"path": "knowledge/ghost_probe.md",
|
|
593
|
+
"layer": "knowledge", "tags": ["doc"],
|
|
594
|
+
"bucket": None, "importance": 0.5}
|
|
595
|
+
g1 = call_tool(cg, "cg", {"op": "ref", "action": "prune"})
|
|
596
|
+
check("幽灵条目被清退(索引有、文件无)",
|
|
597
|
+
len(g1.get("ghost_pruned") or []) >= 1
|
|
598
|
+
and gid not in (cg.index.get("nodes") or {}), str(g1)[:130])
|
|
599
|
+
ck4 = call_tool(cg, "cg", {"op": "ref", "action": "check", "max_nodes": 5000})
|
|
600
|
+
check("清幽灵后悬空仍为零", not (ck4.get("dangling") or []),
|
|
601
|
+
str(ck4.get("dangling"))[:130])
|
|
602
|
+
|
|
603
|
+
# ============================================ ⑪ fence 往返列(binding)
|
|
604
|
+
# 缺口:往返列(取列 / 造键 / 校验 / 漂移 / 反查)是「条件卡 ↔ Markdown 真源」
|
|
605
|
+
# 的可校验面,此前零测试。列形状一旦漂移,对账器会把坏列当好消息(漏报);
|
|
606
|
+
# 键若悄悄带上行位,真源上方插一行就会让整库失配——故两者都要钉死。
|
|
607
|
+
print("\n【11】fence 往返列:形状 / 键不含行位 / 校验 / 漂移 / 反查")
|
|
608
|
+
li = docindex.extract(GUIDE, "guide.md")
|
|
609
|
+
s9b = next(i for i in li if i["heading"] == "9. 分阶段实施")
|
|
610
|
+
rendered = docindex.render(s9b)
|
|
611
|
+
gnode = cg.get(g_id) or {}
|
|
612
|
+
g_ref = (gnode.get("frontmatter") or {}).get("doc_ref") or {}
|
|
613
|
+
b0 = docindex.binding_of(gnode)
|
|
614
|
+
check("必需列齐全且无空值",
|
|
615
|
+
isinstance(b0, dict)
|
|
616
|
+
and all(b0.get(k) not in (None, "") for k in docindex.BINDING_FIELDS),
|
|
617
|
+
str(sorted(docindex.BINDING_FIELDS)))
|
|
618
|
+
check("列面 = 必需列 + 附加列(无第二份列口径)",
|
|
619
|
+
set(b0 or {}) == set(docindex.BINDING_FIELDS)
|
|
620
|
+
| set(docindex.BINDING_OPTIONAL),
|
|
621
|
+
str(sorted(set(b0 or {}) ^ (set(docindex.BINDING_FIELDS)
|
|
622
|
+
| set(docindex.BINDING_OPTIONAL)))))
|
|
623
|
+
check("必需列/附加列 ⊆ 写侧 doc_ref 列(读写同一列面)",
|
|
624
|
+
(set(docindex.BINDING_FIELDS) | set(docindex.BINDING_OPTIONAL))
|
|
625
|
+
<= set(g_ref),
|
|
626
|
+
str(sorted(set(docindex.BINDING_FIELDS) - set(g_ref))))
|
|
627
|
+
check("binding_of 与直接读 frontmatter 同源(同一投影)",
|
|
628
|
+
b0 == docindex.binding_of({"frontmatter": {"doc_ref": g_ref}}))
|
|
629
|
+
check("binding_of 对非索引节点返回 None(不猜、不补默认值)",
|
|
630
|
+
docindex.binding_of(None) is None
|
|
631
|
+
and docindex.binding_of([]) is None
|
|
632
|
+
and docindex.binding_of({}) is None
|
|
633
|
+
and docindex.binding_of({"frontmatter": {"tags": ["doc"]}}) is None)
|
|
634
|
+
check("validate_binding:本仓卡片列形状合法(ok + 零 issue)",
|
|
635
|
+
docindex.validate_binding(b0) == {"ok": True, "issues": []},
|
|
636
|
+
str(docindex.validate_binding(b0)))
|
|
637
|
+
check("validate_binding:非 dict 如实报错且不抛",
|
|
638
|
+
docindex.validate_binding(None)
|
|
639
|
+
== {"ok": False, "issues": ["绑定不是字典(该节点无 doc_ref)"]})
|
|
640
|
+
lo0 = b0["lineno"]
|
|
641
|
+
bad_detail = []
|
|
642
|
+
for _nm, _b, _want in (
|
|
643
|
+
("缺列", {k: v for k, v in b0.items() if k != "anchor"},
|
|
644
|
+
"缺列 anchor"),
|
|
645
|
+
("类型错 root", {**b0, "root": 123},
|
|
646
|
+
"root 应为字符串,实为 int"),
|
|
647
|
+
("类型错 lineno", {**b0, "lineno": "十"}, "行位不是整数"),
|
|
648
|
+
("行位越界", {**b0, "lineno": 0}, "lineno 越界(0 < 1)"),
|
|
649
|
+
("区间倒置", {**b0, "end": lo0 - 1},
|
|
650
|
+
f"end 早于 lineno({lo0}-{lo0 - 1})")):
|
|
651
|
+
_v = docindex.validate_binding(_b)
|
|
652
|
+
if _v["ok"] or _want not in _v["issues"]:
|
|
653
|
+
bad_detail.append(f"{_nm}→{_v['issues']}")
|
|
654
|
+
check("validate_binding 逐类坏列均被点出(缺列/类型错/越界/区间倒置)",
|
|
655
|
+
not bad_detail, "; ".join(bad_detail)[:170])
|
|
656
|
+
b_shift = docindex.binding_of({"frontmatter": {"doc_ref": {
|
|
657
|
+
**g_ref, "lineno": lo0 + 40, "end": g_ref["end"] + 40}}})
|
|
658
|
+
check("键不含行位:行位位移不改键(重排后仍是同一章节)",
|
|
659
|
+
docindex.binding_key(b0) == docindex.binding_key(b_shift)
|
|
660
|
+
== "guide.md#总览/9. 分阶段实施",
|
|
661
|
+
docindex.binding_key(b0))
|
|
662
|
+
check("漂移按固定列序报出变化项(lineno → end → hash)",
|
|
663
|
+
docindex.binding_drift(b0, b_shift) == ["lineno", "end"]
|
|
664
|
+
and docindex.binding_drift(b0, {**b0, "hash": "zz"}) == ["hash"]
|
|
665
|
+
and docindex.binding_drift(
|
|
666
|
+
b0, {**b0, "hash": "zz", "lineno": lo0 + 1})
|
|
667
|
+
== ["lineno", "hash"],
|
|
668
|
+
str(docindex.binding_drift(b0, b_shift)))
|
|
669
|
+
check("同列零漂移(无变化不报 / 空值不炸)",
|
|
670
|
+
docindex.binding_drift(b0, dict(b0)) == []
|
|
671
|
+
and docindex.binding_drift(None, None) == [])
|
|
672
|
+
check("binding_key 缺 heading_path 回落 path#anchor · 非 dict 返空串",
|
|
673
|
+
docindex.binding_key({"path": "a.md", "anchor": "s1"}) == "a.md#s1"
|
|
674
|
+
and docindex.binding_key({"path": "a.md"}) == "a.md#"
|
|
675
|
+
and docindex.binding_key(None) == "")
|
|
676
|
+
check("binding_slug 与 render 正文「本条目属于」逐字同源",
|
|
677
|
+
docindex.binding_slug(b0) == "guide.md#9-分阶段实施"
|
|
678
|
+
and f"本条目属于 {docindex.binding_slug(b0)}" in rendered,
|
|
679
|
+
docindex.binding_slug(b0))
|
|
680
|
+
check("反查取最内层(父节区间内的子节优先,不回落父节)",
|
|
681
|
+
docindex.locate(li, 12)["heading"] == "9. 分阶段实施"
|
|
682
|
+
and docindex.locate(li, 30)["heading"] == "代码示例",
|
|
683
|
+
str((docindex.locate(li, 12) or {}).get("heading")))
|
|
684
|
+
check("反查区间闭合(标题行与末行都算本节)",
|
|
685
|
+
docindex.locate(li, s9b["lineno"])["heading"] == "9. 分阶段实施"
|
|
686
|
+
and docindex.locate(li, s9b["end"])["heading"] == "9. 分阶段实施")
|
|
687
|
+
check("反查越界/非整数/空表 → None(fail-closed 不猜)",
|
|
688
|
+
docindex.locate(li, 10 ** 6) is None
|
|
689
|
+
and docindex.locate(li, 0) is None
|
|
690
|
+
and docindex.locate(li, "x") is None
|
|
691
|
+
and docindex.locate(li, None) is None
|
|
692
|
+
and docindex.locate([], 10) is None)
|
|
693
|
+
|
|
694
|
+
# ================================ ⑫ 逐字节回放回归(往返列无损)
|
|
695
|
+
# ⑪ 只证「列可校验」;本段把口径推到内容无损:卡片必须是真源条目的派生物
|
|
696
|
+
# (render 补一个换行 == 卡片正文)。root 限定为本次真正索引过的两个真源,
|
|
697
|
+
# 断言只落在本轮写入的卡上——历史残留不参与,避免把脏数据当回归。
|
|
698
|
+
print("\n【12】逐字节回放回归:render(条目) + 换行 == 卡片正文")
|
|
699
|
+
cg.flush()
|
|
700
|
+
rp = _replay_cards(cg, roots={_norm(fx), _norm(DOCS)})
|
|
701
|
+
print(f" 取样 {rp['sampled']} 卡:一致 {rp['ok']} · 键未命中 "
|
|
702
|
+
f"{rp['dangling']} · 源已移出 {rp['gone']} · 读不了 "
|
|
703
|
+
f"{rp['unreadable']}")
|
|
704
|
+
check("回放取样命中卡片(root 限定=本次索引的两个真源)",
|
|
705
|
+
rp["sampled"] > 0, str(rp["sampled"]))
|
|
706
|
+
check("可读卡 100% 逐字节回放一致(写侧 render 与读侧同源)",
|
|
707
|
+
not rp["mismatch"],
|
|
708
|
+
f"一致 {rp['ok']}/{rp['sampled']};" + ";".join(rp["mismatch"][:2]))
|
|
709
|
+
check("零键未命中 / 零源已移出 / 零读不了(源在库则必可重定位并读回)",
|
|
710
|
+
rp["dangling"] == 0 and rp["gone"] == 0
|
|
711
|
+
and rp["unreadable"] == 0,
|
|
712
|
+
f"键未命中={rp['dangling']} 源已移出={rp['gone']} "
|
|
713
|
+
f"读不了={rp['unreadable']}")
|
|
714
|
+
check("回放账目闭合(一致 + 键未命中 + 源已移出 + 读不了 = 取样数)",
|
|
715
|
+
(rp["ok"] + rp["dangling"] + rp["gone"] + rp["unreadable"])
|
|
716
|
+
== rp["sampled"],
|
|
717
|
+
f"{rp['ok']}+{rp['dangling']}+{rp['gone']}+{rp['unreadable']}"
|
|
718
|
+
f" vs {rp['sampled']}")
|
|
719
|
+
|
|
720
|
+
# 历史抽样:真实认知图库(仓外数据面,历史卡最多)。
|
|
721
|
+
# ⚠ 必须**显式 opt-in**(2026-09-24 修复):DSH 把 MDCG_ROOT 导出到每个
|
|
722
|
+
# shell,而本段以新身份打开外部库——MdCGSecure 首次打开该身份即
|
|
723
|
+
# provision DEK(写 _keys.json / _crypto.jsonl,见 mdcos._init_crypto),
|
|
724
|
+
# 于是「跑一次测试」= 动生产库;且受限文件沙箱下该写会阻塞(实测本机
|
|
725
|
+
# 900s 超时、日志零字节)。故:MDCG_TEST_LIVE_ROOT=1 才跑。
|
|
726
|
+
# 语义澄清:这里只保证**查询语义只读**,不是「不写盘」。
|
|
727
|
+
_live_opt = os.environ.get("MDCG_TEST_LIVE_ROOT") == "1"
|
|
728
|
+
ext_root = os.environ.get("MDCG_ROOT") or ""
|
|
729
|
+
if not (_live_opt and ext_root and os.path.isdir(ext_root)):
|
|
730
|
+
print("\n【12b】历史抽样:跳过(需 MDCG_TEST_LIVE_ROOT=1 且 MDCG_ROOT "
|
|
731
|
+
"指向真实库)—— 本仓语料回放见【12】")
|
|
732
|
+
else:
|
|
733
|
+
print(f"\n【12b】历史抽样:外部认知图库 {ext_root}"
|
|
734
|
+
"(只读查询;首开新身份会 provision 本身份 DEK)")
|
|
735
|
+
ext = MdCGSecure(ext_root, principal=Principal(
|
|
736
|
+
actor="p27-replay", clearance="secret", can_write=False,
|
|
737
|
+
can_admin=False, role="designer", auth_mode="local-cli"))
|
|
738
|
+
try:
|
|
739
|
+
hp = _replay_cards(ext, sample=120)
|
|
740
|
+
finally:
|
|
741
|
+
ext.close()
|
|
742
|
+
print(f" 抽样 {hp['sampled']} 卡:一致 {hp['ok']} · 键未命中 "
|
|
743
|
+
f"{hp['dangling']} · 源已移出 {hp['gone']} · 读不了 "
|
|
744
|
+
f"{hp['unreadable']}")
|
|
745
|
+
check("【12b】历史库抽样命中卡片(数据面可达)",
|
|
746
|
+
hp["sampled"] > 0, str(hp["sampled"]))
|
|
747
|
+
check("【12b】可读历史卡零回放不符(未被渲染漂移/人工改写腐化)",
|
|
748
|
+
not hp["mismatch"],
|
|
749
|
+
f"一致 {hp['ok']}/{hp['sampled']};"
|
|
750
|
+
+ ";".join(hp["mismatch"][:2]))
|
|
751
|
+
check("【12b】未读回的全部是「源已移出」(非 IO/抽取故障)",
|
|
752
|
+
hp["unreadable"] == 0,
|
|
753
|
+
f"源已移出={hp['gone']} 读不了={hp['unreadable']}")
|
|
754
|
+
check("【12b】抽样账目闭合(无静默丢弃)",
|
|
755
|
+
(hp["ok"] + hp["dangling"] + hp["gone"] + hp["unreadable"])
|
|
756
|
+
== hp["sampled"],
|
|
757
|
+
f"{hp['ok']}+{hp['dangling']}+{hp['gone']}+{hp['unreadable']}"
|
|
758
|
+
f" vs {hp['sampled']}")
|
|
759
|
+
|
|
760
|
+
finally:
|
|
761
|
+
shutil.rmtree(tmp, ignore_errors=True)
|
|
762
|
+
|
|
763
|
+
print("\n" + "=" * 68)
|
|
764
|
+
print(f"通过 {PASS} / 失败 {FAIL}")
|
|
765
|
+
if FAILS:
|
|
766
|
+
print("失败项:\n - " + "\n - ".join(FAILS))
|
|
767
|
+
print("=" * 68)
|
|
768
|
+
return 1 if FAIL else 0
|
|
769
|
+
|
|
770
|
+
|
|
771
|
+
if __name__ == "__main__":
|
|
772
|
+
if hasattr(sys.stdout, "reconfigure"):
|
|
773
|
+
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
|
774
|
+
sys.exit(main())
|