@furongjun1999/dsh-memory 0.4.11 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +552 -465
- package/codebuddy/CODEBUDDY.md +11 -3
- package/codebuddy/README.md +92 -90
- package/codebuddy/mcp.json +27 -27
- package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
- package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
- package/docs/README.md +143 -111
- package/docs/discipline/harnesses.yaml +244 -226
- package/docs/discipline/templates/full.md.tmpl +61 -61
- package/docs/discipline/templates/rules.mdc.tmpl +68 -0
- package/docs/discipline/templates/skill.md.tmpl +23 -23
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
- package/docs/eval/bench_lingshu_self/bench_self.py +140 -0
- package/docs/eval/bench_lingshu_self/self_bench_result.json +404 -0
- package/docs/eval/bench_lingshu_self//347/201/265/346/236/242/350/207/252/345/272/223/347/253/257/345/210/260/347/253/257/346/243/200/347/264/242/345/256/236/346/265/213_v1.0.md +38 -0
- package/docs/eval//344/270/215/345/217/257/351/235/240/346/200/247/350/220/275/345/234/260_P0_v1.0.md +185 -0
- package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
- package/docs/eval//346/225/205/351/232/234/346/263/250/345/205/245/345/256/236/346/265/213_v1.0.md +422 -0
- package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
- package/docs/eval//347/253/257/345/210/260/347/253/257LoCoMoQA/345/220/214/345/217/243/345/276/204/345/257/271/347/205/247_v1.0.md +100 -0
- package/docs/eval//347/253/257/345/210/260/347/253/257/345/271/262/346/211/260/346/261/240/350/257/204/346/265/213_/347/241/256/345/256/232/346/200/247/350/243/201/345/206/263vsLLM_judge_v1.1.md +197 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v1.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v10.md +210 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v11.md +227 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v12.md +203 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v13.md +233 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v14.md +191 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v15.md +213 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v16.md +214 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v17.md +199 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v2.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v3.md +152 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v4.md +128 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v5.md +114 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v6.md +192 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v7.md +187 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v8.md +207 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v9.md +203 -0
- package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
- package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
- package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
- package/docs/hive//345/256/211/345/205/250/345/256/241/350/256/241/345/256/236/351/224/232_v0.1.md +202 -0
- package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
- package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
- package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
- package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +519 -503
- package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
- package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
- package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
- package/docs/images/lingshu-moonlight-covenant-poster-preview.jpg +0 -0
- package/docs/images/lingshu-moonlight-covenant-poster.png +0 -0
- package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
- package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
- package/docs/mdcg/lingshu_tutorial.html +14449 -14449
- package/docs/mdcg/release_v0.4.11.md +49 -0
- package/docs/mdcg/release_v0.4.5.md +55 -55
- package/docs/mdcg/tool_table_v0.3.0.md +117 -117
- package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
- package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
- package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
- package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
- package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
- package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
- package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
- package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
- package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
- package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
- package/docs/theory//344/270/215/345/217/257/351/235/240/345/256/232/347/220/206/344/270/216/345/244/261/346/225/210/344/274/230/345/205/210/346/241/206/346/236/266_v0.3.md +365 -0
- package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
- package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
- package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
- package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
- package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
- package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
- package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
- package/docs//344/270/215/345/217/257/351/235/240/346/200/247/347/220/206/350/256/272_v0.1.md +343 -0
- package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
- package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
- package/dsh/README.md +82 -82
- package/dsh/cordis.yml.example +139 -139
- package/dsh/update-lingshu.bat +11 -11
- package/lib/bridge.d.ts +9 -0
- package/lib/bridge.js +35 -0
- package/lib/hooks.js +36 -2
- package/lib/index.js +7 -1
- package/lib/lib/roleplay_web.js +116 -29
- package/lib/lib/token_store.d.ts +7 -1
- package/lib/lib/token_store.js +12 -3
- package/md_cg/__init__.py +7 -7
- package/md_cg/audit.py +379 -368
- package/md_cg/autonomy.py +287 -287
- package/md_cg/backfill.py +1328 -1327
- package/md_cg/backfill_bigdomain.py +34 -34
- package/md_cg/backfill_bucket_zh.py +35 -0
- package/md_cg/bench6_arms.py +410 -410
- package/md_cg/bench6_common.py +230 -230
- package/md_cg/bench6_competitors.py +212 -212
- package/md_cg/bench_axis_domain.py +257 -257
- package/md_cg/bench_blind_comp.py +308 -308
- package/md_cg/bench_e2e_judge.py +532 -0
- package/md_cg/bench_e2e_locomo_qa.py +368 -0
- package/md_cg/bench_e2e_qa.py +256 -0
- package/md_cg/bench_en_atoms_public.py +230 -230
- package/md_cg/bench_governance.py +348 -348
- package/md_cg/bench_lme_zh.py +410 -410
- package/md_cg/bench_locomo.py +121 -121
- package/md_cg/bench_locomo_zh.py +450 -450
- package/md_cg/bench_locomo_zh_public.py +147 -147
- package/md_cg/bench_longmem.py +112 -112
- package/md_cg/bench_membench.py +632 -632
- package/md_cg/bench_p0.py +149 -149
- package/md_cg/bench_progressive.py +287 -287
- package/md_cg/bench_role_views.py +238 -238
- package/md_cg/bench_task_ab.py +243 -243
- package/md_cg/bench_task_ab_llm.py +408 -408
- package/md_cg/bench_unified_en.py +204 -204
- package/md_cg/bench_zh_mad.py +601 -601
- package/md_cg/blindspot_tickets.py +123 -123
- package/md_cg/branches.py +301 -285
- package/md_cg/build_postings.py +73 -73
- package/md_cg/ccgc.py +1006 -948
- package/md_cg/census.py +132 -132
- package/md_cg/chain.py +315 -300
- package/md_cg/codeindex.py +531 -531
- package/md_cg/coldverify.py +292 -292
- package/md_cg/comment_gate.py +337 -337
- package/md_cg/cond_compose.py +190 -190
- package/md_cg/cond_facts.py +154 -154
- package/md_cg/cond_template.json +106 -106
- package/md_cg/condition_anchor.py +142 -142
- package/md_cg/conformance.py +726 -726
- package/md_cg/consistency.py +717 -717
- package/md_cg/consolidate.py +1537 -1439
- package/md_cg/corpus.py +110 -110
- package/md_cg/crosscheck.py +1098 -1097
- package/md_cg/crypto.py +3 -1
- package/md_cg/d_meta.py +310 -310
- package/md_cg/datapath.py +78 -18
- package/md_cg/docindex.py +473 -473
- package/md_cg/eval_common.py +575 -575
- package/md_cg/evidence.py +4 -2
- package/md_cg/evolution.py +477 -477
- package/md_cg/export.py +222 -220
- package/md_cg/forgetting.py +581 -581
- package/md_cg/fsutil.py +377 -329
- package/md_cg/hotcache.py +48 -7
- package/md_cg/hyperedge.py +251 -251
- package/md_cg/identity.py +390 -390
- package/md_cg/insight.py +500 -500
- package/md_cg/interop.py +338 -0
- package/md_cg/judgment_manifest.py +177 -0
- package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
- package/md_cg/lexicon/build_standard_en.py +171 -171
- package/md_cg/lexicon/expand_en_zh.py +211 -211
- package/md_cg/lifecycle.py +272 -272
- package/md_cg/linkref.py +280 -280
- package/md_cg/links.py +140 -107
- package/md_cg/mcp_server.py +403 -55
- package/md_cg/md_whitebox.py +345 -345
- package/md_cg/mdcg.py +783 -233
- package/md_cg/mdcos.py +500 -70
- package/md_cg/metacognition.py +591 -591
- package/md_cg/migrate.py +119 -119
- package/md_cg/migrate_aeis.py +221 -221
- package/md_cg/migrate_roleplay.py +293 -293
- package/md_cg/migrate_wisdom_graph.py +360 -360
- package/md_cg/mreview/__init__.py +25 -25
- package/md_cg/mreview/__main__.py +110 -110
- package/md_cg/mreview/bundle.py +178 -178
- package/md_cg/mreview/candidates.py +262 -262
- package/md_cg/mreview/govern.py +694 -693
- package/md_cg/mreview/locate.py +939 -939
- package/md_cg/mreview/pipeline.py +728 -728
- package/md_cg/mreview/rules/duplication.json +21 -21
- package/md_cg/mreview/rules/field_coverage.json +54 -54
- package/md_cg/mreview/rules/source_license.json +21 -21
- package/md_cg/mreview/rules/template_flow.json +21 -21
- package/md_cg/mreview/ruleset.py +252 -252
- package/md_cg/nodefile.py +575 -575
- package/md_cg/pooling.py +484 -472
- package/md_cg/postings.py +300 -298
- package/md_cg/predict.py +1100 -1100
- package/md_cg/progressive.py +123 -123
- package/md_cg/protect.py +272 -272
- package/md_cg/protocol/md_cg_gate.proto +33 -33
- package/md_cg/protocol.py +372 -372
- package/md_cg/provenance.py +582 -582
- package/md_cg/reach.py +453 -453
- package/md_cg/readcache.py +143 -0
- package/md_cg/reconcile.py +228 -0
- package/md_cg/refindex.py +833 -833
- package/md_cg/refine.py +604 -604
- package/md_cg/review_cli.py +170 -0
- package/md_cg/roleviews.py +89 -89
- package/md_cg/routing.py +393 -365
- package/md_cg/run_tests.py +211 -0
- package/md_cg/scrub.py +13 -3
- package/md_cg/security.py +128 -18
- package/md_cg/self_state.py +1029 -1029
- package/md_cg/selfreport.py +152 -151
- package/md_cg/semantic/__init__.py +10 -10
- package/md_cg/semantic/canonical.py +122 -122
- package/md_cg/semantic/en_normalizer.py +364 -364
- package/md_cg/semantic/en_zh_map.json +28694 -0
- package/md_cg/semantic/export_en_zh_map.py +64 -0
- package/md_cg/semantic/unify.py +45 -0
- package/md_cg/semantic/zh_en_atoms.py +139 -139
- package/md_cg/signer.py +7 -4
- package/md_cg/sources.py +816 -582
- package/md_cg/statushdr.py +179 -179
- package/md_cg/stg.py +54 -37
- package/md_cg/subgraph.py +729 -729
- package/md_cg/sustain.py +35 -5
- package/md_cg/tasks.py +470 -470
- package/md_cg/test_access_hints.py +147 -0
- package/md_cg/test_action_derive.py +203 -203
- package/md_cg/test_audit_rotate.py +270 -270
- package/md_cg/test_autonomy.py +143 -143
- package/md_cg/test_bench_governance.py +102 -102
- package/md_cg/test_blindspot_tickets.py +166 -166
- package/md_cg/test_branch_discard_tombstone.py +136 -0
- package/md_cg/test_branches.py +13 -3
- package/md_cg/test_ccg_perturb.py +184 -184
- package/md_cg/test_ccgc.py +433 -433
- package/md_cg/test_census_prune.py +81 -81
- package/md_cg/test_chain_read_isolate.py +168 -0
- package/md_cg/test_cond_compose_anchors.py +76 -76
- package/md_cg/test_cond_match.py +165 -165
- package/md_cg/test_condition_anchor.py +81 -81
- package/md_cg/test_d_meta.py +412 -412
- package/md_cg/test_datapath_device_name.py +203 -0
- package/md_cg/test_datapath_root.py +199 -199
- package/md_cg/test_emit_negtail_cache.py +156 -0
- package/md_cg/test_en_pipeline.py +22 -2
- package/md_cg/test_gain_gate.py +212 -212
- package/md_cg/test_govern_directread.py +421 -0
- package/md_cg/test_health_scale.py +173 -173
- package/md_cg/test_hive_ingest.py +285 -0
- package/md_cg/test_hot_cold.py +215 -215
- package/md_cg/test_hyperedge.py +245 -245
- package/md_cg/test_i26_empty_first_write.py +116 -0
- package/md_cg/test_i27_e041_identity.py +128 -0
- package/md_cg/test_i28_hotcache_prodpath.py +122 -0
- package/md_cg/test_i32_hotcache_env_key.py +218 -0
- package/md_cg/test_identity_attribution.py +96 -15
- package/md_cg/test_index_durability.py +17 -3
- package/md_cg/test_interop.py +95 -0
- package/md_cg/test_interop_judgment.py +228 -0
- package/md_cg/test_issue39_utf8_stdio.py +273 -0
- package/md_cg/test_lifecycle.py +309 -309
- package/md_cg/test_linkref.py +306 -306
- package/md_cg/test_links_concurrent_write.py +188 -0
- package/md_cg/test_lock.py +43 -43
- package/md_cg/test_md_access_parity.py +255 -255
- package/md_cg/test_md_writepath.py +345 -345
- package/md_cg/test_mdstore_search_parity.py +160 -0
- package/md_cg/test_merge_upsert.py +168 -0
- package/md_cg/test_mr_m2.py +587 -587
- package/md_cg/test_mr_m3.py +710 -710
- package/md_cg/test_mr_m4.py +485 -485
- package/md_cg/test_n123_derive_expiry_chain.py +205 -0
- package/md_cg/test_n130_verify_falsified_protect.py +185 -0
- package/md_cg/test_n131_merge_gate.py +205 -0
- package/md_cg/test_p0.py +250 -250
- package/md_cg/test_p1.py +316 -316
- package/md_cg/test_p10_identity.py +173 -173
- package/md_cg/test_p11_consistency.py +233 -233
- package/md_cg/test_p12_metacognition.py +212 -212
- package/md_cg/test_p13_encryption.py +241 -241
- package/md_cg/test_p14_sustain.py +249 -249
- package/md_cg/test_p15_scrub.py +280 -280
- package/md_cg/test_p16_self_state.py +301 -301
- package/md_cg/test_p17_predict.py +354 -354
- package/md_cg/test_p18_whitebox.py +171 -171
- package/md_cg/test_p19_migrate_roleplay.py +149 -149
- package/md_cg/test_p1x_ref_root.py +160 -0
- package/md_cg/test_p20_evolution.py +315 -315
- package/md_cg/test_p21_tokens.py +293 -270
- package/md_cg/test_p22_theory.py +175 -175
- package/md_cg/test_p23_links.py +311 -311
- package/md_cg/test_p24_evidence.py +227 -227
- package/md_cg/test_p25_weights.py +156 -156
- package/md_cg/test_p26_refindex.py +416 -416
- package/md_cg/test_p27_docindex.py +16 -7
- package/md_cg/test_p28_refcheck.py +305 -305
- package/md_cg/test_p29_session_ingest_export.py +354 -333
- package/md_cg/test_p2_mcp.py +3 -0
- package/md_cg/test_p3.py +11 -2
- package/md_cg/test_p30_maintain.py +330 -330
- package/md_cg/test_p31_insight.py +534 -534
- package/md_cg/test_p32_backfill.py +7 -1
- package/md_cg/test_p33_ccg_wiring.py +293 -293
- package/md_cg/test_p34_crosscheck.py +331 -331
- package/md_cg/test_p35_conditioned_claim.py +252 -252
- package/md_cg/test_p36_kp_align.py +230 -230
- package/md_cg/test_p37_condition_space.py +248 -248
- package/md_cg/test_p38_concurrent_flush.py +102 -0
- package/md_cg/test_p38_contextualize.py +273 -273
- package/md_cg/test_p39_verify_flow.py +153 -0
- package/md_cg/test_p39_vision_evidence.py +369 -369
- package/md_cg/test_p40_refine_worklist.py +241 -241
- package/md_cg/test_p41_evolve_patrol.py +224 -224
- package/md_cg/test_p42_provenance.py +269 -269
- package/md_cg/test_p43_pooling.py +412 -398
- package/md_cg/test_p44_md_whitebox.py +231 -231
- package/md_cg/test_p45_session_identity.py +219 -219
- package/md_cg/test_p46_unit_scope.py +272 -272
- package/md_cg/test_p47_session_view.py +316 -0
- package/md_cg/test_p4_fuzzy.py +223 -223
- package/md_cg/test_p5_semantic.py +226 -226
- package/md_cg/test_p6_consolidate.py +440 -387
- package/md_cg/test_p7_goals_recent.py +202 -202
- package/md_cg/test_p8_subgraph_chain.py +200 -200
- package/md_cg/test_p9_forget_protect.py +231 -231
- package/md_cg/test_predict_beta.py +135 -135
- package/md_cg/test_preflight_failclosed.py +100 -100
- package/md_cg/test_progressive.py +146 -146
- package/md_cg/test_propose_tail_index.py +157 -0
- package/md_cg/test_protocol.py +243 -243
- package/md_cg/test_reach.py +378 -378
- package/md_cg/test_reach_keys.py +201 -201
- package/md_cg/test_read_clip.py +141 -141
- package/md_cg/test_read_scope_b27.py +277 -0
- package/md_cg/test_readcache_default_on.py +168 -0
- package/md_cg/test_readcache_precise_inval.py +270 -0
- package/md_cg/test_readcache_prodpath.py +203 -0
- package/md_cg/test_reconcile_v0.py +294 -0
- package/md_cg/test_retr_gates_prodpath.py +140 -0
- package/md_cg/test_retr_s1.py +344 -340
- package/md_cg/test_retr_s1b.py +276 -209
- package/md_cg/test_retr_s3.py +194 -194
- package/md_cg/test_retr_s4.py +163 -163
- package/md_cg/test_retr_s5.py +200 -200
- package/md_cg/test_retr_s6.py +157 -157
- package/md_cg/test_retr_s7.py +392 -384
- package/md_cg/test_retr_s8_time.py +369 -316
- package/md_cg/test_retr_s9_edges.py +286 -286
- package/md_cg/test_retr_s9_entity_ctx.py +9 -3
- package/md_cg/test_retr_score_once.py +208 -0
- package/md_cg/test_review_conformance.py +367 -367
- package/md_cg/test_review_onepass.py +170 -0
- package/md_cg/test_role_views.py +354 -354
- package/md_cg/test_rrf_graph_seed_cache.py +154 -0
- package/md_cg/test_security_audit.py +155 -0
- package/md_cg/test_security_audit_b26.py +161 -0
- package/md_cg/test_security_audit_v21.py +250 -0
- package/md_cg/test_sem_noise.py +242 -242
- package/md_cg/test_semantic_canonical.py +16 -2
- package/md_cg/test_session_isolation.py +168 -0
- package/md_cg/test_snapshot_autoclose.py +187 -0
- package/md_cg/test_subproc_encoding.py +192 -192
- package/md_cg/test_sustain_mutual.py +153 -153
- package/md_cg/test_tail_watermark_race.py +208 -0
- package/md_cg/test_tasks.py +409 -409
- package/md_cg/test_tenant_env_override_warn.py +139 -0
- package/md_cg/test_tenant_registry_corrupt_warn.py +151 -0
- package/md_cg/test_tool_face.py +189 -189
- package/md_cg/test_transfer.py +180 -180
- package/md_cg/test_trust.py +361 -361
- package/md_cg/test_twophase.py +286 -286
- package/md_cg/test_v14_fixes.py +38 -20
- package/md_cg/test_validity_filter.py +280 -280
- package/md_cg/test_verify_answer.py +138 -138
- package/md_cg/test_verify_dirty_reconcile.py +157 -0
- package/md_cg/test_wisdom_md_store.py +292 -292
- package/md_cg/test_writelimit.py +197 -197
- package/md_cg/test_writepipe.py +214 -214
- package/md_cg/theory.py +6 -3
- package/md_cg/tokens.py +85 -14
- package/md_cg/tool_face.py +260 -260
- package/md_cg/trust.py +986 -950
- package/md_cg/twophase.py +231 -231
- package/md_cg/units.py +668 -667
- package/md_cg/vision_evidence.py +667 -666
- package/md_cg/weights.py +624 -624
- package/md_cg/whitebox.py +527 -527
- package/md_cg/whitebox_kb/__init__.py +37 -37
- package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
- package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
- package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
- package/md_cg/whitebox_kb/engine.py +310 -310
- package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
- package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
- package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
- package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
- package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
- package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
- package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
- package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
- package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
- package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
- package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
- package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
- package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
- package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
- package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
- package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
- package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
- package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
- package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
- package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
- package/md_cg/whitebox_kb/wisdom/multilang_ir.py +128 -124
- package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
- package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
- package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
- package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
- package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
- package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
- package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
- package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
- package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
- package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
- package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
- package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
- package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
- package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
- package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
- package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
- package/md_cg/writelimit.py +356 -356
- package/md_cg/writepipe.py +20 -8
- package/package.json +101 -96
- package/skills/plugin.json +54 -54
- package/skills/skills/designer-perspective/SKILL.md +158 -158
- package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
- package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
- package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
- package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
- package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
- package/skills/skills/designer-perspective/scripts/designer.py +545 -545
- package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
- package/skills/skills/designer-perspective/tests/selftest.py +61 -61
- package/skills/skills/lingshu-browser/SKILL.md +60 -60
- package/skills/skills/lingshu-compiler/SKILL.md +56 -56
- package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
- package/skills/skills/lingshu-graph/SKILL.md +63 -63
- package/skills/skills/lingshu-net/SKILL.md +48 -48
- package/skills/skills/lingshu-os/SKILL.md +64 -64
- package/skills/skills/lingshu-pylang/SKILL.md +71 -71
- package/src/bridge.ts +33 -0
- package/src/hooks.ts +38 -2
- package/src/index.ts +526 -518
- package/src/lib/datapath.ts +326 -326
- package/src/lib/mdcg_client.ts +413 -413
- package/src/lib/mutual.ts +428 -428
- package/src/lib/prompt_safety.ts +62 -62
- package/src/lib/python_path.ts +71 -71
- package/src/lib/roleplay_web.ts +116 -29
- package/src/lib/token_store.ts +13 -3
- package/src/tools.ts +212 -212
- package/zcode/AGENTS.md +11 -3
- package/zcode/README.md +41 -41
- /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
package/md_cg/docindex.py
CHANGED
|
@@ -1,474 +1,474 @@
|
|
|
1
|
-
# -*- coding: utf-8 -*-
|
|
2
|
-
"""条件文档图:按「章节」索引 md 文档,不存全文。
|
|
3
|
-
|
|
4
|
-
设计(2026-09-10):
|
|
5
|
-
`docs/` 下的 md 是**规范/方案的唯一事实源**,认知图只需要「哪一份文档、哪一节、
|
|
6
|
-
哪几行」这一级坐标,不需要第二份全文(否则文档一改就有两份真相,且必然漂移)。
|
|
7
|
-
因此本模块与 `codeindex` 同构:切块 → 渲染 CCG → frontmatter.doc_ref 指回原文;
|
|
8
|
-
正文用 `op=ref` 回读。
|
|
9
|
-
|
|
10
|
-
切块纪律(对应计划 §八 的风险项):
|
|
11
|
-
· **只切 level<=3**:再深就过细,节点数爆炸且检索噪声上升;
|
|
12
|
-
· 直接正文 < `MIN_BODY` 字且**无子节**的小节**合并进父节**(不单独建节点,
|
|
13
|
-
其文字追加进父节摘要),否则会把「一行小标题」也变成一个节点;
|
|
14
|
-
· **不存全文**:节点正文是 CCG 模板 + 摘要,正文一律回读。
|
|
15
|
-
|
|
16
|
-
md 解析的两处硬约束:
|
|
17
|
-
· **围栏代码块内的 `#` 不是标题**:`docs/` 里大量 python/shell 片段带 `#` 注释,
|
|
18
|
-
若不做围栏跟踪,一节会被切得七零八落(假标题、错行号);
|
|
19
|
-
· **正文里的 `---` 不参与 frontmatter 切分**:这条纪律在 `nodefile.py` 已定,
|
|
20
|
-
本模块额外保证开头 YAML frontmatter 不被当成正文索引,其余 `---` 只当正文。
|
|
21
|
-
|
|
22
|
-
节点正文必须是 **CCG 6 行**(见 `render`):非 CCG 正文会被 `judge_qualification`
|
|
23
|
-
的第一步(ccg_completeness)直接判 **BLINDSPOT**,文档节点会「存得进、判不了、
|
|
24
|
-
检索不到」——这与改造前的 `codeindex` 是同一个坑。
|
|
25
|
-
|
|
26
|
-
区间哈希复用 `codeindex.region_hash`(**唯一实现**),索引侧与 `op=ref` 回读侧共用。
|
|
27
|
-
"""
|
|
28
|
-
from __future__ import annotations
|
|
29
|
-
|
|
30
|
-
import hashlib
|
|
31
|
-
import os
|
|
32
|
-
import re
|
|
33
|
-
|
|
34
|
-
from . import codeindex, nodefile
|
|
35
|
-
|
|
36
|
-
SKIP_DIRS = ("__pycache__", ".git", ".venv", "venv", "node_modules", ".mypy_cache")
|
|
37
|
-
|
|
38
|
-
SUFFIX = (".md", ".markdown")
|
|
39
|
-
|
|
40
|
-
MAX_LEVEL = 3 # 只切 level<=3
|
|
41
|
-
MIN_BODY = 200 # 直接正文 < 200 字且无子节 → 合并进父节
|
|
42
|
-
MAX_SUMMARY = 200 # 「执行」栏摘要上限
|
|
43
|
-
MAX_DOC = 400
|
|
44
|
-
|
|
45
|
-
KIND = "section"
|
|
46
|
-
LANG = "md"
|
|
47
|
-
BASIS = "data" # 文档的验证基底:以原始文档为准
|
|
48
|
-
|
|
49
|
-
# ---- 密级(计划 §1.3-3 的裁定,2026-09-10)--------------------------------
|
|
50
|
-
# 裁定一:layer 默认 knowledge。理由——文档是**可回读、可漂移检测**的参照知识,
|
|
51
|
-
# 与代码节点同层,保证进默认召回;contextual 表示情境绑定、会过期,
|
|
52
|
-
# 用在这里会让文档掉出默认召回。
|
|
53
|
-
# 裁定二:密级**默认 internal 并显式写入 frontmatter**,不依赖节点默认值
|
|
54
|
-
# (mdcos 读隔离取的是 fm.sensitivity;不显式写就等于「靠默认值兜底」,
|
|
55
|
-
# 审计时看不出意图)。且路径段命中私有提示时**再保守一档降为 private**:
|
|
56
|
-
# 宁可漏召回,不可泄漏(计划 §八 的风险项)。
|
|
57
|
-
DEFAULT_SENSITIVITY = "internal"
|
|
58
|
-
PRIVATE_HINTS = ("private", "secret", "internal", "未公开", "私有", "内部")
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
# 生效条件:override 为真值即返回 (override, "调用方显式指定");否则 path(假值按 "")按 "\" 与 "/" 分段,任一段小写含 PRIVATE_HINTS 中任一提示即返回 ("private", 路径段命中理由),全部不命中返回 (DEFAULT_SENSITIVITY, 默认密级理由);
|
|
62
|
-
def sensitivity_for(path, override=None):
|
|
63
|
-
"""返回 (密级, 依据)。override 优先;否则按路径段保守降级。
|
|
64
|
-
|
|
65
|
-
只可能**更严**、不可能更松:命中提示只会把 internal 收紧为 private,
|
|
66
|
-
不会把 private 放开成 public。缺省值显式返回,便于调用方落盘与审计。
|
|
67
|
-
"""
|
|
68
|
-
if override:
|
|
69
|
-
return override, "调用方显式指定"
|
|
70
|
-
for seg in (path or "").replace("\\", "/").split("/"):
|
|
71
|
-
low = seg.lower()
|
|
72
|
-
for hint in PRIVATE_HINTS:
|
|
73
|
-
if hint in low:
|
|
74
|
-
return "private", f"路径段「{seg}」命中私有提示 → 保守降级"
|
|
75
|
-
return DEFAULT_SENSITIVITY, "默认密级(显式写入,不依赖节点默认值)"
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
# --------------------------------------------------------------------------
|
|
79
|
-
# md 解析
|
|
80
|
-
# --------------------------------------------------------------------------
|
|
81
|
-
_FENCE = re.compile(r"^\s*(```+|~~~+)")
|
|
82
|
-
_ATX = re.compile(r"^(#{1,6})\s+(.+?)\s*#*\s*$")
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
# 生效条件:lines 非空且 lines[0].strip() == "---" 时,从下标 1 起找到首个 strip() == "---" 的行并返回其后一行下标 i+1;lines 为空、首行不是 "---" 或找不到闭合 "---" 时返回 0;
|
|
86
|
-
def _body_start(lines):
|
|
87
|
-
"""跳过开头 YAML frontmatter,返回正文起始行下标(0 基)。"""
|
|
88
|
-
if lines and lines[0].strip() == "---":
|
|
89
|
-
for i in range(1, len(lines)):
|
|
90
|
-
if lines[i].strip() == "---":
|
|
91
|
-
return i + 1
|
|
92
|
-
return 0
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
# 生效条件:从 start(默认 0)遍历 lines,未处于围栏时遇 _FENCE 匹配行打开同标记围栏、围栏内遇同标记行关闭并继续;围栏外 _ATX 匹配行追加 {level: 一级 # 个数, title: 去空白后的标题, lineno: i+1};返回 out 列表;start 不小于 len(lines) 时返回空列表;
|
|
96
|
-
def _headings(lines, start=0):
|
|
97
|
-
"""产出 ATX 标题 `{level,title,lineno}`;**围栏代码块内的 `#` 不算标题**。"""
|
|
98
|
-
out, fence = [], None
|
|
99
|
-
for i in range(start, len(lines)):
|
|
100
|
-
line = lines[i]
|
|
101
|
-
m = _FENCE.match(line)
|
|
102
|
-
if m:
|
|
103
|
-
mark = m.group(1)[0]
|
|
104
|
-
if fence is None:
|
|
105
|
-
fence = mark
|
|
106
|
-
elif fence == mark:
|
|
107
|
-
fence = None
|
|
108
|
-
continue
|
|
109
|
-
if fence is not None:
|
|
110
|
-
continue
|
|
111
|
-
m = _ATX.match(line)
|
|
112
|
-
if m:
|
|
113
|
-
out.append({"level": len(m.group(1)), "title": m.group(2).strip(),
|
|
114
|
-
"lineno": i + 1})
|
|
115
|
-
return out
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
# 生效条件:title 真值时取其 strip 后小写,删去 `*[]() 与除 \w\s- 外字符,再把空白/下划线连成 "-" 并 strip("-");title 为假值(含 None、空串)时按空串处理并返回 "";
|
|
119
|
-
def _anchor(title):
|
|
120
|
-
a = (title or "").strip().lower()
|
|
121
|
-
a = re.sub(r"`|\*|\[|\]|\(|\)", "", a)
|
|
122
|
-
a = re.sub(r"[^\w\s-]", "", a, flags=re.UNICODE) # CJK 属 \w,保留
|
|
123
|
-
return re.sub(r"[\s_]+", "-", a).strip("-")
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
# 生效条件:region_lines 逐行 strip 后跳过空行与 "---",去掉行首 #+ 和 >|*- 标记,以空格连接成 text;返回 text[:limit](limit 默认 MAX_SUMMARY;limit=0 返回 "",limit=None 返回全文,limit='' 时切片抛 TypeError,负 limit 按负索引切片);
|
|
127
|
-
def _summary(region_lines, limit=MAX_SUMMARY):
|
|
128
|
-
"""把一段正文压成一行摘要(去 markdown 噪声,不逐字保留)。"""
|
|
129
|
-
parts = []
|
|
130
|
-
for ln in region_lines:
|
|
131
|
-
s = ln.strip()
|
|
132
|
-
if not s or s == "---":
|
|
133
|
-
continue
|
|
134
|
-
s = re.sub(r"^#+\s*", "", s)
|
|
135
|
-
s = re.sub(r"^[>|*-]\s*", "", s)
|
|
136
|
-
parts.append(s)
|
|
137
|
-
text = " ".join(parts).strip()
|
|
138
|
-
return text[:limit]
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
# 生效条件:以 heads[i]["level"]-1 为需匹配层级向前回溯,返回按层级递减补齐的祖先标题列表(不含 heads[i] 自身)。
|
|
142
|
-
def _path_titles(heads, i):
|
|
143
|
-
"""第 i 个标题的祖先链(不含自身),按层级补齐。"""
|
|
144
|
-
out, need = [], heads[i]["level"] - 1
|
|
145
|
-
for k in range(i - 1, -1, -1):
|
|
146
|
-
if heads[k]["level"] == need:
|
|
147
|
-
out.insert(0, heads[k]["title"])
|
|
148
|
-
need -= 1
|
|
149
|
-
if need == 0:
|
|
150
|
-
break
|
|
151
|
-
return out
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
# 生效条件:直接以 lines、lineno、end 调用 codeindex.region_hash 并返回其结果;
|
|
155
|
-
def _region_hash(lines, lineno, end):
|
|
156
|
-
# 唯一实现复用 codeindex.region_hash:两侧各写一份,漂移检测会悄悄失效。
|
|
157
|
-
return codeindex.region_hash(lines, lineno, end)
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
# 生效条件:ext = suffix 真值时原样使用的 suffix,否则取 os.path.splitext(path)[1].lower();ext 不在 SUFFIX 时抛 ValueError;在 SUFFIX 时把 source 按 "\n" 拆分,经 _body_start 与 _headings 得到标题,仅 level<=MAX_LEVEL 且非 small 的标题生成条目,小/过深子节摘要并入父摘要,返回 items;
|
|
161
|
-
def extract(source, path="", suffix=None):
|
|
162
|
-
"""抽取一份 md 的章节条目;按后缀分派。返回条目列表(可能为空)。
|
|
163
|
-
|
|
164
|
-
条目字段与 `codeindex` 对齐(name/kind/lineno/end/hash/lang/precise/basis),
|
|
165
|
-
另带文档专有:heading / heading_path / level / anchor / children。
|
|
166
|
-
"""
|
|
167
|
-
ext = suffix or os.path.splitext(path)[1].lower()
|
|
168
|
-
if ext not in SUFFIX:
|
|
169
|
-
raise ValueError(f"无文档提取器(suffix={ext or '<none>'})")
|
|
170
|
-
lines = source.split("\n")
|
|
171
|
-
heads = _headings(lines, _body_start(lines))
|
|
172
|
-
|
|
173
|
-
# info 以 heads 序号为键:children 存的是 heads 序号,不能拿去过 secs 的下标。
|
|
174
|
-
info = {}
|
|
175
|
-
for i, h in enumerate(heads):
|
|
176
|
-
end = len(lines)
|
|
177
|
-
children = []
|
|
178
|
-
for j in range(i + 1, len(heads)):
|
|
179
|
-
if heads[j]["level"] <= h["level"]:
|
|
180
|
-
end = heads[j]["lineno"] - 1
|
|
181
|
-
break
|
|
182
|
-
children.append(j)
|
|
183
|
-
direct_end = (heads[i + 1]["lineno"] - 1) if i + 1 < len(heads) else len(lines)
|
|
184
|
-
direct = lines[h["lineno"]:max(h["lineno"], min(direct_end, end))]
|
|
185
|
-
info[i] = {"end": max(h["lineno"], end), "children": children, "direct": direct,
|
|
186
|
-
"small": len(_summary(direct)) < MIN_BODY and not children}
|
|
187
|
-
|
|
188
|
-
items, seen = [], {}
|
|
189
|
-
for i, h in enumerate(heads):
|
|
190
|
-
if h["level"] > MAX_LEVEL or info[i]["small"]:
|
|
191
|
-
continue # 合并进父节点:父节点的 end 已覆盖其区间
|
|
192
|
-
summary = _summary(info[i]["direct"])
|
|
193
|
-
# 被合并进来的子节(自身过小,或层级过深从不单独建节点):文字并入父节摘要,
|
|
194
|
-
# 否则这些小节的正文只存在于父节的 ref 区间里,检索不到。
|
|
195
|
-
merged = [_summary(info[k]["direct"], 80) for k in info[i]["children"]
|
|
196
|
-
if info[k]["small"] or heads[k]["level"] > MAX_LEVEL]
|
|
197
|
-
merged = [m for m in merged if m]
|
|
198
|
-
if merged:
|
|
199
|
-
summary = (summary + ";" + ";".join(merged))[:MAX_SUMMARY]
|
|
200
|
-
if not summary:
|
|
201
|
-
summary = "(该节无直接正文,见子节)"
|
|
202
|
-
parent = _path_titles(heads, i)
|
|
203
|
-
heading_path = parent + [h["title"]]
|
|
204
|
-
key = path + "#" + "/".join(heading_path)
|
|
205
|
-
dup = seen.get(key, 0) + 1
|
|
206
|
-
seen[key] = dup
|
|
207
|
-
item = {
|
|
208
|
-
"path": path, "name": h["title"], "heading": h["title"], "kind": KIND,
|
|
209
|
-
"heading_path": heading_path, "level": h["level"],
|
|
210
|
-
"anchor": _anchor(h["title"]), "parent": parent[-1] if parent else "",
|
|
211
|
-
"lineno": h["lineno"], "end": info[i]["end"],
|
|
212
|
-
"summary_parts": summary,
|
|
213
|
-
"children": [heads[k]["title"] for k in info[i]["children"]],
|
|
214
|
-
"dup": dup, "lang": LANG, "precise": True, "basis": BASIS,
|
|
215
|
-
}
|
|
216
|
-
item["hash"] = _region_hash(lines, item["lineno"], item["end"])
|
|
217
|
-
items.append(item)
|
|
218
|
-
return items
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
# --------------------------------------------------------------------------
|
|
222
|
-
# 渲染 / id
|
|
223
|
-
# --------------------------------------------------------------------------
|
|
224
|
-
# 生效条件:item.get("path") 缺键或为假值时 path 取 "",top 取 path.split("/")[0] or "."(故空 path 时 top=".");path 为真值时 top 取其 "/" 前首段,首段为空则 top=".";返回含 observation_position(大域=top)、time_window([nodefile.FULL_TIME_WINDOW_MIN, nodefile.FULL_TIME_WINDOW_MAX])、observation_tool、existence_constraint(以 path 拼入)的四槽字典;
|
|
225
|
-
def condition_space(item):
|
|
226
|
-
"""章节条目 → 条件空间四槽(纯函数,**唯一来源**)。
|
|
227
|
-
|
|
228
|
-
与 `codeindex.condition_space` 同一职责、同一理由:`render` 的正文行与
|
|
229
|
-
`refindex.add_items` 的 frontmatter 必须同源,否则 frontmatter 只剩单槽
|
|
230
|
-
`observation_position`,`nodefile.condition_space_text(require_full=True)`
|
|
231
|
-
恒返回 "" —— 条件空间等于没声明。改造前正文写的是「文档=X;检索…时」,
|
|
232
|
-
是第三种方言,既进不了条件空间,也不可被 `_slot_overlap` 使用。
|
|
233
|
-
|
|
234
|
-
时间槽给全时窗哨兵:文档章节条目声明的是「该文档里有这一节」,
|
|
235
|
-
真值不随索引时刻衰减,不写成 1 小时观测窗。
|
|
236
|
-
"""
|
|
237
|
-
path = item.get("path") or ""
|
|
238
|
-
top = path.split("/")[0] or "."
|
|
239
|
-
return {
|
|
240
|
-
"observation_position": f"本地文档仓(大域={top})",
|
|
241
|
-
"time_window": [nodefile.FULL_TIME_WINDOW_MIN,
|
|
242
|
-
nodefile.FULL_TIME_WINDOW_MAX],
|
|
243
|
-
"observation_tool": (f"{LANG}(md 章节切分,level≤{MAX_LEVEL};"
|
|
244
|
-
f"只存标题+摘要,正文留在源文件)"),
|
|
245
|
-
"existence_constraint": f"源文档 {path} 存在于本地仓且可读",
|
|
246
|
-
}
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
# 生效条件:item 含 heading、path、lineno、end、anchor 键时渲染 7 行 CCG 文本(第2行取 condition_space(item) 文本),item.get("parent") 或 "" 假值回落 "(顶层章节)",item.get("summary_parts") 假值回落 "(该节无直接正文,见子节)",item.get("children") 假值回落空列表且不追加子节行,children 非空时追加 "# 子节:" + 前 12 个;返回以 "\n" 连接的行串;
|
|
250
|
-
def render(item):
|
|
251
|
-
"""章节条目 → CCG 6 行正文(可被 search 命中,不含全文)。
|
|
252
|
-
|
|
253
|
-
**必须渲染成 CCG**:`judge_qualification` 第一步查 ccg_completeness 的 5 要素,
|
|
254
|
-
缺任一即直接判 BLINDSPOT(与 codeindex.render 同一个坑)。
|
|
255
|
-
"""
|
|
256
|
-
heading = item["heading"]
|
|
257
|
-
path = item["path"]
|
|
258
|
-
parent = item.get("parent") or ""
|
|
259
|
-
summary = item.get("summary_parts") or "(该节无直接正文,见子节)"
|
|
260
|
-
sub = f"父章节:{parent}" if parent else "(顶层章节)"
|
|
261
|
-
children = item.get("children") or []
|
|
262
|
-
lines = [
|
|
263
|
-
f"# 功能名:{heading}",
|
|
264
|
-
f"# 生效条件:{nodefile.condition_space_text(condition_space(item))}",
|
|
265
|
-
f"# 子功能:{sub}",
|
|
266
|
-
f"# 执行:{summary[:MAX_DOC]}",
|
|
267
|
-
(f"# 验证方式:{BASIS}(以原始文档为准;"
|
|
268
|
-
f"区间 {path} L{item['lineno']}-L{item['end']})"),
|
|
269
|
-
f"# 不适用条件:其它文档的同名标题(本条目属于 {path}#{item['anchor']})",
|
|
270
|
-
f"# 位置:{path}#{item['anchor']}:{item['lineno']}-{item['end']}"
|
|
271
|
-
f"({item.get('lang')},precise=True)",
|
|
272
|
-
]
|
|
273
|
-
if children:
|
|
274
|
-
lines.append("# 子节:" + ";".join(children[:12]))
|
|
275
|
-
return "\n".join(lines)
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
# 生效条件:以 item["path"] + "#" + "/".join(item["heading_path"]) 为 key,item.get("dup", 1)(缺键取 1)大于 1 时追加 "#" + item["dup"],返回 "doc_" + sha1(key utf-8) hexdigest 前 12 位;
|
|
279
|
-
def node_id(item):
|
|
280
|
-
"""稳定 id:path#heading_path 的短哈希(重复索引幂等;同名用 dup 区分)。"""
|
|
281
|
-
key = item["path"] + "#" + "/".join(item["heading_path"])
|
|
282
|
-
if item.get("dup", 1) > 1:
|
|
283
|
-
key += f"#{item['dup']}"
|
|
284
|
-
return "doc_" + hashlib.sha1(key.encode("utf-8")).hexdigest()[:12]
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
# ==========================================================================
|
|
288
|
-
# fence 真源绑定(§3.5):条件卡 ↔ Markdown 真源的「往返列」
|
|
289
|
-
#
|
|
290
|
-
# 往返列 = 真源(root + path)+ 定位(anchor / heading_path)+ 行位(lineno/end)
|
|
291
|
-
# + 校验(hash)。**行位是易腐化量,键不含行位**:在章节上方插一段话会让整篇行号
|
|
292
|
-
# 位移,但键 `path#heading_path` 不变——故「同一章节」的匹配一律走键,行位只用于
|
|
293
|
-
# 回读原文与漂移判定。写侧唯一入口是 `refindex._doc_ref`(与索引同源,避免第二份
|
|
294
|
-
# 列口径);本节只补读侧:取列 / 校验 / 造键 / 反查行位。
|
|
295
|
-
# · 正向:卡片 → 真源(`doc_ref.lineno/end` + `refindex.read_ref` 回读原文区间)
|
|
296
|
-
# · 反向:真源行 → 卡片(`locate`;对账器 `--at path:line` 即此接口)
|
|
297
|
-
# ==========================================================================
|
|
298
|
-
|
|
299
|
-
#: 必需列:真源(root/path)+ 定位(anchor)+ 行位(lineno/end)+ 校验(hash)
|
|
300
|
-
BINDING_FIELDS = ("root", "path", "anchor", "lineno", "end", "hash")
|
|
301
|
-
#: 附加列:有则参与更精确匹配与人读,缺省不算绑定失败
|
|
302
|
-
BINDING_OPTIONAL = ("heading_path", "heading", "level", "lang", "precise")
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
# 生效条件:node 为 cg.get 产物(含 frontmatter 子字典)或 frontmatter dict 本身,且其 doc_ref 为 dict 时,返回按 BINDING_FIELDS + BINDING_OPTIONAL 投影的列值 dict(缺列以 None 占位);非 dict / 无 doc_ref 返回 None。
|
|
306
|
-
def binding_of(node):
|
|
307
|
-
"""取一条条件卡的往返列;**非索引节点返回 None**(不猜、不补默认值)。"""
|
|
308
|
-
fm = node
|
|
309
|
-
if isinstance(node, dict) and isinstance(node.get("frontmatter"), dict):
|
|
310
|
-
fm = node["frontmatter"]
|
|
311
|
-
if not isinstance(fm, dict):
|
|
312
|
-
return None
|
|
313
|
-
ref = fm.get("doc_ref")
|
|
314
|
-
if not isinstance(ref, dict) or not ref:
|
|
315
|
-
return None
|
|
316
|
-
return {k: ref.get(k) for k in BINDING_FIELDS + BINDING_OPTIONAL}
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
# 生效条件:b 为 dict 且 heading_path 为非空 list/tuple 时返回 path#join(heading_path, '/'),否则回落 path#anchor;b 非 dict 返回空串。
|
|
320
|
-
def binding_key(b):
|
|
321
|
-
"""稳定键 `path#heading_path`:**不含行位**(真源重排只动行位、不动键)。"""
|
|
322
|
-
if not isinstance(b, dict):
|
|
323
|
-
return ""
|
|
324
|
-
path = str(b.get("path") or "")
|
|
325
|
-
hp = b.get("heading_path")
|
|
326
|
-
if isinstance(hp, (list, tuple)) and hp:
|
|
327
|
-
return path + "#" + "/".join(str(x) for x in hp)
|
|
328
|
-
return path + "#" + str(b.get("anchor") or "")
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
# 生效条件:b 为 dict 时返回 'path#anchor'(与 render 正文里的「本条目属于」逐字同源);非 dict 返回空串。
|
|
332
|
-
def binding_slug(b):
|
|
333
|
-
"""人读定位串 `path#anchor`——与 `render` 正文里的「本条目属于」同源。"""
|
|
334
|
-
b = b if isinstance(b, dict) else {}
|
|
335
|
-
return f"{b.get('path') or ''}#{b.get('anchor') or ''}"
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
# 生效条件:b 为 dict 时逐列校验(缺列 / 类型错 / 行位越界),返回 {"ok": bool, "issues": [str, ...]};**不抛异常**——对账要逐条报告,不能因一条坏数据中断全库;行位仅在列齐且类型对时才判,避免级联噪声。
|
|
339
|
-
def validate_binding(b):
|
|
340
|
-
"""校验往返列形状。**不抛异常**:对账逐条报告,不能一条坏数据中断全库。"""
|
|
341
|
-
if not isinstance(b, dict):
|
|
342
|
-
return {"ok": False, "issues": ["绑定不是字典(该节点无 doc_ref)"]}
|
|
343
|
-
issues = []
|
|
344
|
-
for k in BINDING_FIELDS:
|
|
345
|
-
if b.get(k) in (None, ""):
|
|
346
|
-
issues.append(f"缺列 {k}")
|
|
347
|
-
for k in ("root", "path", "anchor", "hash"):
|
|
348
|
-
v = b.get(k)
|
|
349
|
-
if v not in (None, "") and not isinstance(v, str):
|
|
350
|
-
issues.append(f"{k} 应为字符串,实为 {type(v).__name__}")
|
|
351
|
-
if not issues:
|
|
352
|
-
try:
|
|
353
|
-
lo, hi = int(b["lineno"]), int(b["end"])
|
|
354
|
-
except (TypeError, ValueError):
|
|
355
|
-
issues.append("行位不是整数")
|
|
356
|
-
else:
|
|
357
|
-
if lo < 1:
|
|
358
|
-
issues.append(f"lineno 越界({lo} < 1)")
|
|
359
|
-
if hi < lo:
|
|
360
|
-
issues.append(f"end 早于 lineno({lo}-{hi})")
|
|
361
|
-
return {"ok": not issues, "issues": issues}
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
# 生效条件:old/new 任意为 dict 或假值;按 (path, anchor, heading_path, lineno, end, hash) 固定顺序返回取值不同的列名列表(空列表=同一章节同一行位),任一侧假值按空 dict 处理。
|
|
365
|
-
def binding_drift(old, new):
|
|
366
|
-
"""旧往返列 vs 新往返列 → 变化列名(顺序固定,供报告与测试断言)。"""
|
|
367
|
-
old, new = old if isinstance(old, dict) else {}, new if isinstance(new, dict) else {}
|
|
368
|
-
return [k for k in ("path", "anchor", "heading_path", "lineno", "end", "hash")
|
|
369
|
-
if old.get(k) != new.get(k)]
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
# 生效条件:items 为 extract 产出的条目序列、lineno 可转 int;返回覆盖该行的条目中 **lineno 最大者**(嵌套即最内层,h3 优于其父 h2);不可转值或无可覆盖条目返回 None。
|
|
373
|
-
def locate(items, lineno):
|
|
374
|
-
"""反向定位:真源第 `lineno` 行 → 覆盖它的条目(嵌套取**最内层**)。
|
|
375
|
-
|
|
376
|
-
区间闭合:标题行算本节;多层嵌套时取 `lineno` 最大者即最内层。
|
|
377
|
-
"""
|
|
378
|
-
try:
|
|
379
|
-
ln = int(lineno)
|
|
380
|
-
except (TypeError, ValueError):
|
|
381
|
-
return None
|
|
382
|
-
hit = None
|
|
383
|
-
for it in items or []:
|
|
384
|
-
lo, hi = it.get("lineno"), it.get("end")
|
|
385
|
-
if lo is None or hi is None:
|
|
386
|
-
continue
|
|
387
|
-
try:
|
|
388
|
-
lo, hi = int(lo), int(hi)
|
|
389
|
-
except (TypeError, ValueError):
|
|
390
|
-
continue
|
|
391
|
-
if lo <= ln <= hi and (hit is None or lo > int(hit["lineno"])):
|
|
392
|
-
hit = it
|
|
393
|
-
return hit
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
# 生效条件:以 root 为根 os.walk,patterns 假值回落 SUFFIX;files 达到 max_files 或 items 达到 max_items 时提前返回并置 stats["truncated"]/truncated_reason;fresh 非 None 且 fresh(rel, fp) 为真时跳过该文件读盘并计 skipped_unchanged;on_file 非 None 且 open/extract 成功后以 (rel, fp, got) 回调;名字在 SKIP_DIRS 的目录仅剪枝不记录,skip_dirs 经 codeindex.skip_matcher 命中的目录剪枝并记入 stats["skipped_dirs"];返回 (items, errors, stats)。
|
|
397
|
-
def index_dir(root, patterns=None, max_files=500, max_items=2000,
|
|
398
|
-
fresh=None, on_file=None, skip_dirs=None):
|
|
399
|
-
"""按大域(目录)遍历 md,产出 `(items, errors, stats)`。零 LLM。
|
|
400
|
-
|
|
401
|
-
stats 语义与 `codeindex.index_dir` 一致:`truncated`/`truncated_reason` 显式上报
|
|
402
|
-
(截断不静默),`skipped_suffixes` 列出扫到但没被索引的后缀(覆盖缺口可审计)。
|
|
403
|
-
|
|
404
|
-
`fresh(rel, fp)` / `on_file(rel, fp, items)` 是给 `refindex.Ledger` 留的增量钩子
|
|
405
|
-
(默认 None → 行为与改造前逐字一致):未变文件不读盘、计入 `skipped_unchanged`。
|
|
406
|
-
|
|
407
|
-
`skip_dirs` 是**追加**排除,复用 `codeindex.skip_matcher`(唯一实现,避免两条
|
|
408
|
-
索引链路口径漂移):命中的目录整棵剪掉、不计入 `files`。本仓的实例就是
|
|
409
|
-
`docs/experiments/`——`.gitignore` 已整目录忽略、物理却仍有 2358 个 md 的实验
|
|
410
|
-
产物,会把 `max_files` 撑爆并把「索引不全」变成常态。排掉了哪些目录写进
|
|
411
|
-
`stats["skipped_dirs"]`,排除与截断一样**不许静默**。
|
|
412
|
-
"""
|
|
413
|
-
pats = tuple(patterns or SUFFIX)
|
|
414
|
-
hit_skip, skip_rules = codeindex.skip_matcher(skip_dirs)
|
|
415
|
-
items, errors, files = [], [], 0
|
|
416
|
-
seen_suffix = set()
|
|
417
|
-
stats = {"root": root, "patterns": list(pats), "files": 0, "truncated": False,
|
|
418
|
-
"truncated_reason": "", "max_files": max_files, "max_items": max_items,
|
|
419
|
-
"skipped_suffixes": [], "skipped_unchanged": 0,
|
|
420
|
-
"skip_dirs": list(skip_rules), "skipped_dirs": []}
|
|
421
|
-
for dirpath, dirnames, filenames in os.walk(root):
|
|
422
|
-
rel_dir = os.path.relpath(dirpath, root).replace("\\", "/")
|
|
423
|
-
if rel_dir == ".":
|
|
424
|
-
rel_dir = ""
|
|
425
|
-
keep = []
|
|
426
|
-
for d in dirnames:
|
|
427
|
-
if d in SKIP_DIRS:
|
|
428
|
-
continue
|
|
429
|
-
child = f"{rel_dir}/{d}" if rel_dir else d
|
|
430
|
-
if hit_skip is not None and hit_skip(child, d):
|
|
431
|
-
stats["skipped_dirs"].append(child)
|
|
432
|
-
continue
|
|
433
|
-
keep.append(d)
|
|
434
|
-
dirnames[:] = keep
|
|
435
|
-
for fn in sorted(filenames):
|
|
436
|
-
ext = os.path.splitext(fn)[1].lower()
|
|
437
|
-
seen_suffix.add(ext)
|
|
438
|
-
if not fn.lower().endswith(pats):
|
|
439
|
-
continue
|
|
440
|
-
if files >= max_files or len(items) >= max_items:
|
|
441
|
-
stats["truncated"] = True
|
|
442
|
-
stats["truncated_reason"] = (
|
|
443
|
-
f"files={files}>=max_files={max_files}"
|
|
444
|
-
if files >= max_files else
|
|
445
|
-
f"items={len(items)}>=max_items={max_items}")
|
|
446
|
-
stats["files"] = files
|
|
447
|
-
stats["skipped_suffixes"] = sorted(
|
|
448
|
-
s for s in seen_suffix if s and s not in pats)[:12]
|
|
449
|
-
return items, errors, stats
|
|
450
|
-
files += 1
|
|
451
|
-
fp = os.path.join(dirpath, fn)
|
|
452
|
-
rel = os.path.relpath(fp, root).replace("\\", "/")
|
|
453
|
-
if fresh is not None and fresh(rel, fp):
|
|
454
|
-
stats["skipped_unchanged"] += 1
|
|
455
|
-
continue
|
|
456
|
-
try:
|
|
457
|
-
with open(fp, encoding="utf-8") as f:
|
|
458
|
-
src = f.read()
|
|
459
|
-
got = extract(src, rel)
|
|
460
|
-
items.extend(got)
|
|
461
|
-
if on_file is not None:
|
|
462
|
-
on_file(rel, fp, got)
|
|
463
|
-
except (OSError, UnicodeDecodeError, ValueError) as exc:
|
|
464
|
-
errors.append(f"{rel}: {exc}")
|
|
465
|
-
if len(items) >= max_items:
|
|
466
|
-
stats["truncated"] = True
|
|
467
|
-
stats["truncated_reason"] = f"items={len(items)}>=max_items={max_items}"
|
|
468
|
-
stats["files"] = files
|
|
469
|
-
stats["skipped_suffixes"] = sorted(
|
|
470
|
-
s for s in seen_suffix if s and s not in pats)[:12]
|
|
471
|
-
return items, errors, stats
|
|
472
|
-
stats["files"] = files
|
|
473
|
-
stats["skipped_suffixes"] = sorted(s for s in seen_suffix if s and s not in pats)[:12]
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""条件文档图:按「章节」索引 md 文档,不存全文。
|
|
3
|
+
|
|
4
|
+
设计(2026-09-10):
|
|
5
|
+
`docs/` 下的 md 是**规范/方案的唯一事实源**,认知图只需要「哪一份文档、哪一节、
|
|
6
|
+
哪几行」这一级坐标,不需要第二份全文(否则文档一改就有两份真相,且必然漂移)。
|
|
7
|
+
因此本模块与 `codeindex` 同构:切块 → 渲染 CCG → frontmatter.doc_ref 指回原文;
|
|
8
|
+
正文用 `op=ref` 回读。
|
|
9
|
+
|
|
10
|
+
切块纪律(对应计划 §八 的风险项):
|
|
11
|
+
· **只切 level<=3**:再深就过细,节点数爆炸且检索噪声上升;
|
|
12
|
+
· 直接正文 < `MIN_BODY` 字且**无子节**的小节**合并进父节**(不单独建节点,
|
|
13
|
+
其文字追加进父节摘要),否则会把「一行小标题」也变成一个节点;
|
|
14
|
+
· **不存全文**:节点正文是 CCG 模板 + 摘要,正文一律回读。
|
|
15
|
+
|
|
16
|
+
md 解析的两处硬约束:
|
|
17
|
+
· **围栏代码块内的 `#` 不是标题**:`docs/` 里大量 python/shell 片段带 `#` 注释,
|
|
18
|
+
若不做围栏跟踪,一节会被切得七零八落(假标题、错行号);
|
|
19
|
+
· **正文里的 `---` 不参与 frontmatter 切分**:这条纪律在 `nodefile.py` 已定,
|
|
20
|
+
本模块额外保证开头 YAML frontmatter 不被当成正文索引,其余 `---` 只当正文。
|
|
21
|
+
|
|
22
|
+
节点正文必须是 **CCG 6 行**(见 `render`):非 CCG 正文会被 `judge_qualification`
|
|
23
|
+
的第一步(ccg_completeness)直接判 **BLINDSPOT**,文档节点会「存得进、判不了、
|
|
24
|
+
检索不到」——这与改造前的 `codeindex` 是同一个坑。
|
|
25
|
+
|
|
26
|
+
区间哈希复用 `codeindex.region_hash`(**唯一实现**),索引侧与 `op=ref` 回读侧共用。
|
|
27
|
+
"""
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import hashlib
|
|
31
|
+
import os
|
|
32
|
+
import re
|
|
33
|
+
|
|
34
|
+
from . import codeindex, nodefile
|
|
35
|
+
|
|
36
|
+
SKIP_DIRS = ("__pycache__", ".git", ".venv", "venv", "node_modules", ".mypy_cache")
|
|
37
|
+
|
|
38
|
+
SUFFIX = (".md", ".markdown")
|
|
39
|
+
|
|
40
|
+
MAX_LEVEL = 3 # 只切 level<=3
|
|
41
|
+
MIN_BODY = 200 # 直接正文 < 200 字且无子节 → 合并进父节
|
|
42
|
+
MAX_SUMMARY = 200 # 「执行」栏摘要上限
|
|
43
|
+
MAX_DOC = 400
|
|
44
|
+
|
|
45
|
+
KIND = "section"
|
|
46
|
+
LANG = "md"
|
|
47
|
+
BASIS = "data" # 文档的验证基底:以原始文档为准
|
|
48
|
+
|
|
49
|
+
# ---- 密级(计划 §1.3-3 的裁定,2026-09-10)--------------------------------
|
|
50
|
+
# 裁定一:layer 默认 knowledge。理由——文档是**可回读、可漂移检测**的参照知识,
|
|
51
|
+
# 与代码节点同层,保证进默认召回;contextual 表示情境绑定、会过期,
|
|
52
|
+
# 用在这里会让文档掉出默认召回。
|
|
53
|
+
# 裁定二:密级**默认 internal 并显式写入 frontmatter**,不依赖节点默认值
|
|
54
|
+
# (mdcos 读隔离取的是 fm.sensitivity;不显式写就等于「靠默认值兜底」,
|
|
55
|
+
# 审计时看不出意图)。且路径段命中私有提示时**再保守一档降为 private**:
|
|
56
|
+
# 宁可漏召回,不可泄漏(计划 §八 的风险项)。
|
|
57
|
+
DEFAULT_SENSITIVITY = "internal"
|
|
58
|
+
PRIVATE_HINTS = ("private", "secret", "internal", "未公开", "私有", "内部")
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
# 生效条件:override 为真值即返回 (override, "调用方显式指定");否则 path(假值按 "")按 "\" 与 "/" 分段,任一段小写含 PRIVATE_HINTS 中任一提示即返回 ("private", 路径段命中理由),全部不命中返回 (DEFAULT_SENSITIVITY, 默认密级理由);
|
|
62
|
+
def sensitivity_for(path, override=None):
|
|
63
|
+
"""返回 (密级, 依据)。override 优先;否则按路径段保守降级。
|
|
64
|
+
|
|
65
|
+
只可能**更严**、不可能更松:命中提示只会把 internal 收紧为 private,
|
|
66
|
+
不会把 private 放开成 public。缺省值显式返回,便于调用方落盘与审计。
|
|
67
|
+
"""
|
|
68
|
+
if override:
|
|
69
|
+
return override, "调用方显式指定"
|
|
70
|
+
for seg in (path or "").replace("\\", "/").split("/"):
|
|
71
|
+
low = seg.lower()
|
|
72
|
+
for hint in PRIVATE_HINTS:
|
|
73
|
+
if hint in low:
|
|
74
|
+
return "private", f"路径段「{seg}」命中私有提示 → 保守降级"
|
|
75
|
+
return DEFAULT_SENSITIVITY, "默认密级(显式写入,不依赖节点默认值)"
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# --------------------------------------------------------------------------
|
|
79
|
+
# md 解析
|
|
80
|
+
# --------------------------------------------------------------------------
|
|
81
|
+
_FENCE = re.compile(r"^\s*(```+|~~~+)")
|
|
82
|
+
_ATX = re.compile(r"^(#{1,6})\s+(.+?)\s*#*\s*$")
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
# 生效条件:lines 非空且 lines[0].strip() == "---" 时,从下标 1 起找到首个 strip() == "---" 的行并返回其后一行下标 i+1;lines 为空、首行不是 "---" 或找不到闭合 "---" 时返回 0;
|
|
86
|
+
def _body_start(lines):
|
|
87
|
+
"""跳过开头 YAML frontmatter,返回正文起始行下标(0 基)。"""
|
|
88
|
+
if lines and lines[0].strip() == "---":
|
|
89
|
+
for i in range(1, len(lines)):
|
|
90
|
+
if lines[i].strip() == "---":
|
|
91
|
+
return i + 1
|
|
92
|
+
return 0
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
# 生效条件:从 start(默认 0)遍历 lines,未处于围栏时遇 _FENCE 匹配行打开同标记围栏、围栏内遇同标记行关闭并继续;围栏外 _ATX 匹配行追加 {level: 一级 # 个数, title: 去空白后的标题, lineno: i+1};返回 out 列表;start 不小于 len(lines) 时返回空列表;
|
|
96
|
+
def _headings(lines, start=0):
|
|
97
|
+
"""产出 ATX 标题 `{level,title,lineno}`;**围栏代码块内的 `#` 不算标题**。"""
|
|
98
|
+
out, fence = [], None
|
|
99
|
+
for i in range(start, len(lines)):
|
|
100
|
+
line = lines[i]
|
|
101
|
+
m = _FENCE.match(line)
|
|
102
|
+
if m:
|
|
103
|
+
mark = m.group(1)[0]
|
|
104
|
+
if fence is None:
|
|
105
|
+
fence = mark
|
|
106
|
+
elif fence == mark:
|
|
107
|
+
fence = None
|
|
108
|
+
continue
|
|
109
|
+
if fence is not None:
|
|
110
|
+
continue
|
|
111
|
+
m = _ATX.match(line)
|
|
112
|
+
if m:
|
|
113
|
+
out.append({"level": len(m.group(1)), "title": m.group(2).strip(),
|
|
114
|
+
"lineno": i + 1})
|
|
115
|
+
return out
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
# 生效条件:title 真值时取其 strip 后小写,删去 `*[]() 与除 \w\s- 外字符,再把空白/下划线连成 "-" 并 strip("-");title 为假值(含 None、空串)时按空串处理并返回 "";
|
|
119
|
+
def _anchor(title):
|
|
120
|
+
a = (title or "").strip().lower()
|
|
121
|
+
a = re.sub(r"`|\*|\[|\]|\(|\)", "", a)
|
|
122
|
+
a = re.sub(r"[^\w\s-]", "", a, flags=re.UNICODE) # CJK 属 \w,保留
|
|
123
|
+
return re.sub(r"[\s_]+", "-", a).strip("-")
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
# 生效条件:region_lines 逐行 strip 后跳过空行与 "---",去掉行首 #+ 和 >|*- 标记,以空格连接成 text;返回 text[:limit](limit 默认 MAX_SUMMARY;limit=0 返回 "",limit=None 返回全文,limit='' 时切片抛 TypeError,负 limit 按负索引切片);
|
|
127
|
+
def _summary(region_lines, limit=MAX_SUMMARY):
|
|
128
|
+
"""把一段正文压成一行摘要(去 markdown 噪声,不逐字保留)。"""
|
|
129
|
+
parts = []
|
|
130
|
+
for ln in region_lines:
|
|
131
|
+
s = ln.strip()
|
|
132
|
+
if not s or s == "---":
|
|
133
|
+
continue
|
|
134
|
+
s = re.sub(r"^#+\s*", "", s)
|
|
135
|
+
s = re.sub(r"^[>|*-]\s*", "", s)
|
|
136
|
+
parts.append(s)
|
|
137
|
+
text = " ".join(parts).strip()
|
|
138
|
+
return text[:limit]
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
# 生效条件:以 heads[i]["level"]-1 为需匹配层级向前回溯,返回按层级递减补齐的祖先标题列表(不含 heads[i] 自身)。
|
|
142
|
+
def _path_titles(heads, i):
|
|
143
|
+
"""第 i 个标题的祖先链(不含自身),按层级补齐。"""
|
|
144
|
+
out, need = [], heads[i]["level"] - 1
|
|
145
|
+
for k in range(i - 1, -1, -1):
|
|
146
|
+
if heads[k]["level"] == need:
|
|
147
|
+
out.insert(0, heads[k]["title"])
|
|
148
|
+
need -= 1
|
|
149
|
+
if need == 0:
|
|
150
|
+
break
|
|
151
|
+
return out
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
# 生效条件:直接以 lines、lineno、end 调用 codeindex.region_hash 并返回其结果;
|
|
155
|
+
def _region_hash(lines, lineno, end):
|
|
156
|
+
# 唯一实现复用 codeindex.region_hash:两侧各写一份,漂移检测会悄悄失效。
|
|
157
|
+
return codeindex.region_hash(lines, lineno, end)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
# 生效条件:ext = suffix 真值时原样使用的 suffix,否则取 os.path.splitext(path)[1].lower();ext 不在 SUFFIX 时抛 ValueError;在 SUFFIX 时把 source 按 "\n" 拆分,经 _body_start 与 _headings 得到标题,仅 level<=MAX_LEVEL 且非 small 的标题生成条目,小/过深子节摘要并入父摘要,返回 items;
|
|
161
|
+
def extract(source, path="", suffix=None):
|
|
162
|
+
"""抽取一份 md 的章节条目;按后缀分派。返回条目列表(可能为空)。
|
|
163
|
+
|
|
164
|
+
条目字段与 `codeindex` 对齐(name/kind/lineno/end/hash/lang/precise/basis),
|
|
165
|
+
另带文档专有:heading / heading_path / level / anchor / children。
|
|
166
|
+
"""
|
|
167
|
+
ext = suffix or os.path.splitext(path)[1].lower()
|
|
168
|
+
if ext not in SUFFIX:
|
|
169
|
+
raise ValueError(f"无文档提取器(suffix={ext or '<none>'})")
|
|
170
|
+
lines = source.split("\n")
|
|
171
|
+
heads = _headings(lines, _body_start(lines))
|
|
172
|
+
|
|
173
|
+
# info 以 heads 序号为键:children 存的是 heads 序号,不能拿去过 secs 的下标。
|
|
174
|
+
info = {}
|
|
175
|
+
for i, h in enumerate(heads):
|
|
176
|
+
end = len(lines)
|
|
177
|
+
children = []
|
|
178
|
+
for j in range(i + 1, len(heads)):
|
|
179
|
+
if heads[j]["level"] <= h["level"]:
|
|
180
|
+
end = heads[j]["lineno"] - 1
|
|
181
|
+
break
|
|
182
|
+
children.append(j)
|
|
183
|
+
direct_end = (heads[i + 1]["lineno"] - 1) if i + 1 < len(heads) else len(lines)
|
|
184
|
+
direct = lines[h["lineno"]:max(h["lineno"], min(direct_end, end))]
|
|
185
|
+
info[i] = {"end": max(h["lineno"], end), "children": children, "direct": direct,
|
|
186
|
+
"small": len(_summary(direct)) < MIN_BODY and not children}
|
|
187
|
+
|
|
188
|
+
items, seen = [], {}
|
|
189
|
+
for i, h in enumerate(heads):
|
|
190
|
+
if h["level"] > MAX_LEVEL or info[i]["small"]:
|
|
191
|
+
continue # 合并进父节点:父节点的 end 已覆盖其区间
|
|
192
|
+
summary = _summary(info[i]["direct"])
|
|
193
|
+
# 被合并进来的子节(自身过小,或层级过深从不单独建节点):文字并入父节摘要,
|
|
194
|
+
# 否则这些小节的正文只存在于父节的 ref 区间里,检索不到。
|
|
195
|
+
merged = [_summary(info[k]["direct"], 80) for k in info[i]["children"]
|
|
196
|
+
if info[k]["small"] or heads[k]["level"] > MAX_LEVEL]
|
|
197
|
+
merged = [m for m in merged if m]
|
|
198
|
+
if merged:
|
|
199
|
+
summary = (summary + ";" + ";".join(merged))[:MAX_SUMMARY]
|
|
200
|
+
if not summary:
|
|
201
|
+
summary = "(该节无直接正文,见子节)"
|
|
202
|
+
parent = _path_titles(heads, i)
|
|
203
|
+
heading_path = parent + [h["title"]]
|
|
204
|
+
key = path + "#" + "/".join(heading_path)
|
|
205
|
+
dup = seen.get(key, 0) + 1
|
|
206
|
+
seen[key] = dup
|
|
207
|
+
item = {
|
|
208
|
+
"path": path, "name": h["title"], "heading": h["title"], "kind": KIND,
|
|
209
|
+
"heading_path": heading_path, "level": h["level"],
|
|
210
|
+
"anchor": _anchor(h["title"]), "parent": parent[-1] if parent else "",
|
|
211
|
+
"lineno": h["lineno"], "end": info[i]["end"],
|
|
212
|
+
"summary_parts": summary,
|
|
213
|
+
"children": [heads[k]["title"] for k in info[i]["children"]],
|
|
214
|
+
"dup": dup, "lang": LANG, "precise": True, "basis": BASIS,
|
|
215
|
+
}
|
|
216
|
+
item["hash"] = _region_hash(lines, item["lineno"], item["end"])
|
|
217
|
+
items.append(item)
|
|
218
|
+
return items
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
# --------------------------------------------------------------------------
|
|
222
|
+
# 渲染 / id
|
|
223
|
+
# --------------------------------------------------------------------------
|
|
224
|
+
# 生效条件:item.get("path") 缺键或为假值时 path 取 "",top 取 path.split("/")[0] or "."(故空 path 时 top=".");path 为真值时 top 取其 "/" 前首段,首段为空则 top=".";返回含 observation_position(大域=top)、time_window([nodefile.FULL_TIME_WINDOW_MIN, nodefile.FULL_TIME_WINDOW_MAX])、observation_tool、existence_constraint(以 path 拼入)的四槽字典;
|
|
225
|
+
def condition_space(item):
|
|
226
|
+
"""章节条目 → 条件空间四槽(纯函数,**唯一来源**)。
|
|
227
|
+
|
|
228
|
+
与 `codeindex.condition_space` 同一职责、同一理由:`render` 的正文行与
|
|
229
|
+
`refindex.add_items` 的 frontmatter 必须同源,否则 frontmatter 只剩单槽
|
|
230
|
+
`observation_position`,`nodefile.condition_space_text(require_full=True)`
|
|
231
|
+
恒返回 "" —— 条件空间等于没声明。改造前正文写的是「文档=X;检索…时」,
|
|
232
|
+
是第三种方言,既进不了条件空间,也不可被 `_slot_overlap` 使用。
|
|
233
|
+
|
|
234
|
+
时间槽给全时窗哨兵:文档章节条目声明的是「该文档里有这一节」,
|
|
235
|
+
真值不随索引时刻衰减,不写成 1 小时观测窗。
|
|
236
|
+
"""
|
|
237
|
+
path = item.get("path") or ""
|
|
238
|
+
top = path.split("/")[0] or "."
|
|
239
|
+
return {
|
|
240
|
+
"observation_position": f"本地文档仓(大域={top})",
|
|
241
|
+
"time_window": [nodefile.FULL_TIME_WINDOW_MIN,
|
|
242
|
+
nodefile.FULL_TIME_WINDOW_MAX],
|
|
243
|
+
"observation_tool": (f"{LANG}(md 章节切分,level≤{MAX_LEVEL};"
|
|
244
|
+
f"只存标题+摘要,正文留在源文件)"),
|
|
245
|
+
"existence_constraint": f"源文档 {path} 存在于本地仓且可读",
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
# 生效条件:item 含 heading、path、lineno、end、anchor 键时渲染 7 行 CCG 文本(第2行取 condition_space(item) 文本),item.get("parent") 或 "" 假值回落 "(顶层章节)",item.get("summary_parts") 假值回落 "(该节无直接正文,见子节)",item.get("children") 假值回落空列表且不追加子节行,children 非空时追加 "# 子节:" + 前 12 个;返回以 "\n" 连接的行串;
|
|
250
|
+
def render(item):
|
|
251
|
+
"""章节条目 → CCG 6 行正文(可被 search 命中,不含全文)。
|
|
252
|
+
|
|
253
|
+
**必须渲染成 CCG**:`judge_qualification` 第一步查 ccg_completeness 的 5 要素,
|
|
254
|
+
缺任一即直接判 BLINDSPOT(与 codeindex.render 同一个坑)。
|
|
255
|
+
"""
|
|
256
|
+
heading = item["heading"]
|
|
257
|
+
path = item["path"]
|
|
258
|
+
parent = item.get("parent") or ""
|
|
259
|
+
summary = item.get("summary_parts") or "(该节无直接正文,见子节)"
|
|
260
|
+
sub = f"父章节:{parent}" if parent else "(顶层章节)"
|
|
261
|
+
children = item.get("children") or []
|
|
262
|
+
lines = [
|
|
263
|
+
f"# 功能名:{heading}",
|
|
264
|
+
f"# 生效条件:{nodefile.condition_space_text(condition_space(item))}",
|
|
265
|
+
f"# 子功能:{sub}",
|
|
266
|
+
f"# 执行:{summary[:MAX_DOC]}",
|
|
267
|
+
(f"# 验证方式:{BASIS}(以原始文档为准;"
|
|
268
|
+
f"区间 {path} L{item['lineno']}-L{item['end']})"),
|
|
269
|
+
f"# 不适用条件:其它文档的同名标题(本条目属于 {path}#{item['anchor']})",
|
|
270
|
+
f"# 位置:{path}#{item['anchor']}:{item['lineno']}-{item['end']}"
|
|
271
|
+
f"({item.get('lang')},precise=True)",
|
|
272
|
+
]
|
|
273
|
+
if children:
|
|
274
|
+
lines.append("# 子节:" + ";".join(children[:12]))
|
|
275
|
+
return "\n".join(lines)
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
# 生效条件:以 item["path"] + "#" + "/".join(item["heading_path"]) 为 key,item.get("dup", 1)(缺键取 1)大于 1 时追加 "#" + item["dup"],返回 "doc_" + sha1(key utf-8) hexdigest 前 12 位;
|
|
279
|
+
def node_id(item):
|
|
280
|
+
"""稳定 id:path#heading_path 的短哈希(重复索引幂等;同名用 dup 区分)。"""
|
|
281
|
+
key = item["path"] + "#" + "/".join(item["heading_path"])
|
|
282
|
+
if item.get("dup", 1) > 1:
|
|
283
|
+
key += f"#{item['dup']}"
|
|
284
|
+
return "doc_" + hashlib.sha1(key.encode("utf-8")).hexdigest()[:12]
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
# ==========================================================================
|
|
288
|
+
# fence 真源绑定(§3.5):条件卡 ↔ Markdown 真源的「往返列」
|
|
289
|
+
#
|
|
290
|
+
# 往返列 = 真源(root + path)+ 定位(anchor / heading_path)+ 行位(lineno/end)
|
|
291
|
+
# + 校验(hash)。**行位是易腐化量,键不含行位**:在章节上方插一段话会让整篇行号
|
|
292
|
+
# 位移,但键 `path#heading_path` 不变——故「同一章节」的匹配一律走键,行位只用于
|
|
293
|
+
# 回读原文与漂移判定。写侧唯一入口是 `refindex._doc_ref`(与索引同源,避免第二份
|
|
294
|
+
# 列口径);本节只补读侧:取列 / 校验 / 造键 / 反查行位。
|
|
295
|
+
# · 正向:卡片 → 真源(`doc_ref.lineno/end` + `refindex.read_ref` 回读原文区间)
|
|
296
|
+
# · 反向:真源行 → 卡片(`locate`;对账器 `--at path:line` 即此接口)
|
|
297
|
+
# ==========================================================================
|
|
298
|
+
|
|
299
|
+
#: 必需列:真源(root/path)+ 定位(anchor)+ 行位(lineno/end)+ 校验(hash)
|
|
300
|
+
BINDING_FIELDS = ("root", "path", "anchor", "lineno", "end", "hash")
|
|
301
|
+
#: 附加列:有则参与更精确匹配与人读,缺省不算绑定失败
|
|
302
|
+
BINDING_OPTIONAL = ("heading_path", "heading", "level", "lang", "precise")
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
# 生效条件:node 为 cg.get 产物(含 frontmatter 子字典)或 frontmatter dict 本身,且其 doc_ref 为 dict 时,返回按 BINDING_FIELDS + BINDING_OPTIONAL 投影的列值 dict(缺列以 None 占位);非 dict / 无 doc_ref 返回 None。
|
|
306
|
+
def binding_of(node):
|
|
307
|
+
"""取一条条件卡的往返列;**非索引节点返回 None**(不猜、不补默认值)。"""
|
|
308
|
+
fm = node
|
|
309
|
+
if isinstance(node, dict) and isinstance(node.get("frontmatter"), dict):
|
|
310
|
+
fm = node["frontmatter"]
|
|
311
|
+
if not isinstance(fm, dict):
|
|
312
|
+
return None
|
|
313
|
+
ref = fm.get("doc_ref")
|
|
314
|
+
if not isinstance(ref, dict) or not ref:
|
|
315
|
+
return None
|
|
316
|
+
return {k: ref.get(k) for k in BINDING_FIELDS + BINDING_OPTIONAL}
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
# 生效条件:b 为 dict 且 heading_path 为非空 list/tuple 时返回 path#join(heading_path, '/'),否则回落 path#anchor;b 非 dict 返回空串。
|
|
320
|
+
def binding_key(b):
|
|
321
|
+
"""稳定键 `path#heading_path`:**不含行位**(真源重排只动行位、不动键)。"""
|
|
322
|
+
if not isinstance(b, dict):
|
|
323
|
+
return ""
|
|
324
|
+
path = str(b.get("path") or "")
|
|
325
|
+
hp = b.get("heading_path")
|
|
326
|
+
if isinstance(hp, (list, tuple)) and hp:
|
|
327
|
+
return path + "#" + "/".join(str(x) for x in hp)
|
|
328
|
+
return path + "#" + str(b.get("anchor") or "")
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
# 生效条件:b 为 dict 时返回 'path#anchor'(与 render 正文里的「本条目属于」逐字同源);非 dict 返回空串。
|
|
332
|
+
def binding_slug(b):
|
|
333
|
+
"""人读定位串 `path#anchor`——与 `render` 正文里的「本条目属于」同源。"""
|
|
334
|
+
b = b if isinstance(b, dict) else {}
|
|
335
|
+
return f"{b.get('path') or ''}#{b.get('anchor') or ''}"
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
# 生效条件:b 为 dict 时逐列校验(缺列 / 类型错 / 行位越界),返回 {"ok": bool, "issues": [str, ...]};**不抛异常**——对账要逐条报告,不能因一条坏数据中断全库;行位仅在列齐且类型对时才判,避免级联噪声。
|
|
339
|
+
def validate_binding(b):
|
|
340
|
+
"""校验往返列形状。**不抛异常**:对账逐条报告,不能一条坏数据中断全库。"""
|
|
341
|
+
if not isinstance(b, dict):
|
|
342
|
+
return {"ok": False, "issues": ["绑定不是字典(该节点无 doc_ref)"]}
|
|
343
|
+
issues = []
|
|
344
|
+
for k in BINDING_FIELDS:
|
|
345
|
+
if b.get(k) in (None, ""):
|
|
346
|
+
issues.append(f"缺列 {k}")
|
|
347
|
+
for k in ("root", "path", "anchor", "hash"):
|
|
348
|
+
v = b.get(k)
|
|
349
|
+
if v not in (None, "") and not isinstance(v, str):
|
|
350
|
+
issues.append(f"{k} 应为字符串,实为 {type(v).__name__}")
|
|
351
|
+
if not issues:
|
|
352
|
+
try:
|
|
353
|
+
lo, hi = int(b["lineno"]), int(b["end"])
|
|
354
|
+
except (TypeError, ValueError):
|
|
355
|
+
issues.append("行位不是整数")
|
|
356
|
+
else:
|
|
357
|
+
if lo < 1:
|
|
358
|
+
issues.append(f"lineno 越界({lo} < 1)")
|
|
359
|
+
if hi < lo:
|
|
360
|
+
issues.append(f"end 早于 lineno({lo}-{hi})")
|
|
361
|
+
return {"ok": not issues, "issues": issues}
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
# 生效条件:old/new 任意为 dict 或假值;按 (path, anchor, heading_path, lineno, end, hash) 固定顺序返回取值不同的列名列表(空列表=同一章节同一行位),任一侧假值按空 dict 处理。
|
|
365
|
+
def binding_drift(old, new):
|
|
366
|
+
"""旧往返列 vs 新往返列 → 变化列名(顺序固定,供报告与测试断言)。"""
|
|
367
|
+
old, new = old if isinstance(old, dict) else {}, new if isinstance(new, dict) else {}
|
|
368
|
+
return [k for k in ("path", "anchor", "heading_path", "lineno", "end", "hash")
|
|
369
|
+
if old.get(k) != new.get(k)]
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
# 生效条件:items 为 extract 产出的条目序列、lineno 可转 int;返回覆盖该行的条目中 **lineno 最大者**(嵌套即最内层,h3 优于其父 h2);不可转值或无可覆盖条目返回 None。
|
|
373
|
+
def locate(items, lineno):
|
|
374
|
+
"""反向定位:真源第 `lineno` 行 → 覆盖它的条目(嵌套取**最内层**)。
|
|
375
|
+
|
|
376
|
+
区间闭合:标题行算本节;多层嵌套时取 `lineno` 最大者即最内层。
|
|
377
|
+
"""
|
|
378
|
+
try:
|
|
379
|
+
ln = int(lineno)
|
|
380
|
+
except (TypeError, ValueError):
|
|
381
|
+
return None
|
|
382
|
+
hit = None
|
|
383
|
+
for it in items or []:
|
|
384
|
+
lo, hi = it.get("lineno"), it.get("end")
|
|
385
|
+
if lo is None or hi is None:
|
|
386
|
+
continue
|
|
387
|
+
try:
|
|
388
|
+
lo, hi = int(lo), int(hi)
|
|
389
|
+
except (TypeError, ValueError):
|
|
390
|
+
continue
|
|
391
|
+
if lo <= ln <= hi and (hit is None or lo > int(hit["lineno"])):
|
|
392
|
+
hit = it
|
|
393
|
+
return hit
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
# 生效条件:以 root 为根 os.walk,patterns 假值回落 SUFFIX;files 达到 max_files 或 items 达到 max_items 时提前返回并置 stats["truncated"]/truncated_reason;fresh 非 None 且 fresh(rel, fp) 为真时跳过该文件读盘并计 skipped_unchanged;on_file 非 None 且 open/extract 成功后以 (rel, fp, got) 回调;名字在 SKIP_DIRS 的目录仅剪枝不记录,skip_dirs 经 codeindex.skip_matcher 命中的目录剪枝并记入 stats["skipped_dirs"];返回 (items, errors, stats)。
|
|
397
|
+
def index_dir(root, patterns=None, max_files=500, max_items=2000,
|
|
398
|
+
fresh=None, on_file=None, skip_dirs=None):
|
|
399
|
+
"""按大域(目录)遍历 md,产出 `(items, errors, stats)`。零 LLM。
|
|
400
|
+
|
|
401
|
+
stats 语义与 `codeindex.index_dir` 一致:`truncated`/`truncated_reason` 显式上报
|
|
402
|
+
(截断不静默),`skipped_suffixes` 列出扫到但没被索引的后缀(覆盖缺口可审计)。
|
|
403
|
+
|
|
404
|
+
`fresh(rel, fp)` / `on_file(rel, fp, items)` 是给 `refindex.Ledger` 留的增量钩子
|
|
405
|
+
(默认 None → 行为与改造前逐字一致):未变文件不读盘、计入 `skipped_unchanged`。
|
|
406
|
+
|
|
407
|
+
`skip_dirs` 是**追加**排除,复用 `codeindex.skip_matcher`(唯一实现,避免两条
|
|
408
|
+
索引链路口径漂移):命中的目录整棵剪掉、不计入 `files`。本仓的实例就是
|
|
409
|
+
`docs/experiments/`——`.gitignore` 已整目录忽略、物理却仍有 2358 个 md 的实验
|
|
410
|
+
产物,会把 `max_files` 撑爆并把「索引不全」变成常态。排掉了哪些目录写进
|
|
411
|
+
`stats["skipped_dirs"]`,排除与截断一样**不许静默**。
|
|
412
|
+
"""
|
|
413
|
+
pats = tuple(patterns or SUFFIX)
|
|
414
|
+
hit_skip, skip_rules = codeindex.skip_matcher(skip_dirs)
|
|
415
|
+
items, errors, files = [], [], 0
|
|
416
|
+
seen_suffix = set()
|
|
417
|
+
stats = {"root": root, "patterns": list(pats), "files": 0, "truncated": False,
|
|
418
|
+
"truncated_reason": "", "max_files": max_files, "max_items": max_items,
|
|
419
|
+
"skipped_suffixes": [], "skipped_unchanged": 0,
|
|
420
|
+
"skip_dirs": list(skip_rules), "skipped_dirs": []}
|
|
421
|
+
for dirpath, dirnames, filenames in os.walk(root):
|
|
422
|
+
rel_dir = os.path.relpath(dirpath, root).replace("\\", "/")
|
|
423
|
+
if rel_dir == ".":
|
|
424
|
+
rel_dir = ""
|
|
425
|
+
keep = []
|
|
426
|
+
for d in dirnames:
|
|
427
|
+
if d in SKIP_DIRS:
|
|
428
|
+
continue
|
|
429
|
+
child = f"{rel_dir}/{d}" if rel_dir else d
|
|
430
|
+
if hit_skip is not None and hit_skip(child, d):
|
|
431
|
+
stats["skipped_dirs"].append(child)
|
|
432
|
+
continue
|
|
433
|
+
keep.append(d)
|
|
434
|
+
dirnames[:] = keep
|
|
435
|
+
for fn in sorted(filenames):
|
|
436
|
+
ext = os.path.splitext(fn)[1].lower()
|
|
437
|
+
seen_suffix.add(ext)
|
|
438
|
+
if not fn.lower().endswith(pats):
|
|
439
|
+
continue
|
|
440
|
+
if files >= max_files or len(items) >= max_items:
|
|
441
|
+
stats["truncated"] = True
|
|
442
|
+
stats["truncated_reason"] = (
|
|
443
|
+
f"files={files}>=max_files={max_files}"
|
|
444
|
+
if files >= max_files else
|
|
445
|
+
f"items={len(items)}>=max_items={max_items}")
|
|
446
|
+
stats["files"] = files
|
|
447
|
+
stats["skipped_suffixes"] = sorted(
|
|
448
|
+
s for s in seen_suffix if s and s not in pats)[:12]
|
|
449
|
+
return items, errors, stats
|
|
450
|
+
files += 1
|
|
451
|
+
fp = os.path.join(dirpath, fn)
|
|
452
|
+
rel = os.path.relpath(fp, root).replace("\\", "/")
|
|
453
|
+
if fresh is not None and fresh(rel, fp):
|
|
454
|
+
stats["skipped_unchanged"] += 1
|
|
455
|
+
continue
|
|
456
|
+
try:
|
|
457
|
+
with open(fp, encoding="utf-8") as f:
|
|
458
|
+
src = f.read()
|
|
459
|
+
got = extract(src, rel)
|
|
460
|
+
items.extend(got)
|
|
461
|
+
if on_file is not None:
|
|
462
|
+
on_file(rel, fp, got)
|
|
463
|
+
except (OSError, UnicodeDecodeError, ValueError) as exc:
|
|
464
|
+
errors.append(f"{rel}: {exc}")
|
|
465
|
+
if len(items) >= max_items:
|
|
466
|
+
stats["truncated"] = True
|
|
467
|
+
stats["truncated_reason"] = f"items={len(items)}>=max_items={max_items}"
|
|
468
|
+
stats["files"] = files
|
|
469
|
+
stats["skipped_suffixes"] = sorted(
|
|
470
|
+
s for s in seen_suffix if s and s not in pats)[:12]
|
|
471
|
+
return items, errors, stats
|
|
472
|
+
stats["files"] = files
|
|
473
|
+
stats["skipped_suffixes"] = sorted(s for s in seen_suffix if s and s not in pats)[:12]
|
|
474
474
|
return items, errors, stats
|