@furongjun1999/dsh-memory 0.4.11 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +552 -465
- package/codebuddy/CODEBUDDY.md +11 -3
- package/codebuddy/README.md +92 -90
- package/codebuddy/mcp.json +27 -27
- package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
- package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
- package/docs/README.md +143 -111
- package/docs/discipline/harnesses.yaml +244 -226
- package/docs/discipline/templates/full.md.tmpl +61 -61
- package/docs/discipline/templates/rules.mdc.tmpl +68 -0
- package/docs/discipline/templates/skill.md.tmpl +23 -23
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
- package/docs/eval/bench_lingshu_self/bench_self.py +140 -0
- package/docs/eval/bench_lingshu_self/self_bench_result.json +404 -0
- package/docs/eval/bench_lingshu_self//347/201/265/346/236/242/350/207/252/345/272/223/347/253/257/345/210/260/347/253/257/346/243/200/347/264/242/345/256/236/346/265/213_v1.0.md +38 -0
- package/docs/eval//344/270/215/345/217/257/351/235/240/346/200/247/350/220/275/345/234/260_P0_v1.0.md +185 -0
- package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
- package/docs/eval//346/225/205/351/232/234/346/263/250/345/205/245/345/256/236/346/265/213_v1.0.md +422 -0
- package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
- package/docs/eval//347/253/257/345/210/260/347/253/257LoCoMoQA/345/220/214/345/217/243/345/276/204/345/257/271/347/205/247_v1.0.md +100 -0
- package/docs/eval//347/253/257/345/210/260/347/253/257/345/271/262/346/211/260/346/261/240/350/257/204/346/265/213_/347/241/256/345/256/232/346/200/247/350/243/201/345/206/263vsLLM_judge_v1.1.md +197 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v1.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v10.md +210 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v11.md +227 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v12.md +203 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v13.md +233 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v14.md +191 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v15.md +213 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v16.md +214 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v17.md +199 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v2.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v3.md +152 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v4.md +128 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v5.md +114 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v6.md +192 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v7.md +187 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v8.md +207 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v9.md +203 -0
- package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
- package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
- package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
- package/docs/hive//345/256/211/345/205/250/345/256/241/350/256/241/345/256/236/351/224/232_v0.1.md +202 -0
- package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
- package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
- package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
- package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +519 -503
- package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
- package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
- package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
- package/docs/images/lingshu-moonlight-covenant-poster-preview.jpg +0 -0
- package/docs/images/lingshu-moonlight-covenant-poster.png +0 -0
- package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
- package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
- package/docs/mdcg/lingshu_tutorial.html +14449 -14449
- package/docs/mdcg/release_v0.4.11.md +49 -0
- package/docs/mdcg/release_v0.4.5.md +55 -55
- package/docs/mdcg/tool_table_v0.3.0.md +117 -117
- package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
- package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
- package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
- package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
- package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
- package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
- package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
- package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
- package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
- package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
- package/docs/theory//344/270/215/345/217/257/351/235/240/345/256/232/347/220/206/344/270/216/345/244/261/346/225/210/344/274/230/345/205/210/346/241/206/346/236/266_v0.3.md +365 -0
- package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
- package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
- package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
- package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
- package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
- package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
- package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
- package/docs//344/270/215/345/217/257/351/235/240/346/200/247/347/220/206/350/256/272_v0.1.md +343 -0
- package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
- package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
- package/dsh/README.md +82 -82
- package/dsh/cordis.yml.example +139 -139
- package/dsh/update-lingshu.bat +11 -11
- package/lib/bridge.d.ts +9 -0
- package/lib/bridge.js +35 -0
- package/lib/hooks.js +36 -2
- package/lib/index.js +7 -1
- package/lib/lib/roleplay_web.js +116 -29
- package/lib/lib/token_store.d.ts +7 -1
- package/lib/lib/token_store.js +12 -3
- package/md_cg/__init__.py +7 -7
- package/md_cg/audit.py +379 -368
- package/md_cg/autonomy.py +287 -287
- package/md_cg/backfill.py +1328 -1327
- package/md_cg/backfill_bigdomain.py +34 -34
- package/md_cg/backfill_bucket_zh.py +35 -0
- package/md_cg/bench6_arms.py +410 -410
- package/md_cg/bench6_common.py +230 -230
- package/md_cg/bench6_competitors.py +212 -212
- package/md_cg/bench_axis_domain.py +257 -257
- package/md_cg/bench_blind_comp.py +308 -308
- package/md_cg/bench_e2e_judge.py +532 -0
- package/md_cg/bench_e2e_locomo_qa.py +368 -0
- package/md_cg/bench_e2e_qa.py +256 -0
- package/md_cg/bench_en_atoms_public.py +230 -230
- package/md_cg/bench_governance.py +348 -348
- package/md_cg/bench_lme_zh.py +410 -410
- package/md_cg/bench_locomo.py +121 -121
- package/md_cg/bench_locomo_zh.py +450 -450
- package/md_cg/bench_locomo_zh_public.py +147 -147
- package/md_cg/bench_longmem.py +112 -112
- package/md_cg/bench_membench.py +632 -632
- package/md_cg/bench_p0.py +149 -149
- package/md_cg/bench_progressive.py +287 -287
- package/md_cg/bench_role_views.py +238 -238
- package/md_cg/bench_task_ab.py +243 -243
- package/md_cg/bench_task_ab_llm.py +408 -408
- package/md_cg/bench_unified_en.py +204 -204
- package/md_cg/bench_zh_mad.py +601 -601
- package/md_cg/blindspot_tickets.py +123 -123
- package/md_cg/branches.py +301 -285
- package/md_cg/build_postings.py +73 -73
- package/md_cg/ccgc.py +1006 -948
- package/md_cg/census.py +132 -132
- package/md_cg/chain.py +315 -300
- package/md_cg/codeindex.py +531 -531
- package/md_cg/coldverify.py +292 -292
- package/md_cg/comment_gate.py +337 -337
- package/md_cg/cond_compose.py +190 -190
- package/md_cg/cond_facts.py +154 -154
- package/md_cg/cond_template.json +106 -106
- package/md_cg/condition_anchor.py +142 -142
- package/md_cg/conformance.py +726 -726
- package/md_cg/consistency.py +717 -717
- package/md_cg/consolidate.py +1537 -1439
- package/md_cg/corpus.py +110 -110
- package/md_cg/crosscheck.py +1098 -1097
- package/md_cg/crypto.py +3 -1
- package/md_cg/d_meta.py +310 -310
- package/md_cg/datapath.py +78 -18
- package/md_cg/docindex.py +473 -473
- package/md_cg/eval_common.py +575 -575
- package/md_cg/evidence.py +4 -2
- package/md_cg/evolution.py +477 -477
- package/md_cg/export.py +222 -220
- package/md_cg/forgetting.py +581 -581
- package/md_cg/fsutil.py +377 -329
- package/md_cg/hotcache.py +48 -7
- package/md_cg/hyperedge.py +251 -251
- package/md_cg/identity.py +390 -390
- package/md_cg/insight.py +500 -500
- package/md_cg/interop.py +338 -0
- package/md_cg/judgment_manifest.py +177 -0
- package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
- package/md_cg/lexicon/build_standard_en.py +171 -171
- package/md_cg/lexicon/expand_en_zh.py +211 -211
- package/md_cg/lifecycle.py +272 -272
- package/md_cg/linkref.py +280 -280
- package/md_cg/links.py +140 -107
- package/md_cg/mcp_server.py +403 -55
- package/md_cg/md_whitebox.py +345 -345
- package/md_cg/mdcg.py +783 -233
- package/md_cg/mdcos.py +500 -70
- package/md_cg/metacognition.py +591 -591
- package/md_cg/migrate.py +119 -119
- package/md_cg/migrate_aeis.py +221 -221
- package/md_cg/migrate_roleplay.py +293 -293
- package/md_cg/migrate_wisdom_graph.py +360 -360
- package/md_cg/mreview/__init__.py +25 -25
- package/md_cg/mreview/__main__.py +110 -110
- package/md_cg/mreview/bundle.py +178 -178
- package/md_cg/mreview/candidates.py +262 -262
- package/md_cg/mreview/govern.py +694 -693
- package/md_cg/mreview/locate.py +939 -939
- package/md_cg/mreview/pipeline.py +728 -728
- package/md_cg/mreview/rules/duplication.json +21 -21
- package/md_cg/mreview/rules/field_coverage.json +54 -54
- package/md_cg/mreview/rules/source_license.json +21 -21
- package/md_cg/mreview/rules/template_flow.json +21 -21
- package/md_cg/mreview/ruleset.py +252 -252
- package/md_cg/nodefile.py +575 -575
- package/md_cg/pooling.py +484 -472
- package/md_cg/postings.py +300 -298
- package/md_cg/predict.py +1100 -1100
- package/md_cg/progressive.py +123 -123
- package/md_cg/protect.py +272 -272
- package/md_cg/protocol/md_cg_gate.proto +33 -33
- package/md_cg/protocol.py +372 -372
- package/md_cg/provenance.py +582 -582
- package/md_cg/reach.py +453 -453
- package/md_cg/readcache.py +143 -0
- package/md_cg/reconcile.py +228 -0
- package/md_cg/refindex.py +833 -833
- package/md_cg/refine.py +604 -604
- package/md_cg/review_cli.py +170 -0
- package/md_cg/roleviews.py +89 -89
- package/md_cg/routing.py +393 -365
- package/md_cg/run_tests.py +211 -0
- package/md_cg/scrub.py +13 -3
- package/md_cg/security.py +128 -18
- package/md_cg/self_state.py +1029 -1029
- package/md_cg/selfreport.py +152 -151
- package/md_cg/semantic/__init__.py +10 -10
- package/md_cg/semantic/canonical.py +122 -122
- package/md_cg/semantic/en_normalizer.py +364 -364
- package/md_cg/semantic/en_zh_map.json +28694 -0
- package/md_cg/semantic/export_en_zh_map.py +64 -0
- package/md_cg/semantic/unify.py +45 -0
- package/md_cg/semantic/zh_en_atoms.py +139 -139
- package/md_cg/signer.py +7 -4
- package/md_cg/sources.py +816 -582
- package/md_cg/statushdr.py +179 -179
- package/md_cg/stg.py +54 -37
- package/md_cg/subgraph.py +729 -729
- package/md_cg/sustain.py +35 -5
- package/md_cg/tasks.py +470 -470
- package/md_cg/test_access_hints.py +147 -0
- package/md_cg/test_action_derive.py +203 -203
- package/md_cg/test_audit_rotate.py +270 -270
- package/md_cg/test_autonomy.py +143 -143
- package/md_cg/test_bench_governance.py +102 -102
- package/md_cg/test_blindspot_tickets.py +166 -166
- package/md_cg/test_branch_discard_tombstone.py +136 -0
- package/md_cg/test_branches.py +13 -3
- package/md_cg/test_ccg_perturb.py +184 -184
- package/md_cg/test_ccgc.py +433 -433
- package/md_cg/test_census_prune.py +81 -81
- package/md_cg/test_chain_read_isolate.py +168 -0
- package/md_cg/test_cond_compose_anchors.py +76 -76
- package/md_cg/test_cond_match.py +165 -165
- package/md_cg/test_condition_anchor.py +81 -81
- package/md_cg/test_d_meta.py +412 -412
- package/md_cg/test_datapath_device_name.py +203 -0
- package/md_cg/test_datapath_root.py +199 -199
- package/md_cg/test_emit_negtail_cache.py +156 -0
- package/md_cg/test_en_pipeline.py +22 -2
- package/md_cg/test_gain_gate.py +212 -212
- package/md_cg/test_govern_directread.py +421 -0
- package/md_cg/test_health_scale.py +173 -173
- package/md_cg/test_hive_ingest.py +285 -0
- package/md_cg/test_hot_cold.py +215 -215
- package/md_cg/test_hyperedge.py +245 -245
- package/md_cg/test_i26_empty_first_write.py +116 -0
- package/md_cg/test_i27_e041_identity.py +128 -0
- package/md_cg/test_i28_hotcache_prodpath.py +122 -0
- package/md_cg/test_i32_hotcache_env_key.py +218 -0
- package/md_cg/test_identity_attribution.py +96 -15
- package/md_cg/test_index_durability.py +17 -3
- package/md_cg/test_interop.py +95 -0
- package/md_cg/test_interop_judgment.py +228 -0
- package/md_cg/test_issue39_utf8_stdio.py +273 -0
- package/md_cg/test_lifecycle.py +309 -309
- package/md_cg/test_linkref.py +306 -306
- package/md_cg/test_links_concurrent_write.py +188 -0
- package/md_cg/test_lock.py +43 -43
- package/md_cg/test_md_access_parity.py +255 -255
- package/md_cg/test_md_writepath.py +345 -345
- package/md_cg/test_mdstore_search_parity.py +160 -0
- package/md_cg/test_merge_upsert.py +168 -0
- package/md_cg/test_mr_m2.py +587 -587
- package/md_cg/test_mr_m3.py +710 -710
- package/md_cg/test_mr_m4.py +485 -485
- package/md_cg/test_n123_derive_expiry_chain.py +205 -0
- package/md_cg/test_n130_verify_falsified_protect.py +185 -0
- package/md_cg/test_n131_merge_gate.py +205 -0
- package/md_cg/test_p0.py +250 -250
- package/md_cg/test_p1.py +316 -316
- package/md_cg/test_p10_identity.py +173 -173
- package/md_cg/test_p11_consistency.py +233 -233
- package/md_cg/test_p12_metacognition.py +212 -212
- package/md_cg/test_p13_encryption.py +241 -241
- package/md_cg/test_p14_sustain.py +249 -249
- package/md_cg/test_p15_scrub.py +280 -280
- package/md_cg/test_p16_self_state.py +301 -301
- package/md_cg/test_p17_predict.py +354 -354
- package/md_cg/test_p18_whitebox.py +171 -171
- package/md_cg/test_p19_migrate_roleplay.py +149 -149
- package/md_cg/test_p1x_ref_root.py +160 -0
- package/md_cg/test_p20_evolution.py +315 -315
- package/md_cg/test_p21_tokens.py +293 -270
- package/md_cg/test_p22_theory.py +175 -175
- package/md_cg/test_p23_links.py +311 -311
- package/md_cg/test_p24_evidence.py +227 -227
- package/md_cg/test_p25_weights.py +156 -156
- package/md_cg/test_p26_refindex.py +416 -416
- package/md_cg/test_p27_docindex.py +16 -7
- package/md_cg/test_p28_refcheck.py +305 -305
- package/md_cg/test_p29_session_ingest_export.py +354 -333
- package/md_cg/test_p2_mcp.py +3 -0
- package/md_cg/test_p3.py +11 -2
- package/md_cg/test_p30_maintain.py +330 -330
- package/md_cg/test_p31_insight.py +534 -534
- package/md_cg/test_p32_backfill.py +7 -1
- package/md_cg/test_p33_ccg_wiring.py +293 -293
- package/md_cg/test_p34_crosscheck.py +331 -331
- package/md_cg/test_p35_conditioned_claim.py +252 -252
- package/md_cg/test_p36_kp_align.py +230 -230
- package/md_cg/test_p37_condition_space.py +248 -248
- package/md_cg/test_p38_concurrent_flush.py +102 -0
- package/md_cg/test_p38_contextualize.py +273 -273
- package/md_cg/test_p39_verify_flow.py +153 -0
- package/md_cg/test_p39_vision_evidence.py +369 -369
- package/md_cg/test_p40_refine_worklist.py +241 -241
- package/md_cg/test_p41_evolve_patrol.py +224 -224
- package/md_cg/test_p42_provenance.py +269 -269
- package/md_cg/test_p43_pooling.py +412 -398
- package/md_cg/test_p44_md_whitebox.py +231 -231
- package/md_cg/test_p45_session_identity.py +219 -219
- package/md_cg/test_p46_unit_scope.py +272 -272
- package/md_cg/test_p47_session_view.py +316 -0
- package/md_cg/test_p4_fuzzy.py +223 -223
- package/md_cg/test_p5_semantic.py +226 -226
- package/md_cg/test_p6_consolidate.py +440 -387
- package/md_cg/test_p7_goals_recent.py +202 -202
- package/md_cg/test_p8_subgraph_chain.py +200 -200
- package/md_cg/test_p9_forget_protect.py +231 -231
- package/md_cg/test_predict_beta.py +135 -135
- package/md_cg/test_preflight_failclosed.py +100 -100
- package/md_cg/test_progressive.py +146 -146
- package/md_cg/test_propose_tail_index.py +157 -0
- package/md_cg/test_protocol.py +243 -243
- package/md_cg/test_reach.py +378 -378
- package/md_cg/test_reach_keys.py +201 -201
- package/md_cg/test_read_clip.py +141 -141
- package/md_cg/test_read_scope_b27.py +277 -0
- package/md_cg/test_readcache_default_on.py +168 -0
- package/md_cg/test_readcache_precise_inval.py +270 -0
- package/md_cg/test_readcache_prodpath.py +203 -0
- package/md_cg/test_reconcile_v0.py +294 -0
- package/md_cg/test_retr_gates_prodpath.py +140 -0
- package/md_cg/test_retr_s1.py +344 -340
- package/md_cg/test_retr_s1b.py +276 -209
- package/md_cg/test_retr_s3.py +194 -194
- package/md_cg/test_retr_s4.py +163 -163
- package/md_cg/test_retr_s5.py +200 -200
- package/md_cg/test_retr_s6.py +157 -157
- package/md_cg/test_retr_s7.py +392 -384
- package/md_cg/test_retr_s8_time.py +369 -316
- package/md_cg/test_retr_s9_edges.py +286 -286
- package/md_cg/test_retr_s9_entity_ctx.py +9 -3
- package/md_cg/test_retr_score_once.py +208 -0
- package/md_cg/test_review_conformance.py +367 -367
- package/md_cg/test_review_onepass.py +170 -0
- package/md_cg/test_role_views.py +354 -354
- package/md_cg/test_rrf_graph_seed_cache.py +154 -0
- package/md_cg/test_security_audit.py +155 -0
- package/md_cg/test_security_audit_b26.py +161 -0
- package/md_cg/test_security_audit_v21.py +250 -0
- package/md_cg/test_sem_noise.py +242 -242
- package/md_cg/test_semantic_canonical.py +16 -2
- package/md_cg/test_session_isolation.py +168 -0
- package/md_cg/test_snapshot_autoclose.py +187 -0
- package/md_cg/test_subproc_encoding.py +192 -192
- package/md_cg/test_sustain_mutual.py +153 -153
- package/md_cg/test_tail_watermark_race.py +208 -0
- package/md_cg/test_tasks.py +409 -409
- package/md_cg/test_tenant_env_override_warn.py +139 -0
- package/md_cg/test_tenant_registry_corrupt_warn.py +151 -0
- package/md_cg/test_tool_face.py +189 -189
- package/md_cg/test_transfer.py +180 -180
- package/md_cg/test_trust.py +361 -361
- package/md_cg/test_twophase.py +286 -286
- package/md_cg/test_v14_fixes.py +38 -20
- package/md_cg/test_validity_filter.py +280 -280
- package/md_cg/test_verify_answer.py +138 -138
- package/md_cg/test_verify_dirty_reconcile.py +157 -0
- package/md_cg/test_wisdom_md_store.py +292 -292
- package/md_cg/test_writelimit.py +197 -197
- package/md_cg/test_writepipe.py +214 -214
- package/md_cg/theory.py +6 -3
- package/md_cg/tokens.py +85 -14
- package/md_cg/tool_face.py +260 -260
- package/md_cg/trust.py +986 -950
- package/md_cg/twophase.py +231 -231
- package/md_cg/units.py +668 -667
- package/md_cg/vision_evidence.py +667 -666
- package/md_cg/weights.py +624 -624
- package/md_cg/whitebox.py +527 -527
- package/md_cg/whitebox_kb/__init__.py +37 -37
- package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
- package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
- package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
- package/md_cg/whitebox_kb/engine.py +310 -310
- package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
- package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
- package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
- package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
- package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
- package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
- package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
- package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
- package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
- package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
- package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
- package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
- package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
- package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
- package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
- package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
- package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
- package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
- package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
- package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
- package/md_cg/whitebox_kb/wisdom/multilang_ir.py +128 -124
- package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
- package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
- package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
- package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
- package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
- package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
- package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
- package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
- package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
- package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
- package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
- package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
- package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
- package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
- package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
- package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
- package/md_cg/writelimit.py +356 -356
- package/md_cg/writepipe.py +20 -8
- package/package.json +101 -96
- package/skills/plugin.json +54 -54
- package/skills/skills/designer-perspective/SKILL.md +158 -158
- package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
- package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
- package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
- package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
- package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
- package/skills/skills/designer-perspective/scripts/designer.py +545 -545
- package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
- package/skills/skills/designer-perspective/tests/selftest.py +61 -61
- package/skills/skills/lingshu-browser/SKILL.md +60 -60
- package/skills/skills/lingshu-compiler/SKILL.md +56 -56
- package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
- package/skills/skills/lingshu-graph/SKILL.md +63 -63
- package/skills/skills/lingshu-net/SKILL.md +48 -48
- package/skills/skills/lingshu-os/SKILL.md +64 -64
- package/skills/skills/lingshu-pylang/SKILL.md +71 -71
- package/src/bridge.ts +33 -0
- package/src/hooks.ts +38 -2
- package/src/index.ts +526 -518
- package/src/lib/datapath.ts +326 -326
- package/src/lib/mdcg_client.ts +413 -413
- package/src/lib/mutual.ts +428 -428
- package/src/lib/prompt_safety.ts +62 -62
- package/src/lib/python_path.ts +71 -71
- package/src/lib/roleplay_web.ts +116 -29
- package/src/lib/token_store.ts +13 -3
- package/src/tools.ts +212 -212
- package/zcode/AGENTS.md +11 -3
- package/zcode/README.md +41 -41
- /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
package/md_cg/codeindex.py
CHANGED
|
@@ -1,532 +1,532 @@
|
|
|
1
|
-
# -*- coding: utf-8 -*-
|
|
2
|
-
"""条件代码图:按注释/接口索引代码,不存完整代码。
|
|
3
|
-
|
|
4
|
-
设计(2026-09-09;2026-09-10 修订):
|
|
5
|
-
认知图通过「大域」(目录)索引;代码节点存的是**注释与接口**——模块 docstring、
|
|
6
|
-
签名、docstring、前置注释;正文一律不复制,用 frontmatter.code_ref 指回源文件。
|
|
7
|
-
读取走注释索引,零 LLM,纯 AST(弱提取器除外,见下)。
|
|
8
|
-
|
|
9
|
-
节点正文必须是 **CCG 6 行**(见 `render`):非 CCG 正文会被 `judge_qualification`
|
|
10
|
-
的第一步(ccg_completeness)直接判 **BLINDSPOT**,节点存进去了也检索不到可用结论。
|
|
11
|
-
|
|
12
|
-
提取器按后缀注册(`EXTRACTORS`):
|
|
13
|
-
.py → AST 提取,precise=True,区间精确到 end_lineno,基底 compiler
|
|
14
|
-
.ts/.tsx/.js/... → 正则弱提取,precise=False,区间为**上界**,基底 other(诚实降级)
|
|
15
|
-
|
|
16
|
-
区间哈希:每条目带 `hash`(被引用行的 sha1 前 12 位,见 `_region_hash`),
|
|
17
|
-
用于后续判断「索引出来的位置是不是已经漂了」——它不是内容寻址,只做变更探测。
|
|
18
|
-
"""
|
|
19
|
-
from __future__ import annotations
|
|
20
|
-
|
|
21
|
-
import ast
|
|
22
|
-
import hashlib
|
|
23
|
-
import os
|
|
24
|
-
import re
|
|
25
|
-
|
|
26
|
-
from . import nodefile
|
|
27
|
-
|
|
28
|
-
SKIP_DIRS = ("__pycache__", ".git", ".venv", "venv", "node_modules", ".mypy_cache")
|
|
29
|
-
MAX_DOC = 400
|
|
30
|
-
|
|
31
|
-
LANG_COMPILER = "compiler"
|
|
32
|
-
LANG_WEAK = "other"
|
|
33
|
-
|
|
34
|
-
# --------------------------------------------------------------------------
|
|
35
|
-
# 渲染契约代际(版本戳)
|
|
36
|
-
# --------------------------------------------------------------------------
|
|
37
|
-
# 生效条件:`render` 的**产物形态**发生不兼容变化时 +1(形态不变的重构不 +1);
|
|
38
|
-
# 由 refindex 写入 frontmatter.code_ref.render_version(节点侧),由 selfreport
|
|
39
|
-
# 随常驻进程自报(进程侧),两侧同取本常量——**同源,无第二处硬编码**。
|
|
40
|
-
#
|
|
41
|
-
# 用途(第③道防线,防「旧契约静默覆盖重建成果」):
|
|
42
|
-
# ① 节点侧:`scripts/mdcg_verify_render_meta.py` 据 code_ref.render_version 机械
|
|
43
|
-
# 判定「节点由哪一代 render 产出」,不再只靠形态启发式(正文含元条件行);
|
|
44
|
-
# ② 进程侧:`scripts/mdcg_stale_servers.py` 据自报值判定活进程代际,堵住
|
|
45
|
-
# 「进程启动时间晚于源码 mtime 故 stale=False、却持旧 render」的盲区
|
|
46
|
-
# (AGENTS.md §5 运维注记的活体实证:旧契约把全量重建成果刷回 old_synth)。
|
|
47
|
-
#
|
|
48
|
-
# 代际史:
|
|
49
|
-
# 1 = 旧契约(npm 0.4.8 及以前):合成区产出 `# 生效条件:载体/位置:…`,
|
|
50
|
-
# 索引元条件冒用 CCG 字段名(合成即冒充)。
|
|
51
|
-
# 2 = 三分区契约(2026-09-19):源码 CCG 区置首 → 合成 CCG 区(不产出生效
|
|
52
|
-
# 条件行)→ 索引元信息区(`# 索引元条件:…` 非 CCG 字段名 + 位置行)。
|
|
53
|
-
RENDER_VERSION = 2
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
# --------------------------------------------------------------------------
|
|
57
|
-
# Python:AST 提取(精确)
|
|
58
|
-
# --------------------------------------------------------------------------
|
|
59
|
-
# 生效条件:node 传入后,取 ast.get_docstring(node, clean=True) 的返回值,若该返回值为假值则回落空串,返回 strip 后截断到模块级 MAX_DOC 的文本;
|
|
60
|
-
def _doc_of(node):
|
|
61
|
-
return (ast.get_docstring(node, clean=True) or "").strip()[:MAX_DOC]
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
# 生效条件:lines 为源码行列表、lineno 为 1-based 定义行时,从 lines[lineno-2] 向上收集连续以 "#" 起始的行,遇空行且已收集到注释即停止,遇空行且未收集到注释则跳过继续,遇非注释非空行停止,返回按物理顺序排列的注释列表;lineno<=1 或初始无匹配时返回空列表;
|
|
65
|
-
def _leading_comments(lines, lineno):
|
|
66
|
-
"""定义行前的连续注释(# ...)。"""
|
|
67
|
-
out, i = [], lineno - 2
|
|
68
|
-
while i >= 0:
|
|
69
|
-
s = lines[i].strip()
|
|
70
|
-
if s.startswith("#"):
|
|
71
|
-
out.append(s)
|
|
72
|
-
i -= 1
|
|
73
|
-
elif not s and out:
|
|
74
|
-
break
|
|
75
|
-
elif not s:
|
|
76
|
-
i -= 1
|
|
77
|
-
else:
|
|
78
|
-
break
|
|
79
|
-
return list(reversed(out))
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
# 生效条件:node 具有真值 body 属性时返回 body[0].lineno;否则(body 为 None/假值/缺属性)返回 node.lineno;
|
|
83
|
-
def _body_first_line(node):
|
|
84
|
-
"""符号体首个语句的行号(1-based);无体 → 定义行本身(窗口为空)。"""
|
|
85
|
-
body = getattr(node, "body", None) or []
|
|
86
|
-
return body[0].lineno if body else node.lineno
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
# 生效条件:lines 为 source.split("\n") 得到的行列表、lineno 为 1-based 定义行、body_lineno 为 1-based 体首语句行时,在 end=max(lineno, body_lineno-1) 下扫描 lines[lineno:end],收集 strip 后以 "#" 起始的行并返回;body_lineno-1 <= lineno 时返回空列表;
|
|
90
|
-
def _body_comments(lines, lineno, body_lineno):
|
|
91
|
-
"""符号**体内首个语句之前**的连续 `#` 注释(定义行紧下方,声明头区)。
|
|
92
|
-
|
|
93
|
-
这是与 `_leading_comments`(定义行**之上**)并列的**第二个窗口**:
|
|
94
|
-
|
|
95
|
-
· 既有白箱单元库 104+ 处把 CCG 注释块写在**这里**
|
|
96
|
-
(例:`md_cg/whitebox_kb/wisdom/python_code_units.py` 的模板
|
|
97
|
-
`def tokenize(src):` 下一行即 ` # 生效条件:参数 src 合法`);
|
|
98
|
-
· 本函数是 body 窗口的**唯一真源**——`whitebox_kb/wisdom/verifier._ccg_block`
|
|
99
|
-
委托此处(历史文档里那句「与 `codeindex._body_comments` 同款语义」曾是
|
|
100
|
-
**悬空引用**:该名当时并不存在)。
|
|
101
|
-
|
|
102
|
-
参数:`lines`=源码行列表(`source.split("\\n")`);`lineno`=定义行(1-based);
|
|
103
|
-
`body_lineno`=体首个语句行(1-based)。窗口 = `lines[lineno : body_lineno-1]`
|
|
104
|
-
(0-based 切片:定义行之后 → 体首语句之前),只收 `#` 起始行。
|
|
105
|
-
语法上该窗口**结构性地只可能含注释/空行/docstring**——体首语句之前的区域。
|
|
106
|
-
"""
|
|
107
|
-
start = lineno # 0-based 索引 → 定义行的下一行
|
|
108
|
-
end = max(start, body_lineno - 1)
|
|
109
|
-
out = []
|
|
110
|
-
for ln in lines[start:end]:
|
|
111
|
-
s = ln.strip()
|
|
112
|
-
if s.startswith("#"):
|
|
113
|
-
out.append(s)
|
|
114
|
-
return out
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
# 生效条件:lines 为源码行列表且 node 含 lineno 时,返回 _leading_comments(lines, node.lineno) 与 _body_comments(lines, node.lineno, _body_first_line(node)) 的拼接结果(leading 在前、body 在后);
|
|
118
|
-
def _symbol_comments(lines, node):
|
|
119
|
-
"""符号的「源码 CCG 区」= leading 窗口 + body 窗口,**按物理行序**合并。
|
|
120
|
-
|
|
121
|
-
不发明额外优先级:两个窗口在源文件里的物理先后天然确定(leading 在定义行
|
|
122
|
-
之上、body 在其下),而检索侧 `mdcos._ccg_field` 取**首个**匹配——于是
|
|
123
|
-
「靠前者胜出」与「物理序」是同一件事,确定性可复算。
|
|
124
|
-
单窗口文件的行为与改造前逐字一致(只多收 body 窗口)。
|
|
125
|
-
"""
|
|
126
|
-
return (_leading_comments(lines, node.lineno)
|
|
127
|
-
+ _body_comments(lines, node.lineno, _body_first_line(node)))
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
# 生效条件:source 与 node 传入后,取 ast.get_source_segment(source, node) 的返回值,若抛 ValueError/TypeError 或返回假值则 seg 为空串,返回 seg.split("\n",1)[0].strip()[:200];
|
|
131
|
-
def _sig(source, node):
|
|
132
|
-
try:
|
|
133
|
-
seg = ast.get_source_segment(source, node) or ""
|
|
134
|
-
except (ValueError, TypeError):
|
|
135
|
-
seg = ""
|
|
136
|
-
return seg.split("\n", 1)[0].strip()[:200]
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
# 生效条件:tree 为 AST 根节点时,调用嵌套 rec(tree, "") 按 ast.iter_child_nodes 源码顺序递归产出 (定义节点, 所属类名);ClassDef 自身以当前 parent 产出并对其内部递归改用类名,FunctionDef/AsyncFunctionDef 以当前 parent 产出并保持 parent,其他节点递归保持 parent;
|
|
140
|
-
def _walk_defs(tree):
|
|
141
|
-
"""按**源码顺序**产出 (定义节点, 所属类名)。
|
|
142
|
-
|
|
143
|
-
不用 `ast.walk`:它给出的是广度优先、与源码顺序不一致,且丢掉父级归属。
|
|
144
|
-
父级归属是「子功能」与「不适用条件」两项的判定依据(同名方法必须能区分
|
|
145
|
-
是哪个类的),不能省。
|
|
146
|
-
"""
|
|
147
|
-
# 生效条件:node 为 AST 节点、parent 为当前所属类名字符串时,按 ast.iter_child_nodes(node) 顺序递归产出 (定义节点, 所属类名):ClassDef 以 parent 产出并递归改用 child.name,FunctionDef/AsyncFunctionDef 以 parent 产出并递归保持 parent,其他节点递归保持 parent;
|
|
148
|
-
def rec(node, parent):
|
|
149
|
-
for child in ast.iter_child_nodes(node):
|
|
150
|
-
if isinstance(child, ast.ClassDef):
|
|
151
|
-
yield child, parent
|
|
152
|
-
yield from rec(child, child.name)
|
|
153
|
-
elif isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
154
|
-
yield child, parent
|
|
155
|
-
yield from rec(child, parent)
|
|
156
|
-
else:
|
|
157
|
-
yield from rec(child, parent)
|
|
158
|
-
return rec(tree, "")
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
# 生效条件:source 可被 ast.parse 成功解析时,返回首项为 module 条目(path、name=os.path.basename(path) or "<module>"、lineno=1、end=len(source.split("\n"))、doc=_doc_of(tree))后接 _walk_defs(tree) 各定义条目的列表;source 触发 SyntaxError 时抛 ValueError(f"{path}:{exc.lineno}: {exc.msg}");
|
|
162
|
-
def _extract_python(source, path):
|
|
163
|
-
try:
|
|
164
|
-
tree = ast.parse(source)
|
|
165
|
-
except SyntaxError as exc:
|
|
166
|
-
raise ValueError(f"{path}:{exc.lineno}: {exc.msg}") from exc
|
|
167
|
-
lines = source.split("\n")
|
|
168
|
-
items = [{"path": path, "name": os.path.basename(path) or "<module>",
|
|
169
|
-
"kind": "module", "parent": "", "lineno": 1, "end": len(lines),
|
|
170
|
-
"sig": "", "doc": _doc_of(tree), "comments": []}]
|
|
171
|
-
for node, parent in _walk_defs(tree):
|
|
172
|
-
kind = ("class" if isinstance(node, ast.ClassDef)
|
|
173
|
-
else "async_def" if isinstance(node, ast.AsyncFunctionDef)
|
|
174
|
-
else "def")
|
|
175
|
-
items.append({
|
|
176
|
-
"path": path, "name": node.name, "kind": kind, "parent": parent,
|
|
177
|
-
"lineno": node.lineno, "end": getattr(node, "end_lineno", node.lineno),
|
|
178
|
-
"sig": _sig(source, node), "doc": _doc_of(node),
|
|
179
|
-
"comments": _symbol_comments(lines, node)})
|
|
180
|
-
return items
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
# --------------------------------------------------------------------------
|
|
184
|
-
# TS/JS:正则弱提取(不精确,区间为上界)
|
|
185
|
-
# --------------------------------------------------------------------------
|
|
186
|
-
_JS_DEF = re.compile(
|
|
187
|
-
r"^(?P<indent>[ \t]*)(?:export\s+)?(?:default\s+)?(?:declare\s+)?"
|
|
188
|
-
r"(?:abstract\s+)?(?:async\s+)?"
|
|
189
|
-
r"(?P<kind>class|interface|enum|type|function|const)\s+"
|
|
190
|
-
r"(?P<name>[A-Za-z_$][\w$]*)", re.M)
|
|
191
|
-
# 这些关键字只在模块作用域(缩进为空)才算对外接口,否则会把函数内的局部
|
|
192
|
-
# const 全捞进来,把索引淹掉。
|
|
193
|
-
_JS_MODULE_SCOPE_ONLY = ("const", "type", "enum")
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
# 生效条件:lines 为源码行列表、lineno 为 1-based 定义行时,从 lines[lineno-2] 向上收集连续以 "//" 开头的行注释,或遇到 strip 后以 "*/" 结尾的行时向上收集到首个 strip 后以 "/*" 开头的行(含该行)作为块注释;lookback 默认 25 限制收集行数,lookback=0 时循环不进入并返回空列表;遇空行且已有收集即停止,空行且未收集则跳过,其他行停止;
|
|
197
|
-
def _leading_js_comments(lines, lineno, lookback=25):
|
|
198
|
-
"""定义行前的连续行注释块,或紧邻的 /** ... */ 块。"""
|
|
199
|
-
out, i = [], lineno - 2
|
|
200
|
-
while i >= 0 and len(out) < lookback:
|
|
201
|
-
s = lines[i].strip()
|
|
202
|
-
if not s and out:
|
|
203
|
-
break
|
|
204
|
-
if s.endswith("*/"):
|
|
205
|
-
block = []
|
|
206
|
-
j = i
|
|
207
|
-
while j >= 0 and len(out) + len(block) < lookback:
|
|
208
|
-
block.append(lines[j].strip())
|
|
209
|
-
if lines[j].strip().startswith("/*"):
|
|
210
|
-
break
|
|
211
|
-
j -= 1
|
|
212
|
-
out.extend(block)
|
|
213
|
-
break
|
|
214
|
-
if s.startswith("//"):
|
|
215
|
-
out.append(s)
|
|
216
|
-
i -= 1
|
|
217
|
-
continue
|
|
218
|
-
if not s:
|
|
219
|
-
i -= 1
|
|
220
|
-
continue
|
|
221
|
-
break
|
|
222
|
-
return list(reversed(out))
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
# 生效条件:对 source 用 _JS_DEF.finditer 命中项生成条目(首项模块条目 name 为 os.path.basename(path) or "<module>"),跳过 kind 属于 _JS_MODULE_SCOPE_ONLY 且 indent 组非空的命中;其余命中按出现顺序取 lineno、kind、name、sig,end 为下一命中行号减一或 len(source.split("\n")) 且不小于 lineno,comments 由 _leading_js_comments(lines, lineno) 生成。
|
|
226
|
-
def _extract_weak(source, path):
|
|
227
|
-
"""正则弱提取:返回条目,`end` 为**上界**(到下一个定义之前),不保证精确。"""
|
|
228
|
-
lines = source.split("\n")
|
|
229
|
-
items = [{"path": path, "name": os.path.basename(path) or "<module>",
|
|
230
|
-
"kind": "module", "parent": "", "lineno": 1, "end": len(lines),
|
|
231
|
-
"sig": "", "doc": "", "comments": []}]
|
|
232
|
-
hits = []
|
|
233
|
-
for m in _JS_DEF.finditer(source):
|
|
234
|
-
kind, name = m.group("kind"), m.group("name")
|
|
235
|
-
if kind in _JS_MODULE_SCOPE_ONLY and m.group("indent"):
|
|
236
|
-
continue
|
|
237
|
-
lineno = source.count("\n", 0, m.start()) + 1
|
|
238
|
-
hits.append((lineno, kind, name, m.group(0).strip()))
|
|
239
|
-
for idx, (lineno, kind, name, sig) in enumerate(hits):
|
|
240
|
-
end = (hits[idx + 1][0] - 1) if idx + 1 < len(hits) else len(lines)
|
|
241
|
-
items.append({
|
|
242
|
-
"path": path, "name": name, "kind": kind, "parent": "",
|
|
243
|
-
"lineno": lineno, "end": max(lineno, end), "sig": sig[:200],
|
|
244
|
-
"doc": "", "comments": _leading_js_comments(lines, lineno)})
|
|
245
|
-
return items
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
# --------------------------------------------------------------------------
|
|
249
|
-
# 提取器注册表
|
|
250
|
-
# --------------------------------------------------------------------------
|
|
251
|
-
EXTRACTORS = {
|
|
252
|
-
".py": ("py", _extract_python, True, LANG_COMPILER),
|
|
253
|
-
".ts": ("ts", _extract_weak, False, LANG_WEAK),
|
|
254
|
-
".tsx": ("tsx", _extract_weak, False, LANG_WEAK),
|
|
255
|
-
".js": ("js", _extract_weak, False, LANG_WEAK),
|
|
256
|
-
".mjs": ("js", _extract_weak, False, LANG_WEAK),
|
|
257
|
-
".cjs": ("js", _extract_weak, False, LANG_WEAK),
|
|
258
|
-
}
|
|
259
|
-
SUFFIX = tuple(sorted(EXTRACTORS))
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
# 生效条件:lines 为行列表、lineno 与 end 为 1-based 行号时,对 "\n".join(lines[max(0, lineno-1):max(0, end)]) 的 UTF-8 字节求 sha1,返回其十六进制前 12 位;
|
|
263
|
-
def region_hash(lines, lineno, end):
|
|
264
|
-
"""被引用行的 sha1 前 12 位(变更探测用,非内容寻址)。
|
|
265
|
-
|
|
266
|
-
必须是**唯一**定义:索引侧与回读侧(`op=ref`)共用同一个函数。
|
|
267
|
-
两侧各写一份哈希算法,漂移检测就会悄悄失效(永远 hash_match=True)。
|
|
268
|
-
"""
|
|
269
|
-
seg = "\n".join(lines[max(0, lineno - 1):max(0, end)])
|
|
270
|
-
return hashlib.sha1(seg.encode("utf-8")).hexdigest()[:12]
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
# 生效条件:source 与 path 传入后,ext=suffix or os.path.splitext(path)[1].lower();当 ext 存在于模块级 EXTRACTORS 时,用对应 fn(source, path) 抽取并给每个条目补 lang/precise/basis/hash(hash 由 region_hash(lines, it["lineno"], it["end"]) 算)后返回;ext 不在 EXTRACTORS 时抛 ValueError(f"无提取器(suffix={ext or '<none>'})");
|
|
274
|
-
def extract(source, path="", suffix=None):
|
|
275
|
-
"""抽取一个文件的条目;按后缀分派提取器。语法错误抛 ValueError。
|
|
276
|
-
|
|
277
|
-
产出条目带 `lang` / `precise` / `hash`,供 `render` 与 frontmatter.code_ref 使用。
|
|
278
|
-
"""
|
|
279
|
-
ext = suffix or os.path.splitext(path)[1].lower()
|
|
280
|
-
if ext not in EXTRACTORS:
|
|
281
|
-
# 不静默降级成 Python 解析:那会把「没有提取器」伪装成「语法错误」,
|
|
282
|
-
# 让调用方误以为是源码的问题。直接报缺提取器,由 index_dir 收进 errors。
|
|
283
|
-
raise ValueError(f"无提取器(suffix={ext or '<none>'})")
|
|
284
|
-
lang, fn, precise, basis = EXTRACTORS[ext]
|
|
285
|
-
lines = source.split("\n")
|
|
286
|
-
items = fn(source, path)
|
|
287
|
-
for it in items:
|
|
288
|
-
it["lang"] = lang
|
|
289
|
-
it["precise"] = precise
|
|
290
|
-
it["basis"] = basis
|
|
291
|
-
it["hash"] = region_hash(lines, it["lineno"], it["end"])
|
|
292
|
-
return items
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
# 生效条件:item 为条目字典时,path=item.get("path") or ""、top=path.split("/")[0] or ".",按 item.get("precise", True)(缺键默认 True,键存在假值走弱提取)选择 LANG_COMPILER 或 LANG_WEAK 方法文本,返回 observation_position 用 top、time_window 用 nodefile.FULL_TIME_WINDOW_MIN 与 nodefile.FULL_TIME_WINDOW_MAX、observation_tool 用方法文本、existence_constraint 含 path 的四槽字典;
|
|
296
|
-
def condition_space(item):
|
|
297
|
-
"""条目 → 条件空间四槽(纯函数,**唯一来源**)。
|
|
298
|
-
|
|
299
|
-
为什么必须与 `render` 同源:正文的 `# 索引元条件:` 行与 frontmatter 的
|
|
300
|
-
`condition_space` 一旦各写一套,就会出现「正文有声明、条件空间是空的」
|
|
301
|
-
——`nodefile.condition_space_text(require_full=True)` 只看 frontmatter,
|
|
302
|
-
于是节点**存得进、判得了,条件空间却没声明**。改造前正是这样:正文写
|
|
303
|
-
「大域=X;检索…时」(第三种方言),frontmatter 只写 `observation_position`
|
|
304
|
-
**单槽**。单槽不是生效条件(见 nodefile.CONDITION_SLOTS_REQUIRED),
|
|
305
|
-
该四槽在正文里由「索引元条件」行承载(**不再**占用「生效条件」字段——
|
|
306
|
-
Phase 0 契约裁决,见 nodefile.INDEX_META_MARK),
|
|
307
|
-
故本函数按四槽齐备产出,供 `render` 与 `refindex.add_items` 共用。
|
|
308
|
-
|
|
309
|
-
时间槽用**全时窗哨兵**而非 `mdcg.add` 缺省补的「写入时刻锚定 1 小时窗」:
|
|
310
|
-
代码条目声明的是「源文件里存在这个符号」,其真值不随写入时刻衰减,
|
|
311
|
-
写成 1 小时观测窗是把写入副作用伪装成条件。
|
|
312
|
-
"""
|
|
313
|
-
path = item.get("path") or ""
|
|
314
|
-
top = path.split("/")[0] or "."
|
|
315
|
-
if item.get("precise", True):
|
|
316
|
-
method = f"{LANG_COMPILER}(AST 精确提取,区间精确到 end_lineno)"
|
|
317
|
-
else:
|
|
318
|
-
method = (f"{LANG_WEAK}(正则弱提取,未过编译器;"
|
|
319
|
-
f"区间为**上界**,以 op=ref 回读为准)")
|
|
320
|
-
return {
|
|
321
|
-
"observation_position": f"本地源码仓(大域={top})",
|
|
322
|
-
"time_window": [nodefile.FULL_TIME_WINDOW_MIN,
|
|
323
|
-
nodefile.FULL_TIME_WINDOW_MAX],
|
|
324
|
-
"observation_tool": method,
|
|
325
|
-
"existence_constraint": f"源文件 {path} 存在于本地仓且可读",
|
|
326
|
-
}
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
# 生效条件:item 为含 "name"、"kind"、"path"、"lineno"、"end" 键的条目字典时(缺这些必需键会 KeyError),返回由 item.get("comments") 的源码 CCG 区、合成 CCG 区、索引元信息区依次拼接的正文;parent/doc/comments/sig 按 item.get 缺键或假值回落,precise 缺键默认 True、键存在假值走弱提取,item["lineno"]/item["end"] 用于 basis 与位置行;
|
|
330
|
-
def render(item):
|
|
331
|
-
"""条目 → 正文三分区:源码 CCG 区(人工优先)→ 合成 CCG 区 → 索引元信息区。
|
|
332
|
-
|
|
333
|
-
**必须渲染成 CCG 格式**,这是本模块最容易踩的坑:
|
|
334
|
-
`judge_qualification` 第一步就查 `ccg_completeness` 的 5 要素
|
|
335
|
-
(功能名 / 子功能 / 执行 / 验证方式 / 不适用条件),缺任一即**直接判
|
|
336
|
-
BLINDSPOT**,后面的「verification_basis 缺失才 DEFER」根本走不到——
|
|
337
|
-
即便 frontmatter 已正确填了 verification_basis。改造前本函数只产
|
|
338
|
-
`# path::name` / `# sig` / `# doc:` 这类非 CCG 行,于是**所有代码节点
|
|
339
|
-
恒定 BLINDSPOT**:存得进、判不了、检索不到(与「目标节点的 CCG 渲染」
|
|
340
|
-
是同一策略,见 mdcg.py 的对应注释)。
|
|
341
|
-
|
|
342
|
-
Phase 0 修复(行序压制 + 字段语义分家;契约见
|
|
343
|
-
docs/mdcg/代码评审与条件化注释_契约_v0.1.md):
|
|
344
|
-
① **源码 CCG 区置首**——mdcos._ccg_field 取**首个**匹配,置首即
|
|
345
|
-
「人工优先」的确定性序:人工声明的生效条件不再被合成行压制;
|
|
346
|
-
② 合成区**不再产出生效条件行**——索引元条件不是功能前置条件
|
|
347
|
-
(裁定见 nodefile.INDEX_META_MARK)。故源码未声明的条目会**诚实地
|
|
348
|
-
缺该要素(BLINDSPOT)**,而不是被元条件冒充成 DEFER;
|
|
349
|
-
③ 索引元信息区改用非 CCG 字段名(索引元条件行 + 位置行),两个语义
|
|
350
|
-
不再挤同一个字段名(可机械判:nodefile.is_ccg_mark)。
|
|
351
|
-
"""
|
|
352
|
-
name = item["name"]
|
|
353
|
-
kind = item["kind"]
|
|
354
|
-
path = item["path"]
|
|
355
|
-
parent = item.get("parent") or ""
|
|
356
|
-
doc = (item.get("doc") or "").replace("\n", " ").strip()
|
|
357
|
-
comments = [c.lstrip("#").strip() for c in (item.get("comments") or [])]
|
|
358
|
-
sub = doc or (comments[0] if comments else "") or f"{kind} 定义在 {path},无注释"
|
|
359
|
-
sig = (item.get("sig") or "").strip() or "(模块级,无签名)"
|
|
360
|
-
if item.get("precise", True):
|
|
361
|
-
basis = (f"{LANG_COMPILER}(AST 已解析,区间精确:"
|
|
362
|
-
f"{path} L{item['lineno']}-L{item['end']})")
|
|
363
|
-
else:
|
|
364
|
-
basis = (f"{LANG_WEAK}(正则弱提取,未过编译器;区间为**上界**,"
|
|
365
|
-
f"以 op=ref 回读为准:{path} L{item['lineno']}-L{item['end']})")
|
|
366
|
-
# ---- 三分区组装(顺序即语义,不许随手改)------------------------------
|
|
367
|
-
# ① 源码 CCG 区:人工/源码声明逐字保留,**置首**取得「首个匹配」优先权
|
|
368
|
-
# ② 合成 CCG 区:机械补齐 5 要素,保证 ccg_completeness 不因缺行整体失效
|
|
369
|
-
# ③ 索引元信息区:非 CCG 字段名 + 位置行(与 CCG_MARKS 零重名)
|
|
370
|
-
lines = ["# " + c for c in comments]
|
|
371
|
-
lines += [
|
|
372
|
-
f"# 功能名:{name}({kind})",
|
|
373
|
-
f"# 子功能:{parent + '.' if parent else ''}{sub[:MAX_DOC]}",
|
|
374
|
-
f"# 执行:{sig}",
|
|
375
|
-
f"# 验证方式:{basis}",
|
|
376
|
-
"# 不适用条件:其它大域的**同名**符号(同名不同域时以 path 区分;"
|
|
377
|
-
f"本条目属于 {path})",
|
|
378
|
-
# 索引元条件**不占用**生效条件字段:它是「条目在何处/何时可被观测」,
|
|
379
|
-
# 不是「这段代码在何种输入下正确」。合成即冒充(nodefile.INDEX_META_MARK)。
|
|
380
|
-
f"# {nodefile.INDEX_META_MARK}:"
|
|
381
|
-
+ nodefile.condition_space_text(condition_space(item),
|
|
382
|
-
require_full=False),
|
|
383
|
-
f"# 位置:{path}:{item['lineno']}-{item['end']}"
|
|
384
|
-
f"({item.get('lang')},precise={bool(item.get('precise', True))})",
|
|
385
|
-
]
|
|
386
|
-
return "\n".join(lines)
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
# 生效条件:item 为含 "path" 与 "name" 键的字典时(缺任一键会 KeyError),返回 "code_" 加 (item["path"] + "::" + item["name"]).encode("utf-8") 的 sha1 十六进制前 12 位;
|
|
390
|
-
def node_id(item):
|
|
391
|
-
"""稳定 id:path::name 的短哈希(重复索引幂等)。"""
|
|
392
|
-
key = (item["path"] + "::" + item["name"]).encode("utf-8")
|
|
393
|
-
return "code_" + hashlib.sha1(key).hexdigest()[:12]
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
# 生效条件:skip_dirs 传入后,遍历 (skip_dirs or ()) 把每项 str(raw).strip().replace("\\","/").strip("/"),空串跳过;含 "/" 的加入 paths,不含 "/" 的加入 names;若归一化后无规则返回 (None, []),否则返回 (hit, rules),其中 hit(rel_dir, base) 在 base 命中 names 或 rel_dir 等于/前缀匹配 paths 中某条加 "/" 时为 True;
|
|
397
|
-
def skip_matcher(skip_dirs):
|
|
398
|
-
"""把调用方的 `skip_dirs` 编译成「该子目录是否排除」的判定 `hit(rel_dir, base)`。
|
|
399
|
-
|
|
400
|
-
与 `SKIP_DIRS` 在**同一处**生效,且**只增不减**:调用方只能追加排除,不能拿掉
|
|
401
|
-
`.git`/`.venv` 这类内置保护——否则一次参数写错就能把版本库元数据索引进认知图。
|
|
402
|
-
规则口径(两类可混用):
|
|
403
|
-
· 含 `/` → 按**相对 root 的路径**匹配(`docs/experiments` 只排这一处);
|
|
404
|
-
· 不含 `/` → 按**目录名**匹配(`experiments` 排任意层级的同名目录)。
|
|
405
|
-
反斜杠与首尾斜杠一律归一,避免「规则传了却不生效」这类静默失配。
|
|
406
|
-
|
|
407
|
-
返回 `(hit, rules)`;`rules` 为空时 `hit is None` → 调用方走原路径,
|
|
408
|
-
保证**默认行为与改造前逐字一致**(与 `fresh`/`on_file` 同一纪律)。
|
|
409
|
-
排除的**理由**:`.gitignore` 整目录忽略的实验产物物理仍在盘上,会把
|
|
410
|
-
`max_files` 撑爆并让「索引不全」变成常态;用「显式排除 + 回报」比「调大上限」
|
|
411
|
-
诚实。`docindex` 复用本函数(唯一实现,避免两处口径漂移)。
|
|
412
|
-
"""
|
|
413
|
-
rules, names, paths = [], set(), []
|
|
414
|
-
for raw in (skip_dirs or ()):
|
|
415
|
-
s = str(raw).strip().replace("\\", "/").strip("/")
|
|
416
|
-
if not s:
|
|
417
|
-
continue
|
|
418
|
-
rules.append(s)
|
|
419
|
-
if "/" in s:
|
|
420
|
-
paths.append(s)
|
|
421
|
-
else:
|
|
422
|
-
names.add(s)
|
|
423
|
-
if not rules:
|
|
424
|
-
return None, []
|
|
425
|
-
|
|
426
|
-
# 生效条件:在 skip_matcher 返回的闭包中,rel_dir 与 base 传入后,若 base 命中由 skip_dirs 归一化出的不含 "/" 的目录名集合 names 则返回 True;否则若 rel_dir 等于或以其某个含 "/" 的路径规则 paths 加 "/" 为前缀则返回 True;两者都不满足返回 False;
|
|
427
|
-
def hit(rel_dir, base):
|
|
428
|
-
if base in names:
|
|
429
|
-
return True
|
|
430
|
-
return any(rel_dir == p or rel_dir.startswith(p + "/") for p in paths)
|
|
431
|
-
|
|
432
|
-
return hit, rules
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
# 生效条件:当 root 为可 os.walk 的目录时返回 (items, errors, stats);patterns 为 None 时按模块常量 SUFFIX 取后缀,命中 max_files 或 max_items 上限则在 stats['truncated'] 上报截断,fresh 为 None 时逐文件读盘。
|
|
436
|
-
def index_dir(root, patterns=None, max_files=500, max_items=2000,
|
|
437
|
-
fresh=None, on_file=None, skip_dirs=None):
|
|
438
|
-
"""按大域(目录)遍历代码,产出 `(items, errors, stats)`。零 LLM。
|
|
439
|
-
|
|
440
|
-
`stats["truncated"]` 必须显式上报——截断**不再是静默的**:改造前达到上限
|
|
441
|
-
直接 `return`,调用方只看到 `indexed`/`error_count`,**索引不全却不告警**,
|
|
442
|
-
于是「不完整」被当成「完整」用。同时上报 `skipped_suffixes`:扫到但没被
|
|
443
|
-
索引的后缀要能看见,否则「不漏召回」这句话无法审计。
|
|
444
|
-
|
|
445
|
-
`fresh(rel, fp)` / `on_file(rel, fp, items)` 是给 `refindex.Ledger` 留的
|
|
446
|
-
增量钩子(默认 None → 行为与改造前逐字一致):
|
|
447
|
-
· `fresh` 返回 True → 该文件自上次索引后未变,**不读盘**直接跳过,
|
|
448
|
-
计入 `skipped_unchanged`(仍计入 `files`,故截断语义不变);
|
|
449
|
-
· `on_file` 在成功提取后回调,用于记录水位。
|
|
450
|
-
|
|
451
|
-
`skip_dirs` 是**追加**排除(见 `skip_matcher`):命中的目录整棵剪掉、不计入
|
|
452
|
-
`files`;实际排掉了哪些目录写进 `stats["skipped_dirs"]`——**排除与截断一样不许
|
|
453
|
-
静默**,否则「节点数变少」会被误读成「源文件真的少了」。
|
|
454
|
-
"""
|
|
455
|
-
pats = tuple(patterns or SUFFIX)
|
|
456
|
-
hit_skip, skip_rules = skip_matcher(skip_dirs)
|
|
457
|
-
items, errors, files = [], [], 0
|
|
458
|
-
hit_files = 0
|
|
459
|
-
seen_suffix = set()
|
|
460
|
-
stats = {"root": root, "patterns": list(pats), "files": 0, "truncated": False,
|
|
461
|
-
"truncated_reason": "", "max_files": max_files, "max_items": max_items,
|
|
462
|
-
"skipped_suffixes": [], "skipped_unchanged": 0,
|
|
463
|
-
"skip_dirs": list(skip_rules), "skipped_dirs": [],
|
|
464
|
-
# 截断**可复算**:命中(后缀匹配)文件数与本轮已产出条目数。
|
|
465
|
-
# 与 truncated_reason 一起读,调用方才能核对「差多少」而不是只看到"被截断了"。
|
|
466
|
-
"hit_files": 0, "indexed_items": 0}
|
|
467
|
-
|
|
468
|
-
# 生效条件:无参闭包,只在本次扫库作用域内可调用;把 files / hit_files / indexed_items / skipped_suffixes / skipped_dirs 一次性落进 stats,三处 return 共用同一口径;
|
|
469
|
-
def _snap():
|
|
470
|
-
"""把「截断相关」计数一次性落进 stats(三处 return 共用,避免各处口径漂移)。"""
|
|
471
|
-
stats["files"] = files
|
|
472
|
-
stats["hit_files"] = hit_files
|
|
473
|
-
stats["indexed_items"] = len(items)
|
|
474
|
-
stats["skipped_suffixes"] = sorted(
|
|
475
|
-
s for s in seen_suffix if s and s not in pats)[:12]
|
|
476
|
-
|
|
477
|
-
for dirpath, dirnames, filenames in os.walk(root):
|
|
478
|
-
rel_dir = os.path.relpath(dirpath, root).replace("\\", "/")
|
|
479
|
-
if rel_dir == ".":
|
|
480
|
-
rel_dir = ""
|
|
481
|
-
keep = []
|
|
482
|
-
for d in dirnames:
|
|
483
|
-
if d in SKIP_DIRS:
|
|
484
|
-
continue
|
|
485
|
-
child = f"{rel_dir}/{d}" if rel_dir else d
|
|
486
|
-
if hit_skip is not None and hit_skip(child, d):
|
|
487
|
-
# 就地追加、不依赖末尾汇总:截断提前 return 时也带得走(同 skipped_suffixes)。
|
|
488
|
-
stats["skipped_dirs"].append(child)
|
|
489
|
-
continue
|
|
490
|
-
keep.append(d)
|
|
491
|
-
# 目录序**必须确定性**:os.walk 给出的 dirnames 顺序由文件系统决定,
|
|
492
|
-
# 同一次输入两次运行可能不同 → 索引结果不可复算。排序后 walk 顺序唯一。
|
|
493
|
-
dirnames[:] = sorted(keep)
|
|
494
|
-
for fn in sorted(filenames):
|
|
495
|
-
ext = os.path.splitext(fn)[1].lower()
|
|
496
|
-
seen_suffix.add(ext)
|
|
497
|
-
if not fn.lower().endswith(pats):
|
|
498
|
-
continue
|
|
499
|
-
hit_files += 1
|
|
500
|
-
if files >= max_files or len(items) >= max_items:
|
|
501
|
-
stats["truncated"] = True
|
|
502
|
-
stats["truncated_reason"] = (
|
|
503
|
-
f"files={files}>=max_files={max_files}"
|
|
504
|
-
if files >= max_files else
|
|
505
|
-
f"items={len(items)}>=max_items={max_items}")
|
|
506
|
-
_snap()
|
|
507
|
-
return items, errors, stats
|
|
508
|
-
files += 1
|
|
509
|
-
fp = os.path.join(dirpath, fn)
|
|
510
|
-
rel = os.path.relpath(fp, root).replace("\\", "/")
|
|
511
|
-
if fresh is not None and fresh(rel, fp):
|
|
512
|
-
stats["skipped_unchanged"] += 1
|
|
513
|
-
continue
|
|
514
|
-
try:
|
|
515
|
-
with open(fp, encoding="utf-8") as f:
|
|
516
|
-
src = f.read()
|
|
517
|
-
got = extract(src, rel)
|
|
518
|
-
items.extend(got)
|
|
519
|
-
if on_file is not None:
|
|
520
|
-
on_file(rel, fp, got)
|
|
521
|
-
if len(items) >= max_items:
|
|
522
|
-
# 单文件就可能越限:越限即记截断并立刻停,不装看不见、
|
|
523
|
-
# 也不继续往下扫(继续扫只会让「截断」这件事更不明显)。
|
|
524
|
-
stats["truncated"] = True
|
|
525
|
-
stats["truncated_reason"] = (
|
|
526
|
-
f"items={len(items)}>=max_items={max_items}")
|
|
527
|
-
_snap()
|
|
528
|
-
return items, errors, stats
|
|
529
|
-
except (OSError, UnicodeDecodeError, ValueError) as exc:
|
|
530
|
-
errors.append(f"{rel}: {exc}")
|
|
531
|
-
_snap()
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""条件代码图:按注释/接口索引代码,不存完整代码。
|
|
3
|
+
|
|
4
|
+
设计(2026-09-09;2026-09-10 修订):
|
|
5
|
+
认知图通过「大域」(目录)索引;代码节点存的是**注释与接口**——模块 docstring、
|
|
6
|
+
签名、docstring、前置注释;正文一律不复制,用 frontmatter.code_ref 指回源文件。
|
|
7
|
+
读取走注释索引,零 LLM,纯 AST(弱提取器除外,见下)。
|
|
8
|
+
|
|
9
|
+
节点正文必须是 **CCG 6 行**(见 `render`):非 CCG 正文会被 `judge_qualification`
|
|
10
|
+
的第一步(ccg_completeness)直接判 **BLINDSPOT**,节点存进去了也检索不到可用结论。
|
|
11
|
+
|
|
12
|
+
提取器按后缀注册(`EXTRACTORS`):
|
|
13
|
+
.py → AST 提取,precise=True,区间精确到 end_lineno,基底 compiler
|
|
14
|
+
.ts/.tsx/.js/... → 正则弱提取,precise=False,区间为**上界**,基底 other(诚实降级)
|
|
15
|
+
|
|
16
|
+
区间哈希:每条目带 `hash`(被引用行的 sha1 前 12 位,见 `_region_hash`),
|
|
17
|
+
用于后续判断「索引出来的位置是不是已经漂了」——它不是内容寻址,只做变更探测。
|
|
18
|
+
"""
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import ast
|
|
22
|
+
import hashlib
|
|
23
|
+
import os
|
|
24
|
+
import re
|
|
25
|
+
|
|
26
|
+
from . import nodefile
|
|
27
|
+
|
|
28
|
+
SKIP_DIRS = ("__pycache__", ".git", ".venv", "venv", "node_modules", ".mypy_cache")
|
|
29
|
+
MAX_DOC = 400
|
|
30
|
+
|
|
31
|
+
LANG_COMPILER = "compiler"
|
|
32
|
+
LANG_WEAK = "other"
|
|
33
|
+
|
|
34
|
+
# --------------------------------------------------------------------------
|
|
35
|
+
# 渲染契约代际(版本戳)
|
|
36
|
+
# --------------------------------------------------------------------------
|
|
37
|
+
# 生效条件:`render` 的**产物形态**发生不兼容变化时 +1(形态不变的重构不 +1);
|
|
38
|
+
# 由 refindex 写入 frontmatter.code_ref.render_version(节点侧),由 selfreport
|
|
39
|
+
# 随常驻进程自报(进程侧),两侧同取本常量——**同源,无第二处硬编码**。
|
|
40
|
+
#
|
|
41
|
+
# 用途(第③道防线,防「旧契约静默覆盖重建成果」):
|
|
42
|
+
# ① 节点侧:`scripts/mdcg_verify_render_meta.py` 据 code_ref.render_version 机械
|
|
43
|
+
# 判定「节点由哪一代 render 产出」,不再只靠形态启发式(正文含元条件行);
|
|
44
|
+
# ② 进程侧:`scripts/mdcg_stale_servers.py` 据自报值判定活进程代际,堵住
|
|
45
|
+
# 「进程启动时间晚于源码 mtime 故 stale=False、却持旧 render」的盲区
|
|
46
|
+
# (AGENTS.md §5 运维注记的活体实证:旧契约把全量重建成果刷回 old_synth)。
|
|
47
|
+
#
|
|
48
|
+
# 代际史:
|
|
49
|
+
# 1 = 旧契约(npm 0.4.8 及以前):合成区产出 `# 生效条件:载体/位置:…`,
|
|
50
|
+
# 索引元条件冒用 CCG 字段名(合成即冒充)。
|
|
51
|
+
# 2 = 三分区契约(2026-09-19):源码 CCG 区置首 → 合成 CCG 区(不产出生效
|
|
52
|
+
# 条件行)→ 索引元信息区(`# 索引元条件:…` 非 CCG 字段名 + 位置行)。
|
|
53
|
+
RENDER_VERSION = 2
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
# --------------------------------------------------------------------------
|
|
57
|
+
# Python:AST 提取(精确)
|
|
58
|
+
# --------------------------------------------------------------------------
|
|
59
|
+
# 生效条件:node 传入后,取 ast.get_docstring(node, clean=True) 的返回值,若该返回值为假值则回落空串,返回 strip 后截断到模块级 MAX_DOC 的文本;
|
|
60
|
+
def _doc_of(node):
|
|
61
|
+
return (ast.get_docstring(node, clean=True) or "").strip()[:MAX_DOC]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
# 生效条件:lines 为源码行列表、lineno 为 1-based 定义行时,从 lines[lineno-2] 向上收集连续以 "#" 起始的行,遇空行且已收集到注释即停止,遇空行且未收集到注释则跳过继续,遇非注释非空行停止,返回按物理顺序排列的注释列表;lineno<=1 或初始无匹配时返回空列表;
|
|
65
|
+
def _leading_comments(lines, lineno):
|
|
66
|
+
"""定义行前的连续注释(# ...)。"""
|
|
67
|
+
out, i = [], lineno - 2
|
|
68
|
+
while i >= 0:
|
|
69
|
+
s = lines[i].strip()
|
|
70
|
+
if s.startswith("#"):
|
|
71
|
+
out.append(s)
|
|
72
|
+
i -= 1
|
|
73
|
+
elif not s and out:
|
|
74
|
+
break
|
|
75
|
+
elif not s:
|
|
76
|
+
i -= 1
|
|
77
|
+
else:
|
|
78
|
+
break
|
|
79
|
+
return list(reversed(out))
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
# 生效条件:node 具有真值 body 属性时返回 body[0].lineno;否则(body 为 None/假值/缺属性)返回 node.lineno;
|
|
83
|
+
def _body_first_line(node):
|
|
84
|
+
"""符号体首个语句的行号(1-based);无体 → 定义行本身(窗口为空)。"""
|
|
85
|
+
body = getattr(node, "body", None) or []
|
|
86
|
+
return body[0].lineno if body else node.lineno
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
# 生效条件:lines 为 source.split("\n") 得到的行列表、lineno 为 1-based 定义行、body_lineno 为 1-based 体首语句行时,在 end=max(lineno, body_lineno-1) 下扫描 lines[lineno:end],收集 strip 后以 "#" 起始的行并返回;body_lineno-1 <= lineno 时返回空列表;
|
|
90
|
+
def _body_comments(lines, lineno, body_lineno):
|
|
91
|
+
"""符号**体内首个语句之前**的连续 `#` 注释(定义行紧下方,声明头区)。
|
|
92
|
+
|
|
93
|
+
这是与 `_leading_comments`(定义行**之上**)并列的**第二个窗口**:
|
|
94
|
+
|
|
95
|
+
· 既有白箱单元库 104+ 处把 CCG 注释块写在**这里**
|
|
96
|
+
(例:`md_cg/whitebox_kb/wisdom/python_code_units.py` 的模板
|
|
97
|
+
`def tokenize(src):` 下一行即 ` # 生效条件:参数 src 合法`);
|
|
98
|
+
· 本函数是 body 窗口的**唯一真源**——`whitebox_kb/wisdom/verifier._ccg_block`
|
|
99
|
+
委托此处(历史文档里那句「与 `codeindex._body_comments` 同款语义」曾是
|
|
100
|
+
**悬空引用**:该名当时并不存在)。
|
|
101
|
+
|
|
102
|
+
参数:`lines`=源码行列表(`source.split("\\n")`);`lineno`=定义行(1-based);
|
|
103
|
+
`body_lineno`=体首个语句行(1-based)。窗口 = `lines[lineno : body_lineno-1]`
|
|
104
|
+
(0-based 切片:定义行之后 → 体首语句之前),只收 `#` 起始行。
|
|
105
|
+
语法上该窗口**结构性地只可能含注释/空行/docstring**——体首语句之前的区域。
|
|
106
|
+
"""
|
|
107
|
+
start = lineno # 0-based 索引 → 定义行的下一行
|
|
108
|
+
end = max(start, body_lineno - 1)
|
|
109
|
+
out = []
|
|
110
|
+
for ln in lines[start:end]:
|
|
111
|
+
s = ln.strip()
|
|
112
|
+
if s.startswith("#"):
|
|
113
|
+
out.append(s)
|
|
114
|
+
return out
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
# 生效条件:lines 为源码行列表且 node 含 lineno 时,返回 _leading_comments(lines, node.lineno) 与 _body_comments(lines, node.lineno, _body_first_line(node)) 的拼接结果(leading 在前、body 在后);
|
|
118
|
+
def _symbol_comments(lines, node):
|
|
119
|
+
"""符号的「源码 CCG 区」= leading 窗口 + body 窗口,**按物理行序**合并。
|
|
120
|
+
|
|
121
|
+
不发明额外优先级:两个窗口在源文件里的物理先后天然确定(leading 在定义行
|
|
122
|
+
之上、body 在其下),而检索侧 `mdcos._ccg_field` 取**首个**匹配——于是
|
|
123
|
+
「靠前者胜出」与「物理序」是同一件事,确定性可复算。
|
|
124
|
+
单窗口文件的行为与改造前逐字一致(只多收 body 窗口)。
|
|
125
|
+
"""
|
|
126
|
+
return (_leading_comments(lines, node.lineno)
|
|
127
|
+
+ _body_comments(lines, node.lineno, _body_first_line(node)))
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
# 生效条件:source 与 node 传入后,取 ast.get_source_segment(source, node) 的返回值,若抛 ValueError/TypeError 或返回假值则 seg 为空串,返回 seg.split("\n",1)[0].strip()[:200];
|
|
131
|
+
def _sig(source, node):
|
|
132
|
+
try:
|
|
133
|
+
seg = ast.get_source_segment(source, node) or ""
|
|
134
|
+
except (ValueError, TypeError):
|
|
135
|
+
seg = ""
|
|
136
|
+
return seg.split("\n", 1)[0].strip()[:200]
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
# 生效条件:tree 为 AST 根节点时,调用嵌套 rec(tree, "") 按 ast.iter_child_nodes 源码顺序递归产出 (定义节点, 所属类名);ClassDef 自身以当前 parent 产出并对其内部递归改用类名,FunctionDef/AsyncFunctionDef 以当前 parent 产出并保持 parent,其他节点递归保持 parent;
|
|
140
|
+
def _walk_defs(tree):
|
|
141
|
+
"""按**源码顺序**产出 (定义节点, 所属类名)。
|
|
142
|
+
|
|
143
|
+
不用 `ast.walk`:它给出的是广度优先、与源码顺序不一致,且丢掉父级归属。
|
|
144
|
+
父级归属是「子功能」与「不适用条件」两项的判定依据(同名方法必须能区分
|
|
145
|
+
是哪个类的),不能省。
|
|
146
|
+
"""
|
|
147
|
+
# 生效条件:node 为 AST 节点、parent 为当前所属类名字符串时,按 ast.iter_child_nodes(node) 顺序递归产出 (定义节点, 所属类名):ClassDef 以 parent 产出并递归改用 child.name,FunctionDef/AsyncFunctionDef 以 parent 产出并递归保持 parent,其他节点递归保持 parent;
|
|
148
|
+
def rec(node, parent):
|
|
149
|
+
for child in ast.iter_child_nodes(node):
|
|
150
|
+
if isinstance(child, ast.ClassDef):
|
|
151
|
+
yield child, parent
|
|
152
|
+
yield from rec(child, child.name)
|
|
153
|
+
elif isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
154
|
+
yield child, parent
|
|
155
|
+
yield from rec(child, parent)
|
|
156
|
+
else:
|
|
157
|
+
yield from rec(child, parent)
|
|
158
|
+
return rec(tree, "")
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
# 生效条件:source 可被 ast.parse 成功解析时,返回首项为 module 条目(path、name=os.path.basename(path) or "<module>"、lineno=1、end=len(source.split("\n"))、doc=_doc_of(tree))后接 _walk_defs(tree) 各定义条目的列表;source 触发 SyntaxError 时抛 ValueError(f"{path}:{exc.lineno}: {exc.msg}");
|
|
162
|
+
def _extract_python(source, path):
|
|
163
|
+
try:
|
|
164
|
+
tree = ast.parse(source)
|
|
165
|
+
except SyntaxError as exc:
|
|
166
|
+
raise ValueError(f"{path}:{exc.lineno}: {exc.msg}") from exc
|
|
167
|
+
lines = source.split("\n")
|
|
168
|
+
items = [{"path": path, "name": os.path.basename(path) or "<module>",
|
|
169
|
+
"kind": "module", "parent": "", "lineno": 1, "end": len(lines),
|
|
170
|
+
"sig": "", "doc": _doc_of(tree), "comments": []}]
|
|
171
|
+
for node, parent in _walk_defs(tree):
|
|
172
|
+
kind = ("class" if isinstance(node, ast.ClassDef)
|
|
173
|
+
else "async_def" if isinstance(node, ast.AsyncFunctionDef)
|
|
174
|
+
else "def")
|
|
175
|
+
items.append({
|
|
176
|
+
"path": path, "name": node.name, "kind": kind, "parent": parent,
|
|
177
|
+
"lineno": node.lineno, "end": getattr(node, "end_lineno", node.lineno),
|
|
178
|
+
"sig": _sig(source, node), "doc": _doc_of(node),
|
|
179
|
+
"comments": _symbol_comments(lines, node)})
|
|
180
|
+
return items
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
# --------------------------------------------------------------------------
|
|
184
|
+
# TS/JS:正则弱提取(不精确,区间为上界)
|
|
185
|
+
# --------------------------------------------------------------------------
|
|
186
|
+
_JS_DEF = re.compile(
|
|
187
|
+
r"^(?P<indent>[ \t]*)(?:export\s+)?(?:default\s+)?(?:declare\s+)?"
|
|
188
|
+
r"(?:abstract\s+)?(?:async\s+)?"
|
|
189
|
+
r"(?P<kind>class|interface|enum|type|function|const)\s+"
|
|
190
|
+
r"(?P<name>[A-Za-z_$][\w$]*)", re.M)
|
|
191
|
+
# 这些关键字只在模块作用域(缩进为空)才算对外接口,否则会把函数内的局部
|
|
192
|
+
# const 全捞进来,把索引淹掉。
|
|
193
|
+
_JS_MODULE_SCOPE_ONLY = ("const", "type", "enum")
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
# 生效条件:lines 为源码行列表、lineno 为 1-based 定义行时,从 lines[lineno-2] 向上收集连续以 "//" 开头的行注释,或遇到 strip 后以 "*/" 结尾的行时向上收集到首个 strip 后以 "/*" 开头的行(含该行)作为块注释;lookback 默认 25 限制收集行数,lookback=0 时循环不进入并返回空列表;遇空行且已有收集即停止,空行且未收集则跳过,其他行停止;
|
|
197
|
+
def _leading_js_comments(lines, lineno, lookback=25):
|
|
198
|
+
"""定义行前的连续行注释块,或紧邻的 /** ... */ 块。"""
|
|
199
|
+
out, i = [], lineno - 2
|
|
200
|
+
while i >= 0 and len(out) < lookback:
|
|
201
|
+
s = lines[i].strip()
|
|
202
|
+
if not s and out:
|
|
203
|
+
break
|
|
204
|
+
if s.endswith("*/"):
|
|
205
|
+
block = []
|
|
206
|
+
j = i
|
|
207
|
+
while j >= 0 and len(out) + len(block) < lookback:
|
|
208
|
+
block.append(lines[j].strip())
|
|
209
|
+
if lines[j].strip().startswith("/*"):
|
|
210
|
+
break
|
|
211
|
+
j -= 1
|
|
212
|
+
out.extend(block)
|
|
213
|
+
break
|
|
214
|
+
if s.startswith("//"):
|
|
215
|
+
out.append(s)
|
|
216
|
+
i -= 1
|
|
217
|
+
continue
|
|
218
|
+
if not s:
|
|
219
|
+
i -= 1
|
|
220
|
+
continue
|
|
221
|
+
break
|
|
222
|
+
return list(reversed(out))
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
# 生效条件:对 source 用 _JS_DEF.finditer 命中项生成条目(首项模块条目 name 为 os.path.basename(path) or "<module>"),跳过 kind 属于 _JS_MODULE_SCOPE_ONLY 且 indent 组非空的命中;其余命中按出现顺序取 lineno、kind、name、sig,end 为下一命中行号减一或 len(source.split("\n")) 且不小于 lineno,comments 由 _leading_js_comments(lines, lineno) 生成。
|
|
226
|
+
def _extract_weak(source, path):
|
|
227
|
+
"""正则弱提取:返回条目,`end` 为**上界**(到下一个定义之前),不保证精确。"""
|
|
228
|
+
lines = source.split("\n")
|
|
229
|
+
items = [{"path": path, "name": os.path.basename(path) or "<module>",
|
|
230
|
+
"kind": "module", "parent": "", "lineno": 1, "end": len(lines),
|
|
231
|
+
"sig": "", "doc": "", "comments": []}]
|
|
232
|
+
hits = []
|
|
233
|
+
for m in _JS_DEF.finditer(source):
|
|
234
|
+
kind, name = m.group("kind"), m.group("name")
|
|
235
|
+
if kind in _JS_MODULE_SCOPE_ONLY and m.group("indent"):
|
|
236
|
+
continue
|
|
237
|
+
lineno = source.count("\n", 0, m.start()) + 1
|
|
238
|
+
hits.append((lineno, kind, name, m.group(0).strip()))
|
|
239
|
+
for idx, (lineno, kind, name, sig) in enumerate(hits):
|
|
240
|
+
end = (hits[idx + 1][0] - 1) if idx + 1 < len(hits) else len(lines)
|
|
241
|
+
items.append({
|
|
242
|
+
"path": path, "name": name, "kind": kind, "parent": "",
|
|
243
|
+
"lineno": lineno, "end": max(lineno, end), "sig": sig[:200],
|
|
244
|
+
"doc": "", "comments": _leading_js_comments(lines, lineno)})
|
|
245
|
+
return items
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
# --------------------------------------------------------------------------
|
|
249
|
+
# 提取器注册表
|
|
250
|
+
# --------------------------------------------------------------------------
|
|
251
|
+
EXTRACTORS = {
|
|
252
|
+
".py": ("py", _extract_python, True, LANG_COMPILER),
|
|
253
|
+
".ts": ("ts", _extract_weak, False, LANG_WEAK),
|
|
254
|
+
".tsx": ("tsx", _extract_weak, False, LANG_WEAK),
|
|
255
|
+
".js": ("js", _extract_weak, False, LANG_WEAK),
|
|
256
|
+
".mjs": ("js", _extract_weak, False, LANG_WEAK),
|
|
257
|
+
".cjs": ("js", _extract_weak, False, LANG_WEAK),
|
|
258
|
+
}
|
|
259
|
+
SUFFIX = tuple(sorted(EXTRACTORS))
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
# 生效条件:lines 为行列表、lineno 与 end 为 1-based 行号时,对 "\n".join(lines[max(0, lineno-1):max(0, end)]) 的 UTF-8 字节求 sha1,返回其十六进制前 12 位;
|
|
263
|
+
def region_hash(lines, lineno, end):
|
|
264
|
+
"""被引用行的 sha1 前 12 位(变更探测用,非内容寻址)。
|
|
265
|
+
|
|
266
|
+
必须是**唯一**定义:索引侧与回读侧(`op=ref`)共用同一个函数。
|
|
267
|
+
两侧各写一份哈希算法,漂移检测就会悄悄失效(永远 hash_match=True)。
|
|
268
|
+
"""
|
|
269
|
+
seg = "\n".join(lines[max(0, lineno - 1):max(0, end)])
|
|
270
|
+
return hashlib.sha1(seg.encode("utf-8")).hexdigest()[:12]
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
# 生效条件:source 与 path 传入后,ext=suffix or os.path.splitext(path)[1].lower();当 ext 存在于模块级 EXTRACTORS 时,用对应 fn(source, path) 抽取并给每个条目补 lang/precise/basis/hash(hash 由 region_hash(lines, it["lineno"], it["end"]) 算)后返回;ext 不在 EXTRACTORS 时抛 ValueError(f"无提取器(suffix={ext or '<none>'})");
|
|
274
|
+
def extract(source, path="", suffix=None):
|
|
275
|
+
"""抽取一个文件的条目;按后缀分派提取器。语法错误抛 ValueError。
|
|
276
|
+
|
|
277
|
+
产出条目带 `lang` / `precise` / `hash`,供 `render` 与 frontmatter.code_ref 使用。
|
|
278
|
+
"""
|
|
279
|
+
ext = suffix or os.path.splitext(path)[1].lower()
|
|
280
|
+
if ext not in EXTRACTORS:
|
|
281
|
+
# 不静默降级成 Python 解析:那会把「没有提取器」伪装成「语法错误」,
|
|
282
|
+
# 让调用方误以为是源码的问题。直接报缺提取器,由 index_dir 收进 errors。
|
|
283
|
+
raise ValueError(f"无提取器(suffix={ext or '<none>'})")
|
|
284
|
+
lang, fn, precise, basis = EXTRACTORS[ext]
|
|
285
|
+
lines = source.split("\n")
|
|
286
|
+
items = fn(source, path)
|
|
287
|
+
for it in items:
|
|
288
|
+
it["lang"] = lang
|
|
289
|
+
it["precise"] = precise
|
|
290
|
+
it["basis"] = basis
|
|
291
|
+
it["hash"] = region_hash(lines, it["lineno"], it["end"])
|
|
292
|
+
return items
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
# 生效条件:item 为条目字典时,path=item.get("path") or ""、top=path.split("/")[0] or ".",按 item.get("precise", True)(缺键默认 True,键存在假值走弱提取)选择 LANG_COMPILER 或 LANG_WEAK 方法文本,返回 observation_position 用 top、time_window 用 nodefile.FULL_TIME_WINDOW_MIN 与 nodefile.FULL_TIME_WINDOW_MAX、observation_tool 用方法文本、existence_constraint 含 path 的四槽字典;
|
|
296
|
+
def condition_space(item):
|
|
297
|
+
"""条目 → 条件空间四槽(纯函数,**唯一来源**)。
|
|
298
|
+
|
|
299
|
+
为什么必须与 `render` 同源:正文的 `# 索引元条件:` 行与 frontmatter 的
|
|
300
|
+
`condition_space` 一旦各写一套,就会出现「正文有声明、条件空间是空的」
|
|
301
|
+
——`nodefile.condition_space_text(require_full=True)` 只看 frontmatter,
|
|
302
|
+
于是节点**存得进、判得了,条件空间却没声明**。改造前正是这样:正文写
|
|
303
|
+
「大域=X;检索…时」(第三种方言),frontmatter 只写 `observation_position`
|
|
304
|
+
**单槽**。单槽不是生效条件(见 nodefile.CONDITION_SLOTS_REQUIRED),
|
|
305
|
+
该四槽在正文里由「索引元条件」行承载(**不再**占用「生效条件」字段——
|
|
306
|
+
Phase 0 契约裁决,见 nodefile.INDEX_META_MARK),
|
|
307
|
+
故本函数按四槽齐备产出,供 `render` 与 `refindex.add_items` 共用。
|
|
308
|
+
|
|
309
|
+
时间槽用**全时窗哨兵**而非 `mdcg.add` 缺省补的「写入时刻锚定 1 小时窗」:
|
|
310
|
+
代码条目声明的是「源文件里存在这个符号」,其真值不随写入时刻衰减,
|
|
311
|
+
写成 1 小时观测窗是把写入副作用伪装成条件。
|
|
312
|
+
"""
|
|
313
|
+
path = item.get("path") or ""
|
|
314
|
+
top = path.split("/")[0] or "."
|
|
315
|
+
if item.get("precise", True):
|
|
316
|
+
method = f"{LANG_COMPILER}(AST 精确提取,区间精确到 end_lineno)"
|
|
317
|
+
else:
|
|
318
|
+
method = (f"{LANG_WEAK}(正则弱提取,未过编译器;"
|
|
319
|
+
f"区间为**上界**,以 op=ref 回读为准)")
|
|
320
|
+
return {
|
|
321
|
+
"observation_position": f"本地源码仓(大域={top})",
|
|
322
|
+
"time_window": [nodefile.FULL_TIME_WINDOW_MIN,
|
|
323
|
+
nodefile.FULL_TIME_WINDOW_MAX],
|
|
324
|
+
"observation_tool": method,
|
|
325
|
+
"existence_constraint": f"源文件 {path} 存在于本地仓且可读",
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
# 生效条件:item 为含 "name"、"kind"、"path"、"lineno"、"end" 键的条目字典时(缺这些必需键会 KeyError),返回由 item.get("comments") 的源码 CCG 区、合成 CCG 区、索引元信息区依次拼接的正文;parent/doc/comments/sig 按 item.get 缺键或假值回落,precise 缺键默认 True、键存在假值走弱提取,item["lineno"]/item["end"] 用于 basis 与位置行;
|
|
330
|
+
def render(item):
|
|
331
|
+
"""条目 → 正文三分区:源码 CCG 区(人工优先)→ 合成 CCG 区 → 索引元信息区。
|
|
332
|
+
|
|
333
|
+
**必须渲染成 CCG 格式**,这是本模块最容易踩的坑:
|
|
334
|
+
`judge_qualification` 第一步就查 `ccg_completeness` 的 5 要素
|
|
335
|
+
(功能名 / 子功能 / 执行 / 验证方式 / 不适用条件),缺任一即**直接判
|
|
336
|
+
BLINDSPOT**,后面的「verification_basis 缺失才 DEFER」根本走不到——
|
|
337
|
+
即便 frontmatter 已正确填了 verification_basis。改造前本函数只产
|
|
338
|
+
`# path::name` / `# sig` / `# doc:` 这类非 CCG 行,于是**所有代码节点
|
|
339
|
+
恒定 BLINDSPOT**:存得进、判不了、检索不到(与「目标节点的 CCG 渲染」
|
|
340
|
+
是同一策略,见 mdcg.py 的对应注释)。
|
|
341
|
+
|
|
342
|
+
Phase 0 修复(行序压制 + 字段语义分家;契约见
|
|
343
|
+
docs/mdcg/代码评审与条件化注释_契约_v0.1.md):
|
|
344
|
+
① **源码 CCG 区置首**——mdcos._ccg_field 取**首个**匹配,置首即
|
|
345
|
+
「人工优先」的确定性序:人工声明的生效条件不再被合成行压制;
|
|
346
|
+
② 合成区**不再产出生效条件行**——索引元条件不是功能前置条件
|
|
347
|
+
(裁定见 nodefile.INDEX_META_MARK)。故源码未声明的条目会**诚实地
|
|
348
|
+
缺该要素(BLINDSPOT)**,而不是被元条件冒充成 DEFER;
|
|
349
|
+
③ 索引元信息区改用非 CCG 字段名(索引元条件行 + 位置行),两个语义
|
|
350
|
+
不再挤同一个字段名(可机械判:nodefile.is_ccg_mark)。
|
|
351
|
+
"""
|
|
352
|
+
name = item["name"]
|
|
353
|
+
kind = item["kind"]
|
|
354
|
+
path = item["path"]
|
|
355
|
+
parent = item.get("parent") or ""
|
|
356
|
+
doc = (item.get("doc") or "").replace("\n", " ").strip()
|
|
357
|
+
comments = [c.lstrip("#").strip() for c in (item.get("comments") or [])]
|
|
358
|
+
sub = doc or (comments[0] if comments else "") or f"{kind} 定义在 {path},无注释"
|
|
359
|
+
sig = (item.get("sig") or "").strip() or "(模块级,无签名)"
|
|
360
|
+
if item.get("precise", True):
|
|
361
|
+
basis = (f"{LANG_COMPILER}(AST 已解析,区间精确:"
|
|
362
|
+
f"{path} L{item['lineno']}-L{item['end']})")
|
|
363
|
+
else:
|
|
364
|
+
basis = (f"{LANG_WEAK}(正则弱提取,未过编译器;区间为**上界**,"
|
|
365
|
+
f"以 op=ref 回读为准:{path} L{item['lineno']}-L{item['end']})")
|
|
366
|
+
# ---- 三分区组装(顺序即语义,不许随手改)------------------------------
|
|
367
|
+
# ① 源码 CCG 区:人工/源码声明逐字保留,**置首**取得「首个匹配」优先权
|
|
368
|
+
# ② 合成 CCG 区:机械补齐 5 要素,保证 ccg_completeness 不因缺行整体失效
|
|
369
|
+
# ③ 索引元信息区:非 CCG 字段名 + 位置行(与 CCG_MARKS 零重名)
|
|
370
|
+
lines = ["# " + c for c in comments]
|
|
371
|
+
lines += [
|
|
372
|
+
f"# 功能名:{name}({kind})",
|
|
373
|
+
f"# 子功能:{parent + '.' if parent else ''}{sub[:MAX_DOC]}",
|
|
374
|
+
f"# 执行:{sig}",
|
|
375
|
+
f"# 验证方式:{basis}",
|
|
376
|
+
"# 不适用条件:其它大域的**同名**符号(同名不同域时以 path 区分;"
|
|
377
|
+
f"本条目属于 {path})",
|
|
378
|
+
# 索引元条件**不占用**生效条件字段:它是「条目在何处/何时可被观测」,
|
|
379
|
+
# 不是「这段代码在何种输入下正确」。合成即冒充(nodefile.INDEX_META_MARK)。
|
|
380
|
+
f"# {nodefile.INDEX_META_MARK}:"
|
|
381
|
+
+ nodefile.condition_space_text(condition_space(item),
|
|
382
|
+
require_full=False),
|
|
383
|
+
f"# 位置:{path}:{item['lineno']}-{item['end']}"
|
|
384
|
+
f"({item.get('lang')},precise={bool(item.get('precise', True))})",
|
|
385
|
+
]
|
|
386
|
+
return "\n".join(lines)
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
# 生效条件:item 为含 "path" 与 "name" 键的字典时(缺任一键会 KeyError),返回 "code_" 加 (item["path"] + "::" + item["name"]).encode("utf-8") 的 sha1 十六进制前 12 位;
|
|
390
|
+
def node_id(item):
|
|
391
|
+
"""稳定 id:path::name 的短哈希(重复索引幂等)。"""
|
|
392
|
+
key = (item["path"] + "::" + item["name"]).encode("utf-8")
|
|
393
|
+
return "code_" + hashlib.sha1(key).hexdigest()[:12]
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
# 生效条件:skip_dirs 传入后,遍历 (skip_dirs or ()) 把每项 str(raw).strip().replace("\\","/").strip("/"),空串跳过;含 "/" 的加入 paths,不含 "/" 的加入 names;若归一化后无规则返回 (None, []),否则返回 (hit, rules),其中 hit(rel_dir, base) 在 base 命中 names 或 rel_dir 等于/前缀匹配 paths 中某条加 "/" 时为 True;
|
|
397
|
+
def skip_matcher(skip_dirs):
|
|
398
|
+
"""把调用方的 `skip_dirs` 编译成「该子目录是否排除」的判定 `hit(rel_dir, base)`。
|
|
399
|
+
|
|
400
|
+
与 `SKIP_DIRS` 在**同一处**生效,且**只增不减**:调用方只能追加排除,不能拿掉
|
|
401
|
+
`.git`/`.venv` 这类内置保护——否则一次参数写错就能把版本库元数据索引进认知图。
|
|
402
|
+
规则口径(两类可混用):
|
|
403
|
+
· 含 `/` → 按**相对 root 的路径**匹配(`docs/experiments` 只排这一处);
|
|
404
|
+
· 不含 `/` → 按**目录名**匹配(`experiments` 排任意层级的同名目录)。
|
|
405
|
+
反斜杠与首尾斜杠一律归一,避免「规则传了却不生效」这类静默失配。
|
|
406
|
+
|
|
407
|
+
返回 `(hit, rules)`;`rules` 为空时 `hit is None` → 调用方走原路径,
|
|
408
|
+
保证**默认行为与改造前逐字一致**(与 `fresh`/`on_file` 同一纪律)。
|
|
409
|
+
排除的**理由**:`.gitignore` 整目录忽略的实验产物物理仍在盘上,会把
|
|
410
|
+
`max_files` 撑爆并让「索引不全」变成常态;用「显式排除 + 回报」比「调大上限」
|
|
411
|
+
诚实。`docindex` 复用本函数(唯一实现,避免两处口径漂移)。
|
|
412
|
+
"""
|
|
413
|
+
rules, names, paths = [], set(), []
|
|
414
|
+
for raw in (skip_dirs or ()):
|
|
415
|
+
s = str(raw).strip().replace("\\", "/").strip("/")
|
|
416
|
+
if not s:
|
|
417
|
+
continue
|
|
418
|
+
rules.append(s)
|
|
419
|
+
if "/" in s:
|
|
420
|
+
paths.append(s)
|
|
421
|
+
else:
|
|
422
|
+
names.add(s)
|
|
423
|
+
if not rules:
|
|
424
|
+
return None, []
|
|
425
|
+
|
|
426
|
+
# 生效条件:在 skip_matcher 返回的闭包中,rel_dir 与 base 传入后,若 base 命中由 skip_dirs 归一化出的不含 "/" 的目录名集合 names 则返回 True;否则若 rel_dir 等于或以其某个含 "/" 的路径规则 paths 加 "/" 为前缀则返回 True;两者都不满足返回 False;
|
|
427
|
+
def hit(rel_dir, base):
|
|
428
|
+
if base in names:
|
|
429
|
+
return True
|
|
430
|
+
return any(rel_dir == p or rel_dir.startswith(p + "/") for p in paths)
|
|
431
|
+
|
|
432
|
+
return hit, rules
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
# 生效条件:当 root 为可 os.walk 的目录时返回 (items, errors, stats);patterns 为 None 时按模块常量 SUFFIX 取后缀,命中 max_files 或 max_items 上限则在 stats['truncated'] 上报截断,fresh 为 None 时逐文件读盘。
|
|
436
|
+
def index_dir(root, patterns=None, max_files=500, max_items=2000,
|
|
437
|
+
fresh=None, on_file=None, skip_dirs=None):
|
|
438
|
+
"""按大域(目录)遍历代码,产出 `(items, errors, stats)`。零 LLM。
|
|
439
|
+
|
|
440
|
+
`stats["truncated"]` 必须显式上报——截断**不再是静默的**:改造前达到上限
|
|
441
|
+
直接 `return`,调用方只看到 `indexed`/`error_count`,**索引不全却不告警**,
|
|
442
|
+
于是「不完整」被当成「完整」用。同时上报 `skipped_suffixes`:扫到但没被
|
|
443
|
+
索引的后缀要能看见,否则「不漏召回」这句话无法审计。
|
|
444
|
+
|
|
445
|
+
`fresh(rel, fp)` / `on_file(rel, fp, items)` 是给 `refindex.Ledger` 留的
|
|
446
|
+
增量钩子(默认 None → 行为与改造前逐字一致):
|
|
447
|
+
· `fresh` 返回 True → 该文件自上次索引后未变,**不读盘**直接跳过,
|
|
448
|
+
计入 `skipped_unchanged`(仍计入 `files`,故截断语义不变);
|
|
449
|
+
· `on_file` 在成功提取后回调,用于记录水位。
|
|
450
|
+
|
|
451
|
+
`skip_dirs` 是**追加**排除(见 `skip_matcher`):命中的目录整棵剪掉、不计入
|
|
452
|
+
`files`;实际排掉了哪些目录写进 `stats["skipped_dirs"]`——**排除与截断一样不许
|
|
453
|
+
静默**,否则「节点数变少」会被误读成「源文件真的少了」。
|
|
454
|
+
"""
|
|
455
|
+
pats = tuple(patterns or SUFFIX)
|
|
456
|
+
hit_skip, skip_rules = skip_matcher(skip_dirs)
|
|
457
|
+
items, errors, files = [], [], 0
|
|
458
|
+
hit_files = 0
|
|
459
|
+
seen_suffix = set()
|
|
460
|
+
stats = {"root": root, "patterns": list(pats), "files": 0, "truncated": False,
|
|
461
|
+
"truncated_reason": "", "max_files": max_files, "max_items": max_items,
|
|
462
|
+
"skipped_suffixes": [], "skipped_unchanged": 0,
|
|
463
|
+
"skip_dirs": list(skip_rules), "skipped_dirs": [],
|
|
464
|
+
# 截断**可复算**:命中(后缀匹配)文件数与本轮已产出条目数。
|
|
465
|
+
# 与 truncated_reason 一起读,调用方才能核对「差多少」而不是只看到"被截断了"。
|
|
466
|
+
"hit_files": 0, "indexed_items": 0}
|
|
467
|
+
|
|
468
|
+
# 生效条件:无参闭包,只在本次扫库作用域内可调用;把 files / hit_files / indexed_items / skipped_suffixes / skipped_dirs 一次性落进 stats,三处 return 共用同一口径;
|
|
469
|
+
def _snap():
|
|
470
|
+
"""把「截断相关」计数一次性落进 stats(三处 return 共用,避免各处口径漂移)。"""
|
|
471
|
+
stats["files"] = files
|
|
472
|
+
stats["hit_files"] = hit_files
|
|
473
|
+
stats["indexed_items"] = len(items)
|
|
474
|
+
stats["skipped_suffixes"] = sorted(
|
|
475
|
+
s for s in seen_suffix if s and s not in pats)[:12]
|
|
476
|
+
|
|
477
|
+
for dirpath, dirnames, filenames in os.walk(root):
|
|
478
|
+
rel_dir = os.path.relpath(dirpath, root).replace("\\", "/")
|
|
479
|
+
if rel_dir == ".":
|
|
480
|
+
rel_dir = ""
|
|
481
|
+
keep = []
|
|
482
|
+
for d in dirnames:
|
|
483
|
+
if d in SKIP_DIRS:
|
|
484
|
+
continue
|
|
485
|
+
child = f"{rel_dir}/{d}" if rel_dir else d
|
|
486
|
+
if hit_skip is not None and hit_skip(child, d):
|
|
487
|
+
# 就地追加、不依赖末尾汇总:截断提前 return 时也带得走(同 skipped_suffixes)。
|
|
488
|
+
stats["skipped_dirs"].append(child)
|
|
489
|
+
continue
|
|
490
|
+
keep.append(d)
|
|
491
|
+
# 目录序**必须确定性**:os.walk 给出的 dirnames 顺序由文件系统决定,
|
|
492
|
+
# 同一次输入两次运行可能不同 → 索引结果不可复算。排序后 walk 顺序唯一。
|
|
493
|
+
dirnames[:] = sorted(keep)
|
|
494
|
+
for fn in sorted(filenames):
|
|
495
|
+
ext = os.path.splitext(fn)[1].lower()
|
|
496
|
+
seen_suffix.add(ext)
|
|
497
|
+
if not fn.lower().endswith(pats):
|
|
498
|
+
continue
|
|
499
|
+
hit_files += 1
|
|
500
|
+
if files >= max_files or len(items) >= max_items:
|
|
501
|
+
stats["truncated"] = True
|
|
502
|
+
stats["truncated_reason"] = (
|
|
503
|
+
f"files={files}>=max_files={max_files}"
|
|
504
|
+
if files >= max_files else
|
|
505
|
+
f"items={len(items)}>=max_items={max_items}")
|
|
506
|
+
_snap()
|
|
507
|
+
return items, errors, stats
|
|
508
|
+
files += 1
|
|
509
|
+
fp = os.path.join(dirpath, fn)
|
|
510
|
+
rel = os.path.relpath(fp, root).replace("\\", "/")
|
|
511
|
+
if fresh is not None and fresh(rel, fp):
|
|
512
|
+
stats["skipped_unchanged"] += 1
|
|
513
|
+
continue
|
|
514
|
+
try:
|
|
515
|
+
with open(fp, encoding="utf-8") as f:
|
|
516
|
+
src = f.read()
|
|
517
|
+
got = extract(src, rel)
|
|
518
|
+
items.extend(got)
|
|
519
|
+
if on_file is not None:
|
|
520
|
+
on_file(rel, fp, got)
|
|
521
|
+
if len(items) >= max_items:
|
|
522
|
+
# 单文件就可能越限:越限即记截断并立刻停,不装看不见、
|
|
523
|
+
# 也不继续往下扫(继续扫只会让「截断」这件事更不明显)。
|
|
524
|
+
stats["truncated"] = True
|
|
525
|
+
stats["truncated_reason"] = (
|
|
526
|
+
f"items={len(items)}>=max_items={max_items}")
|
|
527
|
+
_snap()
|
|
528
|
+
return items, errors, stats
|
|
529
|
+
except (OSError, UnicodeDecodeError, ValueError) as exc:
|
|
530
|
+
errors.append(f"{rel}: {exc}")
|
|
531
|
+
_snap()
|
|
532
532
|
return items, errors, stats
|