@furongjun1999/dsh-memory 0.4.11 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -16
- package/codebuddy/CODEBUDDY.md +11 -3
- package/codebuddy/README.md +92 -90
- package/codebuddy/mcp.json +27 -27
- package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
- package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
- package/docs/README.md +142 -111
- package/docs/discipline/harnesses.yaml +244 -226
- package/docs/discipline/templates/full.md.tmpl +61 -61
- package/docs/discipline/templates/rules.mdc.tmpl +68 -0
- package/docs/discipline/templates/skill.md.tmpl +23 -23
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
- package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
- package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
- package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
- package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
- package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
- package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
- package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
- package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
- package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
- package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
- package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
- package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
- package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
- package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
- package/docs/mdcg/lingshu_tutorial.html +14449 -14449
- package/docs/mdcg/release_v0.4.11.md +49 -0
- package/docs/mdcg/release_v0.4.5.md +55 -55
- package/docs/mdcg/tool_table_v0.3.0.md +117 -117
- package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
- package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
- package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
- package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
- package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
- package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
- package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
- package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
- package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
- package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
- package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
- package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
- package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
- package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
- package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
- package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
- package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
- package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
- package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
- package/dsh/README.md +82 -82
- package/dsh/cordis.yml.example +139 -139
- package/dsh/update-lingshu.bat +11 -11
- package/lib/hooks.js +36 -2
- package/lib/lib/roleplay_web.js +427 -427
- package/md_cg/__init__.py +7 -7
- package/md_cg/audit.py +368 -368
- package/md_cg/autonomy.py +287 -287
- package/md_cg/backfill.py +1327 -1327
- package/md_cg/backfill_bigdomain.py +34 -34
- package/md_cg/bench6_arms.py +410 -410
- package/md_cg/bench6_common.py +230 -230
- package/md_cg/bench6_competitors.py +212 -212
- package/md_cg/bench_axis_domain.py +257 -257
- package/md_cg/bench_blind_comp.py +308 -308
- package/md_cg/bench_en_atoms_public.py +230 -230
- package/md_cg/bench_governance.py +348 -348
- package/md_cg/bench_lme_zh.py +410 -410
- package/md_cg/bench_locomo.py +121 -121
- package/md_cg/bench_locomo_zh.py +450 -450
- package/md_cg/bench_locomo_zh_public.py +147 -147
- package/md_cg/bench_longmem.py +112 -112
- package/md_cg/bench_membench.py +632 -632
- package/md_cg/bench_p0.py +149 -149
- package/md_cg/bench_progressive.py +287 -287
- package/md_cg/bench_role_views.py +238 -238
- package/md_cg/bench_task_ab.py +243 -243
- package/md_cg/bench_task_ab_llm.py +408 -408
- package/md_cg/bench_unified_en.py +204 -204
- package/md_cg/bench_zh_mad.py +601 -601
- package/md_cg/blindspot_tickets.py +123 -123
- package/md_cg/branches.py +285 -285
- package/md_cg/build_postings.py +73 -73
- package/md_cg/ccgc.py +1005 -948
- package/md_cg/census.py +132 -132
- package/md_cg/chain.py +300 -300
- package/md_cg/codeindex.py +531 -531
- package/md_cg/coldverify.py +292 -292
- package/md_cg/comment_gate.py +337 -337
- package/md_cg/cond_compose.py +190 -190
- package/md_cg/cond_facts.py +154 -154
- package/md_cg/cond_template.json +106 -106
- package/md_cg/condition_anchor.py +142 -142
- package/md_cg/conformance.py +726 -726
- package/md_cg/consistency.py +717 -717
- package/md_cg/consolidate.py +1536 -1439
- package/md_cg/corpus.py +110 -110
- package/md_cg/crosscheck.py +1097 -1097
- package/md_cg/crypto.py +437 -437
- package/md_cg/d_meta.py +310 -310
- package/md_cg/datapath.py +334 -334
- package/md_cg/docindex.py +473 -473
- package/md_cg/eval_common.py +575 -575
- package/md_cg/evidence.py +580 -580
- package/md_cg/evolution.py +477 -477
- package/md_cg/export.py +220 -220
- package/md_cg/forgetting.py +581 -581
- package/md_cg/fsutil.py +329 -329
- package/md_cg/hotcache.py +238 -214
- package/md_cg/hyperedge.py +251 -251
- package/md_cg/identity.py +390 -390
- package/md_cg/insight.py +500 -500
- package/md_cg/interop.py +199 -0
- package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
- package/md_cg/lexicon/build_standard_en.py +171 -171
- package/md_cg/lexicon/expand_en_zh.py +211 -211
- package/md_cg/lifecycle.py +272 -272
- package/md_cg/linkref.py +280 -280
- package/md_cg/links.py +622 -622
- package/md_cg/mcp_server.py +129 -32
- package/md_cg/md_whitebox.py +345 -345
- package/md_cg/mdcg.py +176 -117
- package/md_cg/mdcos.py +79 -13
- package/md_cg/metacognition.py +591 -591
- package/md_cg/migrate.py +119 -119
- package/md_cg/migrate_aeis.py +221 -221
- package/md_cg/migrate_roleplay.py +293 -293
- package/md_cg/migrate_wisdom_graph.py +360 -360
- package/md_cg/mreview/__init__.py +25 -25
- package/md_cg/mreview/__main__.py +110 -110
- package/md_cg/mreview/bundle.py +178 -178
- package/md_cg/mreview/candidates.py +262 -262
- package/md_cg/mreview/govern.py +693 -693
- package/md_cg/mreview/locate.py +939 -939
- package/md_cg/mreview/pipeline.py +728 -728
- package/md_cg/mreview/rules/duplication.json +21 -21
- package/md_cg/mreview/rules/field_coverage.json +54 -54
- package/md_cg/mreview/rules/source_license.json +21 -21
- package/md_cg/mreview/rules/template_flow.json +21 -21
- package/md_cg/mreview/ruleset.py +252 -252
- package/md_cg/nodefile.py +575 -575
- package/md_cg/pooling.py +484 -472
- package/md_cg/postings.py +298 -298
- package/md_cg/predict.py +1100 -1100
- package/md_cg/progressive.py +123 -123
- package/md_cg/protect.py +272 -272
- package/md_cg/protocol/md_cg_gate.proto +33 -33
- package/md_cg/protocol.py +372 -372
- package/md_cg/provenance.py +582 -582
- package/md_cg/reach.py +453 -453
- package/md_cg/readcache.py +85 -0
- package/md_cg/refindex.py +833 -833
- package/md_cg/refine.py +604 -604
- package/md_cg/roleviews.py +89 -89
- package/md_cg/routing.py +365 -365
- package/md_cg/scrub.py +852 -852
- package/md_cg/security.py +274 -274
- package/md_cg/self_state.py +1029 -1029
- package/md_cg/selfreport.py +151 -151
- package/md_cg/semantic/__init__.py +10 -10
- package/md_cg/semantic/canonical.py +122 -122
- package/md_cg/semantic/en_normalizer.py +364 -364
- package/md_cg/semantic/en_zh_map.json +28694 -0
- package/md_cg/semantic/export_en_zh_map.py +64 -0
- package/md_cg/semantic/unify.py +45 -0
- package/md_cg/semantic/zh_en_atoms.py +139 -139
- package/md_cg/signer.py +562 -562
- package/md_cg/sources.py +815 -582
- package/md_cg/statushdr.py +179 -179
- package/md_cg/stg.py +48 -37
- package/md_cg/subgraph.py +729 -729
- package/md_cg/sustain.py +1138 -1138
- package/md_cg/tasks.py +470 -470
- package/md_cg/test_action_derive.py +203 -203
- package/md_cg/test_audit_rotate.py +270 -270
- package/md_cg/test_autonomy.py +143 -143
- package/md_cg/test_bench_governance.py +102 -102
- package/md_cg/test_blindspot_tickets.py +166 -166
- package/md_cg/test_branches.py +249 -249
- package/md_cg/test_ccg_perturb.py +184 -184
- package/md_cg/test_ccgc.py +433 -433
- package/md_cg/test_census_prune.py +81 -81
- package/md_cg/test_cond_compose_anchors.py +76 -76
- package/md_cg/test_cond_match.py +165 -165
- package/md_cg/test_condition_anchor.py +81 -81
- package/md_cg/test_d_meta.py +412 -412
- package/md_cg/test_datapath_root.py +199 -199
- package/md_cg/test_en_pipeline.py +166 -166
- package/md_cg/test_gain_gate.py +212 -212
- package/md_cg/test_health_scale.py +173 -173
- package/md_cg/test_hive_ingest.py +285 -0
- package/md_cg/test_hot_cold.py +215 -215
- package/md_cg/test_hyperedge.py +245 -245
- package/md_cg/test_i26_empty_first_write.py +116 -0
- package/md_cg/test_i27_e041_identity.py +128 -0
- package/md_cg/test_i28_hotcache_prodpath.py +122 -0
- package/md_cg/test_identity_attribution.py +147 -147
- package/md_cg/test_index_durability.py +224 -224
- package/md_cg/test_interop.py +93 -0
- package/md_cg/test_lifecycle.py +309 -309
- package/md_cg/test_linkref.py +306 -306
- package/md_cg/test_lock.py +43 -43
- package/md_cg/test_md_access_parity.py +255 -255
- package/md_cg/test_md_writepath.py +345 -345
- package/md_cg/test_mdstore_search_parity.py +160 -0
- package/md_cg/test_mr_m2.py +587 -587
- package/md_cg/test_mr_m3.py +710 -710
- package/md_cg/test_mr_m4.py +485 -485
- package/md_cg/test_p0.py +250 -250
- package/md_cg/test_p1.py +316 -316
- package/md_cg/test_p10_identity.py +173 -173
- package/md_cg/test_p11_consistency.py +233 -233
- package/md_cg/test_p12_metacognition.py +212 -212
- package/md_cg/test_p13_encryption.py +241 -241
- package/md_cg/test_p14_sustain.py +249 -249
- package/md_cg/test_p15_scrub.py +280 -280
- package/md_cg/test_p16_self_state.py +301 -301
- package/md_cg/test_p17_predict.py +354 -354
- package/md_cg/test_p18_whitebox.py +171 -171
- package/md_cg/test_p19_migrate_roleplay.py +149 -149
- package/md_cg/test_p20_evolution.py +315 -315
- package/md_cg/test_p21_tokens.py +293 -270
- package/md_cg/test_p22_theory.py +175 -175
- package/md_cg/test_p23_links.py +311 -311
- package/md_cg/test_p24_evidence.py +227 -227
- package/md_cg/test_p25_weights.py +156 -156
- package/md_cg/test_p26_refindex.py +416 -416
- package/md_cg/test_p27_docindex.py +765 -765
- package/md_cg/test_p28_refcheck.py +305 -305
- package/md_cg/test_p29_session_ingest_export.py +354 -333
- package/md_cg/test_p3.py +11 -2
- package/md_cg/test_p30_maintain.py +330 -330
- package/md_cg/test_p31_insight.py +534 -534
- package/md_cg/test_p32_backfill.py +298 -298
- package/md_cg/test_p33_ccg_wiring.py +293 -293
- package/md_cg/test_p34_crosscheck.py +331 -331
- package/md_cg/test_p35_conditioned_claim.py +252 -252
- package/md_cg/test_p36_kp_align.py +230 -230
- package/md_cg/test_p37_condition_space.py +248 -248
- package/md_cg/test_p38_concurrent_flush.py +102 -0
- package/md_cg/test_p38_contextualize.py +273 -273
- package/md_cg/test_p39_verify_flow.py +113 -0
- package/md_cg/test_p39_vision_evidence.py +369 -369
- package/md_cg/test_p40_refine_worklist.py +241 -241
- package/md_cg/test_p41_evolve_patrol.py +224 -224
- package/md_cg/test_p42_provenance.py +269 -269
- package/md_cg/test_p43_pooling.py +412 -398
- package/md_cg/test_p44_md_whitebox.py +231 -231
- package/md_cg/test_p45_session_identity.py +219 -219
- package/md_cg/test_p46_unit_scope.py +272 -272
- package/md_cg/test_p47_session_view.py +281 -0
- package/md_cg/test_p4_fuzzy.py +223 -223
- package/md_cg/test_p5_semantic.py +226 -226
- package/md_cg/test_p6_consolidate.py +440 -387
- package/md_cg/test_p7_goals_recent.py +202 -202
- package/md_cg/test_p8_subgraph_chain.py +200 -200
- package/md_cg/test_p9_forget_protect.py +231 -231
- package/md_cg/test_predict_beta.py +135 -135
- package/md_cg/test_preflight_failclosed.py +100 -100
- package/md_cg/test_progressive.py +146 -146
- package/md_cg/test_protocol.py +243 -243
- package/md_cg/test_reach.py +378 -378
- package/md_cg/test_reach_keys.py +201 -201
- package/md_cg/test_read_clip.py +141 -141
- package/md_cg/test_readcache_prodpath.py +155 -0
- package/md_cg/test_retr_gates_prodpath.py +140 -0
- package/md_cg/test_retr_s1.py +340 -340
- package/md_cg/test_retr_s1b.py +209 -209
- package/md_cg/test_retr_s3.py +194 -194
- package/md_cg/test_retr_s4.py +163 -163
- package/md_cg/test_retr_s5.py +200 -200
- package/md_cg/test_retr_s6.py +157 -157
- package/md_cg/test_retr_s7.py +384 -384
- package/md_cg/test_retr_s8_time.py +369 -316
- package/md_cg/test_retr_s9_edges.py +286 -286
- package/md_cg/test_retr_s9_entity_ctx.py +175 -175
- package/md_cg/test_review_conformance.py +367 -367
- package/md_cg/test_role_views.py +354 -354
- package/md_cg/test_sem_noise.py +242 -242
- package/md_cg/test_semantic_canonical.py +241 -241
- package/md_cg/test_subproc_encoding.py +192 -192
- package/md_cg/test_sustain_mutual.py +153 -153
- package/md_cg/test_tasks.py +409 -409
- package/md_cg/test_tool_face.py +189 -189
- package/md_cg/test_transfer.py +180 -180
- package/md_cg/test_trust.py +361 -361
- package/md_cg/test_twophase.py +286 -286
- package/md_cg/test_v14_fixes.py +397 -397
- package/md_cg/test_validity_filter.py +280 -280
- package/md_cg/test_verify_answer.py +138 -138
- package/md_cg/test_wisdom_md_store.py +292 -292
- package/md_cg/test_writelimit.py +197 -197
- package/md_cg/test_writepipe.py +214 -214
- package/md_cg/theory.py +273 -273
- package/md_cg/tokens.py +677 -663
- package/md_cg/tool_face.py +260 -260
- package/md_cg/trust.py +986 -950
- package/md_cg/twophase.py +231 -231
- package/md_cg/units.py +667 -667
- package/md_cg/vision_evidence.py +666 -666
- package/md_cg/weights.py +624 -624
- package/md_cg/whitebox.py +527 -527
- package/md_cg/whitebox_kb/__init__.py +37 -37
- package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
- package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
- package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
- package/md_cg/whitebox_kb/engine.py +310 -310
- package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
- package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
- package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
- package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
- package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
- package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
- package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
- package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
- package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
- package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
- package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
- package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
- package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
- package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
- package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
- package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
- package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
- package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
- package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
- package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
- package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
- package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
- package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
- package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
- package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
- package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
- package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
- package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
- package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
- package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
- package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
- package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
- package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
- package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
- package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
- package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
- package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
- package/md_cg/writelimit.py +356 -356
- package/md_cg/writepipe.py +550 -542
- package/package.json +97 -96
- package/skills/plugin.json +54 -54
- package/skills/skills/designer-perspective/SKILL.md +158 -158
- package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
- package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
- package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
- package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
- package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
- package/skills/skills/designer-perspective/scripts/designer.py +545 -545
- package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
- package/skills/skills/designer-perspective/tests/selftest.py +61 -61
- package/skills/skills/lingshu-browser/SKILL.md +60 -60
- package/skills/skills/lingshu-compiler/SKILL.md +56 -56
- package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
- package/skills/skills/lingshu-graph/SKILL.md +63 -63
- package/skills/skills/lingshu-net/SKILL.md +48 -48
- package/skills/skills/lingshu-os/SKILL.md +64 -64
- package/skills/skills/lingshu-pylang/SKILL.md +71 -71
- package/src/bridge.ts +401 -401
- package/src/hooks.ts +38 -2
- package/src/lib/datapath.ts +326 -326
- package/src/lib/mdcg_client.ts +413 -413
- package/src/lib/mutual.ts +428 -428
- package/src/lib/prompt_safety.ts +62 -62
- package/src/lib/python_path.ts +71 -71
- package/src/lib/roleplay_web.ts +932 -932
- package/src/lib/token_store.ts +192 -192
- package/src/tools.ts +212 -212
- package/zcode/AGENTS.md +11 -3
- package/zcode/README.md +41 -41
- /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
package/md_cg/refindex.py
CHANGED
|
@@ -1,834 +1,834 @@
|
|
|
1
|
-
# -*- coding: utf-8 -*-
|
|
2
|
-
"""md_cg · 统一 ref 协议 + 索引水位(增量)+ 漂移/悬空巡检
|
|
3
|
-
|
|
4
|
-
对照 `docs/mdcg/认知图_索引与工程规范化_计划_v0.1.md` 的 R3(修 D + 修 F):
|
|
5
|
-
|
|
6
|
-
**D 漂移 / 悬空检测(本模块 `check_refs`)**
|
|
7
|
-
扫描带 `code_ref` / `doc_ref` 的节点,回两类问题:
|
|
8
|
-
· `stale` —— 源文件被改(区间哈希不再匹配)
|
|
9
|
-
· `dangling` —— 源文件被删(索引指向不存在的文件)
|
|
10
|
-
巡检**只读**、**不抛**、**不改源文件**;修复动作是「重跑 index_code / index_doc」,
|
|
11
|
-
因为索引是派生物(对齐 sustain.heal 的既有边界)。
|
|
12
|
-
|
|
13
|
-
**F 全量重扫 + 静默截断(本模块 `Ledger` + `index_dir`)**
|
|
14
|
-
`<root>/_refindex.json` 是 ref 索引水位(抄 `sources.Ingestor` 的 `_sources.json` 范式),
|
|
15
|
-
以**源文件绝对路径**为键,记每个源文件的 (size, mtime) 与节点区间 + 它所属的**源大域
|
|
16
|
-
root**;`incremental=True` 时未变文件**不再读盘重切**,直接跳过(`skipped_unchanged`)
|
|
17
|
-
——这就是「不全量重扫」。
|
|
18
|
-
键用绝对路径、且逐文件记 root,是因为一份认知图可以索引多个大域:只按 rel 记会在同名
|
|
19
|
-
文件上互相覆盖,巡检时若拿认知图根去拼路径则会把一切都误判成 dangling。
|
|
20
|
-
截断(`max_files` / `max_items`)由 codeindex / docindex 显式上报,本模块把
|
|
21
|
-
「最近一次索引被截断」写进水位,交给 `sustain.diagnose` 巡检看见(不再静默)。
|
|
22
|
-
|
|
23
|
-
**为什么回读要收进本模块**
|
|
24
|
-
`op=ref` 的回读与 `check_refs` 的判定**必须共用同一实现**,否则会出现
|
|
25
|
-
「回读说没漂、巡检说有漂」。与 `region_hash` 的教训同源:区间哈希只允许一份实现,
|
|
26
|
-
这里连「怎么判定 ok / stale / dangling」也只允许一份。
|
|
27
|
-
|
|
28
|
-
零第三方依赖。
|
|
29
|
-
"""
|
|
30
|
-
from __future__ import annotations
|
|
31
|
-
|
|
32
|
-
import json
|
|
33
|
-
import os
|
|
34
|
-
import time
|
|
35
|
-
|
|
36
|
-
from .fsutil import atomic_write
|
|
37
|
-
|
|
38
|
-
SCHEMA = 2 # v2:水位以「源文件绝对路径」为键(v1 按 rel 会跨大域撞名)
|
|
39
|
-
LEDGER_FILE = "_refindex.json"
|
|
40
|
-
REF_KEYS = ("code_ref", "doc_ref")
|
|
41
|
-
MAX_CHECK = 2000 # 巡检节点上限(超出报 truncated,不静默截断)
|
|
42
|
-
STATUSES = ("ok", "stale", "dangling", "unresolved", "error")
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
# 生效条件:给定 fp 时返回 os.path.abspath(fp or '')(fp 为空/None 则返回当前目录的绝对路径),作为水位键以绝对路径保证不同 root 下同名文件不互相覆盖。
|
|
46
|
-
def _src_key(fp: str) -> str:
|
|
47
|
-
"""水位的键 = 源文件绝对路径。
|
|
48
|
-
|
|
49
|
-
不能用 rel:一份认知图可以索引多个大域(不同 root),只按 rel 记会在
|
|
50
|
-
`alpha.py` 这种同名文件上互相覆盖——水位被静默丢掉,巡检就漏报。
|
|
51
|
-
"""
|
|
52
|
-
return os.path.abspath(fp or "")
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
# 生效条件:无 required 形参,任何调用都返回 round(time.time(), 1),把时间戳压到 1 位小数以稳定 `_refindex.json` 字节数。
|
|
56
|
-
def _now() -> float:
|
|
57
|
-
"""时间戳压到 1 位小数:让 `_refindex.json` 字节数稳定(重跑不涨),
|
|
58
|
-
同时保留足够的「多久以前」信息(float 的最短 repr 保证小数位固定为 1)。"""
|
|
59
|
-
return round(time.time(), 1)
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
# --------------------------------------------------------------------------
|
|
63
|
-
# 提取器注册表(统一调度:调用方只说 kind,不说「用哪个模块」)
|
|
64
|
-
# --------------------------------------------------------------------------
|
|
65
|
-
|
|
66
|
-
# 生效条件:kind == 'code_ref' 返回 codeindex、kind == 'doc_ref' 返回 docindex,其他 kind 抛 ValueError(提示支持 REF_KEYS)。
|
|
67
|
-
def _mod(kind: str):
|
|
68
|
-
from . import codeindex, docindex
|
|
69
|
-
if kind == "code_ref":
|
|
70
|
-
return codeindex
|
|
71
|
-
if kind == "doc_ref":
|
|
72
|
-
return docindex
|
|
73
|
-
raise ValueError(f"未知 ref kind:{kind!r}(支持 {REF_KEYS})")
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
# 生效条件:无 required 形参,调用即返回 {'code_ref': {'suffixes': tuple(codeindex.SUFFIX)}, 'doc_ref': {'suffixes': tuple(docindex.SUFFIX)}}。
|
|
77
|
-
def registry() -> dict:
|
|
78
|
-
"""后缀 → kind 的注册表(code / doc 各一份提取器)。"""
|
|
79
|
-
from . import codeindex, docindex
|
|
80
|
-
return {
|
|
81
|
-
"code_ref": {"suffixes": tuple(codeindex.SUFFIX)},
|
|
82
|
-
"doc_ref": {"suffixes": tuple(docindex.SUFFIX)},
|
|
83
|
-
}
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
# 生效条件:path 的小写后缀在 codeindex.EXTRACTORS 中返回 'code_ref',在 docindex.SUFFIX 中返回 'doc_ref',无后缀或均不匹配返回 ''。
|
|
87
|
-
def kind_of_path(path: str) -> str:
|
|
88
|
-
"""按后缀判 kind;无提取器返回 ''(由调用方决定是报错还是跳过)。"""
|
|
89
|
-
from . import codeindex, docindex
|
|
90
|
-
ext = os.path.splitext(path or "")[1].lower()
|
|
91
|
-
if not ext:
|
|
92
|
-
return ""
|
|
93
|
-
if ext in codeindex.EXTRACTORS:
|
|
94
|
-
return "code_ref"
|
|
95
|
-
if ext in docindex.SUFFIX:
|
|
96
|
-
return "doc_ref"
|
|
97
|
-
return ""
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
# 生效条件:source 为待提取文本,kind 非空或 path 后缀能推出 kind 时返回 _mod(k).extract(source, path),推不出 kind 时抛 ValueError。
|
|
101
|
-
def extract(source: str, path: str = "", kind: str = ""):
|
|
102
|
-
"""统一提取入口:按 kind(或从 path 推断)分发到对应 extractor。"""
|
|
103
|
-
k = kind or kind_of_path(path)
|
|
104
|
-
if not k:
|
|
105
|
-
ext = os.path.splitext(path or "")[1] or "<none>"
|
|
106
|
-
raise ValueError(f"无索引提取器(suffix={ext})")
|
|
107
|
-
return _mod(k).extract(source, path)
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
# 生效条件:kind 为 'code_ref'/'doc_ref' 时返回 _mod(kind).node_id(item),其他 kind 由 _mod 抛 ValueError。
|
|
111
|
-
def node_id_of(item: dict, kind: str) -> str:
|
|
112
|
-
return _mod(kind).node_id(item)
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
# 生效条件:kind 为 'code_ref'/'doc_ref' 时返回 _mod(kind).render(item),其他 kind 由 _mod 抛 ValueError。
|
|
116
|
-
def render_of(item: dict, kind: str) -> str:
|
|
117
|
-
return _mod(kind).render(item)
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
# 生效条件:node 的 frontmatter 中 REF_KEYS 命中且值为非空 dict 时返回 {'ref': ref, 'ref_kind': kind},否则返回 {'ref': None, 'ref_kind': ''}。
|
|
121
|
-
def ref_fields(node) -> dict:
|
|
122
|
-
"""节点 → 检索结果要带的两字段(读侧只加字段,不改召回逻辑)。"""
|
|
123
|
-
kind, ref = ref_of(node)
|
|
124
|
-
return {"ref": ref, "ref_kind": kind} if ref else {"ref": None, "ref_kind": ""}
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
# 生效条件:node 的 frontmatter 按 REF_KEYS 顺序取到第一个非空 dict 时返回 (k, r),否则返回 ('', None)。
|
|
128
|
-
def ref_of(node) -> tuple:
|
|
129
|
-
"""从节点 frontmatter 取 ref:返回 (kind, ref) 或 ('', None)。"""
|
|
130
|
-
fm = (node or {}).get("frontmatter") or {}
|
|
131
|
-
for k in REF_KEYS:
|
|
132
|
-
r = fm.get(k)
|
|
133
|
-
if isinstance(r, dict) and r:
|
|
134
|
-
return k, r
|
|
135
|
-
return "", None
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
# --------------------------------------------------------------------------
|
|
139
|
-
# 索引水位(_refindex.json):增量 + 截断留痕
|
|
140
|
-
# --------------------------------------------------------------------------
|
|
141
|
-
|
|
142
|
-
# 生效条件:以 root 为必填实参构造,实例化即置 self.root=root、self.path=os.path.join(root, LEDGER_FILE)、self._d=None;
|
|
143
|
-
class Ledger:
|
|
144
|
-
"""`<root>/_refindex.json`:每个源文件的 (size, mtime) 水位 + 节点区间。"""
|
|
145
|
-
|
|
146
|
-
# 生效条件:当传入 root 时,self.root 取该 root,self.path 为 os.path.join(root, LEDGER_FILE),self._d 置为 None;
|
|
147
|
-
def __init__(self, root: str):
|
|
148
|
-
self.root = root
|
|
149
|
-
self.path = os.path.join(root, LEDGER_FILE)
|
|
150
|
-
self._d = None
|
|
151
|
-
|
|
152
|
-
# 生效条件:当 self._d is not None 时直接返回 self._d;否则读取 self.path 的 JSON,仅当 obj 是 dict 且 obj.get("schema") == SCHEMA 且 obj.get("files") 是 dict 时用 obj,否则(含 OSError/ValueError、结构不符)回落为 {"schema": SCHEMA, "updated_at": 0.0, "files": {}} 并缓存返回;
|
|
153
|
-
def load(self) -> dict:
|
|
154
|
-
if self._d is not None:
|
|
155
|
-
return self._d
|
|
156
|
-
d = None
|
|
157
|
-
try:
|
|
158
|
-
with open(self.path, "r", encoding="utf-8") as f:
|
|
159
|
-
obj = json.load(f)
|
|
160
|
-
if isinstance(obj, dict) and obj.get("schema") == SCHEMA \
|
|
161
|
-
and isinstance(obj.get("files"), dict):
|
|
162
|
-
d = obj
|
|
163
|
-
except (OSError, ValueError):
|
|
164
|
-
d = None
|
|
165
|
-
self._d = d or {"schema": SCHEMA, "updated_at": 0.0, "files": {}}
|
|
166
|
-
return self._d
|
|
167
|
-
|
|
168
|
-
# 生效条件:传入 rel、fp 时,若 self.load()["files"].get(_src_key(fp)) 缺失或为假值、或 os.stat(fp) 抛 OSError、或条目 e.get("size") != st.st_size,则返回 False;否则返回 abs(float(e.get("mtime") or 0.0) - st.st_mtime) < 1e-6(mtime 缺失或假值时按 0.0);
|
|
169
|
-
def is_fresh(self, rel: str, fp: str) -> bool:
|
|
170
|
-
"""源文件自上次索引后未变(size + mtime 双等)→ 可跳过不重切。"""
|
|
171
|
-
e = self.load()["files"].get(_src_key(fp))
|
|
172
|
-
if not e:
|
|
173
|
-
return False
|
|
174
|
-
try:
|
|
175
|
-
st = os.stat(fp)
|
|
176
|
-
except OSError:
|
|
177
|
-
return False
|
|
178
|
-
if e.get("size") != st.st_size:
|
|
179
|
-
return False
|
|
180
|
-
return abs(float(e.get("mtime") or 0.0) - st.st_mtime) < 1e-6
|
|
181
|
-
|
|
182
|
-
# 生效条件:当 rel、fp、kind、nodes 传入且 os.stat(fp) 成功时,向 self.load()["files"][_src_key(fp)] 写条目,其中 root 为 root if root else os.path.dirname(key)、path 为 rel、kind 为 kind、size/mtime 取 st、nodes 为每项 n.get("id")/n.get("lineno")/n.get("end")/n.get("hash");os.stat(fp) 抛 OSError 时不写入;
|
|
183
|
-
def record(self, rel: str, fp: str, kind: str, nodes,
|
|
184
|
-
root: str = None) -> None:
|
|
185
|
-
"""记一个源文件的水位(节点区间用于判 stale)。
|
|
186
|
-
|
|
187
|
-
`root` 是**源**大域的根(≠ 认知图根):巡检要拿它拼 `root/rel` 才能
|
|
188
|
-
找到源文件,缺了它就会把「源在别处」误判成 dangling。
|
|
189
|
-
"""
|
|
190
|
-
key = _src_key(fp)
|
|
191
|
-
try:
|
|
192
|
-
st = os.stat(fp)
|
|
193
|
-
except OSError:
|
|
194
|
-
return
|
|
195
|
-
self.load()["files"][key] = {
|
|
196
|
-
"root": root if root else os.path.dirname(key),
|
|
197
|
-
"path": rel,
|
|
198
|
-
"size": st.st_size,
|
|
199
|
-
"mtime": st.st_mtime,
|
|
200
|
-
"kind": kind,
|
|
201
|
-
"nodes": [
|
|
202
|
-
{"id": n.get("id"), "lineno": n.get("lineno"),
|
|
203
|
-
"end": n.get("end"), "hash": n.get("hash")}
|
|
204
|
-
for n in nodes
|
|
205
|
-
],
|
|
206
|
-
}
|
|
207
|
-
|
|
208
|
-
# 生效条件:当传入 fp 时,self.load()["files"].pop(_src_key(fp), None),即删除对应键(不存在也静默);
|
|
209
|
-
def drop(self, fp: str) -> None:
|
|
210
|
-
self.load()["files"].pop(_src_key(fp), None)
|
|
211
|
-
|
|
212
|
-
# 生效条件:当传入 root、kind、seen 时,对 self.load()["files"] 中满足 os.path.abspath(e.get("root") or "") == os.path.abspath(root) 且 e.get("kind") == kind 且键 k 不在 seen 的条目删除,返回删除数量;
|
|
213
|
-
def reconcile(self, root: str, kind: str, seen) -> int:
|
|
214
|
-
"""一次**完整**索引后对账:本 (root, kind) 下没被扫到的旧条目剪掉。
|
|
215
|
-
|
|
216
|
-
否则「源文件被删 → 索引悬空 → heal 重建」之后条目还在,巡检就永远报
|
|
217
|
-
dangling,heal 是治不好的。`seen` 是本次真正走过(提取成功或判定未变)
|
|
218
|
-
的源文件键集合。**截断的索引不能对账**——没扫完不等于剩下的都消失了。
|
|
219
|
-
"""
|
|
220
|
-
r = os.path.abspath(root)
|
|
221
|
-
files = self.load()["files"]
|
|
222
|
-
dead = [k for k, e in files.items()
|
|
223
|
-
if os.path.abspath(e.get("root") or "") == r
|
|
224
|
-
and e.get("kind") == kind and k not in seen]
|
|
225
|
-
for k in dead:
|
|
226
|
-
files.pop(k, None)
|
|
227
|
-
return len(dead)
|
|
228
|
-
|
|
229
|
-
# 生效条件:对 load()["files"] 中「条目 root(为假值时用 os.path.dirname(键) 兜底)不是目录」的条目逐一 pop 并返回删除条数,无匹配时返回 0。
|
|
230
|
-
def prune(self) -> int:
|
|
231
|
-
"""剪掉「源大域已不存在」的条目(整个目录被搬走/删除)。
|
|
232
|
-
|
|
233
|
-
这类条目已不可能再被任何大域索引到,留着只会在巡检里报永不消失的
|
|
234
|
-
dangling;而节点自带的 ref 仍会兜底探测,所以剪掉不会漏报真实悬空。
|
|
235
|
-
"""
|
|
236
|
-
files = self.load()["files"]
|
|
237
|
-
dead = [k for k, e in files.items()
|
|
238
|
-
if not os.path.isdir(e.get("root") or os.path.dirname(k))]
|
|
239
|
-
for k in dead:
|
|
240
|
-
files.pop(k, None)
|
|
241
|
-
return len(dead)
|
|
242
|
-
|
|
243
|
-
# 生效条件:当 kind、root、files、indexed、truncated 传入时,self.load()["last_index"] 被设为含 ts=_now()、kind、root、files、indexed、truncated=bool(truncated)、truncated_reason=reason or "" 的字典;reason 为假值(默认 ""/None)时 truncated_reason 回落 "";
|
|
244
|
-
def note_index(self, *, kind: str, root: str, files: int, indexed: int,
|
|
245
|
-
truncated: bool, reason: str = "") -> None:
|
|
246
|
-
"""记「最近一次索引」结果——截断在这里留痕,供 diagnose 看见。"""
|
|
247
|
-
self.load()["last_index"] = {
|
|
248
|
-
"ts": _now(), "kind": kind, "root": root, "files": files,
|
|
249
|
-
"indexed": indexed, "truncated": bool(truncated),
|
|
250
|
-
"truncated_reason": reason or "",
|
|
251
|
-
}
|
|
252
|
-
|
|
253
|
-
# 生效条件:无参数调用即生效,取 self.load() 结果把 updated_at 置为 _now(),再以 atomic_write 把 json.dumps(..., ensure_ascii=False, indent=1, sort_keys=True) 写入 self.path,无返回值。
|
|
254
|
-
def save(self) -> None:
|
|
255
|
-
d = self.load()
|
|
256
|
-
d["updated_at"] = _now()
|
|
257
|
-
atomic_write(self.path, json.dumps(d, ensure_ascii=False,
|
|
258
|
-
indent=1, sort_keys=True))
|
|
259
|
-
|
|
260
|
-
# 生效条件:无参数调用即生效,返回含 path、schema、load()["files"] 条目数、nodes 总数(各条目 nodes 列表长度之和)、updated_at、exists=os.path.isfile(self.path) 的 out;age_s 在 updated_at 为假值(0.0)时为 None,否则为 max(0.0, time.time()-up);仅当 load() 的 last_index 为 dict 时才并入 out["last_index"]。
|
|
261
|
-
def summary(self) -> dict:
|
|
262
|
-
d = self.load()
|
|
263
|
-
files = d.get("files") or {}
|
|
264
|
-
nodes = sum(len(e.get("nodes") or []) for e in files.values())
|
|
265
|
-
up = float(d.get("updated_at") or 0.0)
|
|
266
|
-
out = {
|
|
267
|
-
"path": self.path,
|
|
268
|
-
"schema": d.get("schema"),
|
|
269
|
-
"files": len(files),
|
|
270
|
-
"nodes": nodes,
|
|
271
|
-
"updated_at": up,
|
|
272
|
-
"age_s": None if not up else max(0.0, time.time() - up),
|
|
273
|
-
"exists": os.path.isfile(self.path),
|
|
274
|
-
}
|
|
275
|
-
if isinstance(d.get("last_index"), dict):
|
|
276
|
-
out["last_index"] = d["last_index"]
|
|
277
|
-
return out
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
# --------------------------------------------------------------------------
|
|
281
|
-
# 统一 index_dir:调度 + 水位 + 落盘(供 op=index_code / op=index_doc / heal 共用)
|
|
282
|
-
# --------------------------------------------------------------------------
|
|
283
|
-
|
|
284
|
-
# 生效条件:root 为源大域根、kind 为 'code_ref'/'doc_ref' 时经 _mod(kind) 调度底层 index_dir 并返回 (items, errors, stats);ledger 非空时逐文件 record,incremental 为真时跳过 ledger.is_fresh 为真的文件,且 stats 未截断时执行 reconcile。
|
|
285
|
-
def index_dir(root: str, *, kind: str, patterns=None, max_files: int = 500,
|
|
286
|
-
max_items: int = 2000, incremental: bool = False,
|
|
287
|
-
ledger: "Ledger" = None, skip_dirs=None):
|
|
288
|
-
"""按 kind 调度 codeindex / docindex 的全量(或增量)索引。
|
|
289
|
-
|
|
290
|
-
incremental=True 且给了 ledger 时:未变文件跳过(`skipped_unchanged`)。
|
|
291
|
-
返回 (items, errors, stats),与底层 index_dir 的返回一致(多一个
|
|
292
|
-
`skipped_unchanged`)。
|
|
293
|
-
|
|
294
|
-
`skip_dirs` 透传给底层:**追加**排除、只增不减(内置 `.git`/`.venv`/
|
|
295
|
-
`node_modules` 等不可被关闭),见 `codeindex.skip_matcher`。实际排掉了哪些目录
|
|
296
|
-
由 `stats["skipped_dirs"]` 回报,仍不静默。
|
|
297
|
-
"""
|
|
298
|
-
mod = _mod(kind)
|
|
299
|
-
fresh = None
|
|
300
|
-
on_file = None
|
|
301
|
-
seen = set() # 本次真正走过的源文件(用于对账)
|
|
302
|
-
if ledger is not None:
|
|
303
|
-
if incremental:
|
|
304
|
-
# 生效条件:当 rel、fp 传入时,ok = ledger.is_fresh(rel, fp);若 ok 为真则将 _src_key(fp) 加入 seen 并返回 ok,若 ok 为假则直接返回 False;
|
|
305
|
-
def fresh(rel, fp): # noqa: E306
|
|
306
|
-
ok = ledger.is_fresh(rel, fp)
|
|
307
|
-
if ok:
|
|
308
|
-
seen.add(_src_key(fp))
|
|
309
|
-
return ok
|
|
310
|
-
|
|
311
|
-
# 生效条件:当 rel、fp、got 传入时,将 _src_key(fp) 加入 seen,并以 root=root 调用 ledger.record(rel, fp, kind, [{"id": node_id_of(it, kind), "lineno": it.get("lineno"), "end": it.get("end"), "hash": it.get("hash")} for it in got]);
|
|
312
|
-
def on_file(rel, fp, got): # noqa: E306
|
|
313
|
-
seen.add(_src_key(fp))
|
|
314
|
-
ledger.record(rel, fp, kind,
|
|
315
|
-
[{"id": node_id_of(it, kind), "lineno": it.get("lineno"),
|
|
316
|
-
"end": it.get("end"), "hash": it.get("hash")} for it in got],
|
|
317
|
-
root=root)
|
|
318
|
-
|
|
319
|
-
items, errors, stats = mod.index_dir(
|
|
320
|
-
root, patterns=patterns, max_files=max_files, max_items=max_items,
|
|
321
|
-
fresh=fresh, on_file=on_file, skip_dirs=skip_dirs,
|
|
322
|
-
)
|
|
323
|
-
if ledger is not None:
|
|
324
|
-
ledger.prune()
|
|
325
|
-
if not stats.get("truncated"):
|
|
326
|
-
# 没扫完就不能对账:截断时「没见到」不等于「源已消失」。
|
|
327
|
-
ledger.reconcile(root, kind, seen)
|
|
328
|
-
ledger.note_index(kind=kind, root=root, files=stats.get("files", 0),
|
|
329
|
-
indexed=len(items), truncated=bool(stats.get("truncated")),
|
|
330
|
-
reason=stats.get("truncated_reason") or "")
|
|
331
|
-
ledger.save()
|
|
332
|
-
return items, errors, stats
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
# 生效条件:it['path'] 非空时返回其首段 path.split('/')[0] 作为 domain 键,path 为空返回 'orphan'。
|
|
336
|
-
def _domain_of(it: dict) -> str:
|
|
337
|
-
"""条目 → 路由域键(供 `tags` 的 `domain:` 显式声明)。
|
|
338
|
-
|
|
339
|
-
与 `observation_position` **分开**:position 是给人读的条件文本(「本地
|
|
340
|
-
源码仓(大域=md_cg)」),domain 是给 `routing.route_key` 直取的短键。
|
|
341
|
-
两者混成一个字段就会重演普查里的退化:实例名嵌进条件字段 → 3037 桶 /
|
|
342
|
-
3048 节点(99.9% 单例桶),路由等于失效。
|
|
343
|
-
|
|
344
|
-
取 path 首段,与改造前 `normalize_domain(observation_position)` 的产物
|
|
345
|
-
**逐字相同**,故本次加标签不改变任何既有节点的分桶结果。
|
|
346
|
-
"""
|
|
347
|
-
path = it.get("path") or ""
|
|
348
|
-
return path.split("/")[0] or "orphan"
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
# 生效条件:kind == 'code_ref' 时按 codeindex.node_id/render 写入 cg(tags 含 'code'、code_ref=_code_ref(it, root)),kind == 'doc_ref' 时按 docindex 写入(tags 含 'doc'、doc_ref=_doc_ref(it, root)、密级取自 docindex.sensitivity_for(it['path'], sensitivity)),其他 kind 抛 ValueError,返回 (ids, sens)。
|
|
352
|
-
def add_items(cg, items, *, kind: str, root: str, layer=None, sensitivity=None,
|
|
353
|
-
layer_of=None):
|
|
354
|
-
"""把索引条目写进认知图(code / doc 的落盘细节收在这里,唯一实现)。
|
|
355
|
-
|
|
356
|
-
- `layer=None` → 默认 `knowledge`(与代码节点同层,保证进默认召回)。
|
|
357
|
-
- `layer_of(nid)` 可逐节点覆盖 layer(heal 重建时保留原层)。
|
|
358
|
-
- doc 节点:密级走 `docindex.sensitivity_for`(只可能更严);返回密级分布。
|
|
359
|
-
- `condition_space` 走 `codeindex/docindex.condition_space`,与正文的
|
|
360
|
-
`# 生效条件:` 行**同源**——改造前此处只写 `observation_position` 单槽,
|
|
361
|
-
而单槽不是生效条件,于是 frontmatter 的条件空间形同未声明。
|
|
362
|
-
返回 (ids, sens_counts)。
|
|
363
|
-
"""
|
|
364
|
-
from . import codeindex, docindex
|
|
365
|
-
ids, sens = [], {}
|
|
366
|
-
for it in items:
|
|
367
|
-
if kind == "code_ref":
|
|
368
|
-
nid = codeindex.node_id(it)
|
|
369
|
-
cg.add(
|
|
370
|
-
nid, codeindex.render(it),
|
|
371
|
-
layer=(layer_of(nid) if layer_of else None) or layer or "knowledge",
|
|
372
|
-
tags=["code", "code:" + it.get("kind", ""),
|
|
373
|
-
"domain:" + _domain_of(it)],
|
|
374
|
-
condition_space=codeindex.condition_space(it),
|
|
375
|
-
verification_basis=it.get("basis") or "compiler",
|
|
376
|
-
code_ref=_code_ref(it, root),
|
|
377
|
-
)
|
|
378
|
-
elif kind == "doc_ref":
|
|
379
|
-
nid = docindex.node_id(it)
|
|
380
|
-
level = it.get("level")
|
|
381
|
-
s, _basis = docindex.sensitivity_for(it.get("path") or "", sensitivity)
|
|
382
|
-
sens[s] = sens.get(s, 0) + 1
|
|
383
|
-
cg.add(
|
|
384
|
-
nid, docindex.render(it),
|
|
385
|
-
layer=(layer_of(nid) if layer_of else None) or layer or "knowledge",
|
|
386
|
-
tags=["doc", "doc:md", f"level:{level}",
|
|
387
|
-
"domain:" + _domain_of(it)],
|
|
388
|
-
condition_space=docindex.condition_space(it),
|
|
389
|
-
verification_basis="data",
|
|
390
|
-
sensitivity=s,
|
|
391
|
-
doc_ref=_doc_ref(it, root),
|
|
392
|
-
)
|
|
393
|
-
else:
|
|
394
|
-
raise ValueError(f"未知 ref kind:{kind!r}")
|
|
395
|
-
ids.append(nid)
|
|
396
|
-
return ids, sens
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
# 生效条件:把入参 root 原样写入返回 dict 的 'root',path/name/kind/lineno/end/lang/hash 按 it.get 取值(缺省 None),precise 取 bool(it.get('precise', True)),render_version 取传入值(传入 None 时延迟 import codeindex 取 codeindex.RENDER_VERSION,保证与 render 契约**同源**、无第二处硬编码)。
|
|
400
|
-
def _code_ref(it: dict, root: str, render_version=None) -> dict:
|
|
401
|
-
if render_version is None: # 直接调用点的兜底:与 render 产物同源
|
|
402
|
-
from . import codeindex
|
|
403
|
-
render_version = codeindex.RENDER_VERSION
|
|
404
|
-
return {
|
|
405
|
-
"path": it.get("path"), "name": it.get("name"),
|
|
406
|
-
"kind": it.get("kind"), "lineno": it.get("lineno"), "end": it.get("end"),
|
|
407
|
-
"lang": it.get("lang"), "precise": bool(it.get("precise", True)),
|
|
408
|
-
"hash": it.get("hash"), "root": root,
|
|
409
|
-
"render_version": render_version,
|
|
410
|
-
}
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
# 生效条件:把入参 root 原样写入返回 dict 的 'root',path/heading/heading_path/level/lineno/end/anchor/hash/lang 按 it.get 取值(缺省 None),precise 取 bool(it.get('precise', True))。
|
|
414
|
-
def _doc_ref(it: dict, root: str) -> dict:
|
|
415
|
-
return {
|
|
416
|
-
"path": it.get("path"), "heading": it.get("heading"),
|
|
417
|
-
"heading_path": it.get("heading_path"), "level": it.get("level"),
|
|
418
|
-
"lineno": it.get("lineno"), "end": it.get("end"),
|
|
419
|
-
"anchor": it.get("anchor"), "hash": it.get("hash"), "lang": it.get("lang"),
|
|
420
|
-
"precise": bool(it.get("precise", True)), "root": root,
|
|
421
|
-
}
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
# --------------------------------------------------------------------------
|
|
425
|
-
# 回读(唯一实现:op=ref 与 check_refs 共用)
|
|
426
|
-
# --------------------------------------------------------------------------
|
|
427
|
-
|
|
428
|
-
# 生效条件:传入 ref 为假值(如 None/{})时按 {} 处理,rel 取 ref.get("path") or "";root 与 ref.get("root") 均为假值时返回含 ref/path/status:"unresolved"/ok:False/error:"ref 未记录 root..." 的 out;否则用 root or ref.get("root") 与 rel 拼 fp,os.path.isfile(fp) 为假时返回 status:"dangling"、stale:True,读取抛 OSError/UnicodeDecodeError 时返回 status:"error";读取成功时 lineno 取 int(ref.get("lineno") or 1)(假值回落 1)、end 取 int(ref.get("end") or lineno)(假值回落 lineno),ref.get("hash") 为 None 时 match=None、ok=True、status:"ok",ref.get("hash") 为真值且等于 region_hash 时 ok=True/status:"ok"、不等时 ok=False/status:"stale",ref.get("hash") 为假值但非 None(如 ""/0/False)时 ok=False/status:"stale";with_text 为真时 out["text"] 取 lines[max(0,lineno-1):max(max(0,lineno-1),end)] 的 join;
|
|
429
|
-
def probe_ref(ref: dict, *, root: str = None, with_text: bool = False) -> dict:
|
|
430
|
-
"""只读探测单个 ref 的状态(不回读整篇,除非 with_text)。"""
|
|
431
|
-
from . import codeindex
|
|
432
|
-
ref = ref or {}
|
|
433
|
-
rel = ref.get("path") or ""
|
|
434
|
-
r = root or ref.get("root") or ""
|
|
435
|
-
base = {"ref": ref, "path": rel, "status": "unresolved", "ok": False}
|
|
436
|
-
if not r:
|
|
437
|
-
return {**base, "error": "ref 未记录 root,请显式传 root 参数"
|
|
438
|
-
"(索引里存的是相对 root 的 path)"}
|
|
439
|
-
fp = os.path.join(r, rel)
|
|
440
|
-
base["root"] = r
|
|
441
|
-
if not os.path.isfile(fp):
|
|
442
|
-
return {**base, "status": "dangling", "abspath": fp, "stale": True,
|
|
443
|
-
"error": f"源文件不存在(索引已悬空):{fp}"}
|
|
444
|
-
try:
|
|
445
|
-
with open(fp, "r", encoding="utf-8") as f:
|
|
446
|
-
lines = f.read().split("\n")
|
|
447
|
-
except (OSError, UnicodeDecodeError) as exc:
|
|
448
|
-
return {**base, "status": "error", "abspath": fp,
|
|
449
|
-
"error": f"读取失败:{exc}"}
|
|
450
|
-
total = len(lines)
|
|
451
|
-
lineno = int(ref.get("lineno") or 1)
|
|
452
|
-
end = int(ref.get("end") or lineno)
|
|
453
|
-
got = codeindex.region_hash(lines, lineno, end)
|
|
454
|
-
expect = ref.get("hash")
|
|
455
|
-
match = (got == expect) if expect else None
|
|
456
|
-
out = {
|
|
457
|
-
**base, "abspath": fp, "total_lines": total,
|
|
458
|
-
"hash": got, "hash_expected": expect, "hash_match": match,
|
|
459
|
-
"stale": bool(expect) and not match,
|
|
460
|
-
"ok": expect is None or bool(match),
|
|
461
|
-
"status": "ok" if (expect is None or match) else "stale",
|
|
462
|
-
}
|
|
463
|
-
if with_text:
|
|
464
|
-
lo = max(0, lineno - 1)
|
|
465
|
-
out["text"] = "\n".join(lines[lo:max(lo, end)])
|
|
466
|
-
return out
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
# 生效条件:ref 经 probe_ref(root=root, with_text=True) 后 status 为 'ok'/'stale' 时返回 ok=True 及 text/total_lines/hash/hash_match/stale/precise,status 为 'unresolved'/'error'/'dangling' 时返回 ok=False 与 error。
|
|
470
|
-
def read_ref(ref: dict, *, root: str = None, ref_kind: str = "ref") -> dict:
|
|
471
|
-
"""按 ref 回读源区间——`op=ref` 与 `check_refs` 的唯一实现。"""
|
|
472
|
-
p = probe_ref(ref, root=root, with_text=True)
|
|
473
|
-
base = {"ref": ref, "ref_kind": ref_kind}
|
|
474
|
-
if p["status"] in ("unresolved", "error", "dangling"):
|
|
475
|
-
out = {**base, "ok": False, "error": p["error"]}
|
|
476
|
-
if p["status"] == "dangling":
|
|
477
|
-
out["stale"] = True
|
|
478
|
-
return out
|
|
479
|
-
return {
|
|
480
|
-
**base, "ok": True, "text": p["text"], "total_lines": p["total_lines"],
|
|
481
|
-
"hash": p["hash"], "hash_expected": p["hash_expected"],
|
|
482
|
-
"hash_match": p["hash_match"], "stale": p["stale"],
|
|
483
|
-
"precise": bool((ref or {}).get("precise", True)),
|
|
484
|
-
"note": "按 ref 区间回读;hash_match=False 说明源已改动,"
|
|
485
|
-
"重跑 index_code / index_doc 重建",
|
|
486
|
-
}
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
# 生效条件:cg 的 index['nodes'] 非空时汇总 stale/dangling/unresolved/errors 并返回 ok =(无 stale 且无 dangling);only_tagged 为真时只探测 tags 含 'code'/'doc' 的节点,ledger 非空时先走 (size, mtime) 快路径。
|
|
490
|
-
def check_refs(cg, *, ledger: "Ledger" = None, max_nodes: int = MAX_CHECK,
|
|
491
|
-
only_tagged: bool = True) -> dict:
|
|
492
|
-
"""漂移 / 悬空巡检(只读、不抛)。
|
|
493
|
-
|
|
494
|
-
优先走 ledger 的 (size, mtime) 快路径:未变文件**不读盘**直接判 ok;
|
|
495
|
-
变了的文件读一次、按记录区间重算哈希判 stale。
|
|
496
|
-
ledger 覆盖不到的节点(早期索引 / 未开增量)再回退逐节点探测。
|
|
497
|
-
|
|
498
|
-
`only_tagged=True`(默认)只探测带 `code` / `doc` 标签的节点——ref 只由
|
|
499
|
-
`index_code` / `index_doc` 产生,两者都会打这两个标签;这样巡检不必为每条
|
|
500
|
-
记忆节点都读一次盘。需要穷举(含手写 ref)时传 False。
|
|
501
|
-
"""
|
|
502
|
-
nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
|
|
503
|
-
stale, dangling, unresolved, errors = [], [], [], []
|
|
504
|
-
covered = set()
|
|
505
|
-
|
|
506
|
-
# 生效条件:当 nid、ref、kind、rel 传入时,p = probe_ref(ref) 后按 p["status"] 分派:为 "dangling" 时把含 node_id/ref_kind/path/lineno/end/error 的 row 加入 dangling,为 "stale" 时补 hash_expected/hash 加入 stale,为 "unresolved" 时加入 unresolved,为 "error" 时加入 errors;其他状态不加入;
|
|
507
|
-
def _probe_one(nid, ref, kind, rel):
|
|
508
|
-
p = probe_ref(ref)
|
|
509
|
-
row = {"node_id": nid, "ref_kind": kind, "path": rel,
|
|
510
|
-
"lineno": ref.get("lineno"), "end": ref.get("end"),
|
|
511
|
-
"error": p.get("error", "")}
|
|
512
|
-
if p["status"] == "dangling":
|
|
513
|
-
dangling.append(row)
|
|
514
|
-
elif p["status"] == "stale":
|
|
515
|
-
row["hash_expected"] = p.get("hash_expected")
|
|
516
|
-
row["hash"] = p.get("hash")
|
|
517
|
-
stale.append(row)
|
|
518
|
-
elif p["status"] == "unresolved":
|
|
519
|
-
unresolved.append(row)
|
|
520
|
-
elif p["status"] == "error":
|
|
521
|
-
errors.append(row)
|
|
522
|
-
|
|
523
|
-
# 快路径:ledger 记录的文件(键 = 源文件绝对路径,天然跨大域不撞名)
|
|
524
|
-
if ledger is not None:
|
|
525
|
-
for key, e in sorted((ledger.load().get("files") or {}).items()):
|
|
526
|
-
rec_nodes = [n for n in (e.get("nodes") or [])
|
|
527
|
-
if n.get("id") in nodes]
|
|
528
|
-
if not rec_nodes:
|
|
529
|
-
continue
|
|
530
|
-
covered.update(n.get("id") for n in rec_nodes)
|
|
531
|
-
rel = e.get("path") or ""
|
|
532
|
-
# 必须用**每个源文件自己的 root**(索引时的源大域),不能用 ledger.root:
|
|
533
|
-
# 后者是认知图根,拿它拼路径会指向不存在的位置、把一切都误判成 dangling。
|
|
534
|
-
src_root = e.get("root") or os.path.dirname(key)
|
|
535
|
-
try:
|
|
536
|
-
st = os.stat(key)
|
|
537
|
-
unchanged = (e.get("size") == st.st_size
|
|
538
|
-
and abs(float(e.get("mtime") or 0.0) - st.st_mtime) < 1e-6)
|
|
539
|
-
except OSError:
|
|
540
|
-
unchanged = False
|
|
541
|
-
if not unchanged:
|
|
542
|
-
kind = e.get("kind") or kind_of_path(rel)
|
|
543
|
-
for n in rec_nodes:
|
|
544
|
-
_probe_one(n.get("id"), {**n, "path": rel, "root": src_root},
|
|
545
|
-
kind, rel)
|
|
546
|
-
|
|
547
|
-
# 回退:ledger 未覆盖的索引节点
|
|
548
|
-
# 生效条件:当 nid 传入时,若 only_tagged 为假值立即返回 True;否则取 (nodes.get(nid) or {}).get("tags") or [],仅当其中存在 "code" 或 "doc" 返回 True,否则返回 False;
|
|
549
|
-
def _candidate(nid):
|
|
550
|
-
if not only_tagged:
|
|
551
|
-
return True
|
|
552
|
-
tags = (nodes.get(nid) or {}).get("tags") or []
|
|
553
|
-
return any(t in ("code", "doc") for t in tags)
|
|
554
|
-
|
|
555
|
-
todo = [nid for nid in nodes if nid not in covered and _candidate(nid)]
|
|
556
|
-
truncated = len(todo) > max_nodes
|
|
557
|
-
checked = 0
|
|
558
|
-
for nid in todo[:max_nodes]:
|
|
559
|
-
try:
|
|
560
|
-
node = cg.get(nid)
|
|
561
|
-
except Exception:
|
|
562
|
-
continue
|
|
563
|
-
if not node:
|
|
564
|
-
continue
|
|
565
|
-
kind, ref = ref_of(node)
|
|
566
|
-
if not ref:
|
|
567
|
-
continue
|
|
568
|
-
checked += 1
|
|
569
|
-
try:
|
|
570
|
-
_probe_one(nid, ref, kind, ref.get("path") or "")
|
|
571
|
-
except Exception as exc: # 巡检不抛
|
|
572
|
-
errors.append({"node_id": nid, "ref_kind": kind, "error": str(exc)})
|
|
573
|
-
|
|
574
|
-
return {
|
|
575
|
-
"ok": not (stale or dangling),
|
|
576
|
-
"status": "ok" if not (stale or dangling) else ("dangling" if dangling else "stale"),
|
|
577
|
-
"checked": checked + len(covered),
|
|
578
|
-
"ledger_files": len((ledger.load().get("files") or {})) if ledger else 0,
|
|
579
|
-
"stale": stale, "dangling": dangling,
|
|
580
|
-
"unresolved": unresolved, "errors": errors,
|
|
581
|
-
"truncated": truncated, "max_nodes": max_nodes,
|
|
582
|
-
}
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
# --------------------------------------------------------------------------
|
|
586
|
-
# 节点级对账:孤儿清退 + 悬空清退
|
|
587
|
-
#
|
|
588
|
-
# 水位层的 `Ledger.reconcile` 只剪**水位条目**、`prune` 只剪「源大域已消失」
|
|
589
|
-
# 的条目,两者都不碰**节点**。于是节点层的两类残留无人处置:
|
|
590
|
-
# · 孤儿(同一文档的过期代)——标题路径一变 id 全量重算,旧代与新代并存;
|
|
591
|
-
# · 悬空(源文件已删)——回读必然失败,巡检永远报 dangling。
|
|
592
|
-
# 本节的唯一实现同时供 `op=index_code / index_doc`(孤儿)与
|
|
593
|
-
# `op=ref action=prune`(悬空)使用,避免两处各写一套口径。
|
|
594
|
-
# --------------------------------------------------------------------------
|
|
595
|
-
|
|
596
|
-
# 生效条件:返回 os.path.normcase(os.path.abspath(str(p or ''))),即 p 为 None/空串时返回当前目录的归一绝对路径。
|
|
597
|
-
def _norm_root(p) -> str:
|
|
598
|
-
"""root 归一:同一目录的大小写/分隔符差异不得影响「同一大域」判定。"""
|
|
599
|
-
return os.path.normcase(os.path.abspath(str(p or "")))
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
# 生效条件:a 与 b 都非空且 _norm_root(a) == _norm_root(b) 时返回 True,否则(含 TypeError/ValueError)返回 False。
|
|
603
|
-
def _same_root(a, b) -> bool:
|
|
604
|
-
try:
|
|
605
|
-
return bool(a) and bool(b) and _norm_root(a) == _norm_root(b)
|
|
606
|
-
except (TypeError, ValueError):
|
|
607
|
-
return False
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
# 生效条件:返回 str(p or '').replace('\\', '/').lstrip('./'),即 p 为 None/空串时返回 ''。
|
|
611
|
-
def _norm_rel(p) -> str:
|
|
612
|
-
return str(p or "").replace("\\", "/").lstrip("./")
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
# 生效条件:cg 具备可调用的 forget 方法时对 plan 中每个 nid 调 cg.forget(nid, why),返回 (成功 id 列表, 被拦下/失败的 {node_id, error} 列表);cg 无 forget 时返回 ([], plan 中每 nid 一条错误)。
|
|
616
|
-
def _forget_many(cg, plan, why: str) -> tuple:
|
|
617
|
-
"""逐条软删(进 trash/、写删除清单、可 restore);受保护节点拦下不删。
|
|
618
|
-
|
|
619
|
-
`forget` 属 **MdCGOS 层**能力(保护裁决 + 回收站 + 删除清单),基础层
|
|
620
|
-
`MdCG` 没有任何删除原语。缺能力时**明确报错、不静默跳过**——否则
|
|
621
|
-
「清退了 N 条」看着成功、实际一条没删(P27 §10 实测过这个坑)。
|
|
622
|
-
"""
|
|
623
|
-
fn = getattr(cg, "forget", None)
|
|
624
|
-
if not callable(fn):
|
|
625
|
-
err = (f"{type(cg).__name__} 无 forget 能力(对账须能删除节点);"
|
|
626
|
-
f"生产路径是 MdCGOS,测试请用 MdCGOS")
|
|
627
|
-
return [], [{"node_id": nid, "error": err} for nid in sorted(plan)]
|
|
628
|
-
done, blocked = [], []
|
|
629
|
-
for nid in sorted(plan):
|
|
630
|
-
try:
|
|
631
|
-
res = fn(nid, why) or {}
|
|
632
|
-
except Exception as exc: # ProtectionError 等 → 拦下,不越权
|
|
633
|
-
blocked.append({"node_id": nid, "error": str(exc)[:120]})
|
|
634
|
-
continue
|
|
635
|
-
if res.get("ok"):
|
|
636
|
-
done.append(nid)
|
|
637
|
-
else:
|
|
638
|
-
blocked.append({"node_id": nid, "error": str(res.get("error"))[:120]})
|
|
639
|
-
return done, blocked
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
# 生效条件:cg 具备可调用的 _unstage 时对 ghosts 逐个调用并收集成功 id(单条异常跳过),cg 无该能力时返回 [](不假装成功)。
|
|
643
|
-
def _drop_ghosts(cg, ghosts) -> list:
|
|
644
|
-
"""摘除幽灵条目的索引记录(节点文件已不存在,没有可软删的实体)。
|
|
645
|
-
|
|
646
|
-
走 `MdCG._unstage`:它顺带落删除记录,保证幽灵不会再次从分片日志里复活。
|
|
647
|
-
基础层没有该能力时如实返回空列表(不假装成功)。
|
|
648
|
-
"""
|
|
649
|
-
drop = getattr(cg, "_unstage", None)
|
|
650
|
-
if not callable(drop):
|
|
651
|
-
return []
|
|
652
|
-
out = []
|
|
653
|
-
for nid in ghosts:
|
|
654
|
-
try:
|
|
655
|
-
drop(nid)
|
|
656
|
-
out.append(nid)
|
|
657
|
-
except Exception: # 单条失败不拖垮整批
|
|
658
|
-
continue
|
|
659
|
-
return out
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
# 生效条件:items 中同 kind 的节点若其 ref['path'] 命中本次 items 的文件、id 不在本次产出内且 ref['root'] 与入参 root 同一(_same_root),则列入清退计划;dry_run 为真只返回计划,否则经 _forget_many 软删;items 为空时返回 count 0。
|
|
663
|
-
def prune_orphans(cg, *, kind: str, root: str, items, dry_run: bool = False,
|
|
664
|
-
reason: str = "") -> dict:
|
|
665
|
-
"""清退「同 root + 同 path,但已不在本次产出里」的**过期代**节点。
|
|
666
|
-
|
|
667
|
-
为什么必须有:`node_id = sha1(相对path + "#" + heading_path)`,而
|
|
668
|
-
`add_items` 只做**同 id 幂等 upsert**——文档标题结构一变,整篇 id 全量
|
|
669
|
-
重算,旧代节点无人清退,与新代并存(同一文档召回两份,且旧代引用的区间
|
|
670
|
-
已失效)。截断的索引由调用方负责不调用本函数(没扫完 ≠ 剩下的都过期)。
|
|
671
|
-
|
|
672
|
-
范围**只限本次真正重切过的文件**(`items` 的 path):增量索引跳过的未变
|
|
673
|
-
文件不在 items 里,其节点不进判定——否则会把完好的节点整片误删。
|
|
674
|
-
"""
|
|
675
|
-
nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
|
|
676
|
-
touched: dict = {}
|
|
677
|
-
for it in items or []:
|
|
678
|
-
rel = _norm_rel(it.get("path"))
|
|
679
|
-
if rel:
|
|
680
|
-
touched.setdefault(rel, set()).add(node_id_of(it, kind))
|
|
681
|
-
base = {"scanned": len(nodes), "touched_files": len(touched),
|
|
682
|
-
"dry_run": bool(dry_run)}
|
|
683
|
-
if not touched:
|
|
684
|
-
return {**base, "ok": True, "count": 0, "pruned": [],
|
|
685
|
-
"skipped_protected": []}
|
|
686
|
-
|
|
687
|
-
# ⚠ ref 只存在于**节点 frontmatter**里;`cg.index['nodes']` 是元数据快照
|
|
688
|
-
# (path/layer/tags/…,见 mdcg._scan_nodes),**不含 ref**。因此必须
|
|
689
|
-
# `cg.get(nid)` 取回节点再 ref_of —— 否则 ref_of 恒返回 ('', None)、
|
|
690
|
-
# 整个对账静默失效(P27 §10 实测过这个坑)。标签预筛与 check_refs 同口径,
|
|
691
|
-
# 免得为全库每条记忆都读一次盘。
|
|
692
|
-
tag = "doc" if kind == "doc_ref" else "code"
|
|
693
|
-
plan = []
|
|
694
|
-
for nid, e in nodes.items():
|
|
695
|
-
if tag not in ((e or {}).get("tags") or []):
|
|
696
|
-
continue
|
|
697
|
-
try:
|
|
698
|
-
node = cg.get(nid)
|
|
699
|
-
except Exception:
|
|
700
|
-
continue
|
|
701
|
-
if not node:
|
|
702
|
-
continue
|
|
703
|
-
k, ref = ref_of(node)
|
|
704
|
-
if k != kind or not isinstance(ref, dict):
|
|
705
|
-
continue
|
|
706
|
-
keep = touched.get(_norm_rel(ref.get("path")))
|
|
707
|
-
if keep is None or nid in keep or not _same_root(ref.get("root"), root):
|
|
708
|
-
continue
|
|
709
|
-
plan.append(nid)
|
|
710
|
-
|
|
711
|
-
why = reason or ("索引重建:本节点已不在同文档新代产出中(标题路径变更致 "
|
|
712
|
-
"node_id 重算),清退过期代以消除重复召回")
|
|
713
|
-
if dry_run:
|
|
714
|
-
return {**base, "ok": True, "count": len(plan), "pruned": sorted(plan)[:50],
|
|
715
|
-
"skipped_protected": [], "reason": why}
|
|
716
|
-
done, blocked = _forget_many(cg, plan, why)
|
|
717
|
-
return {**base, "ok": True, "count": len(done), "pruned": done[:50],
|
|
718
|
-
"skipped_protected": blocked[:20], "reason": why}
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
# 生效条件:cg 中带 'code'/'doc' 标签且 ref 带 root 的节点经 probe_ref 判为 'dangling' 时列入清退计划;only_roots 为空时另收集 cg.root 下取不到对应节点 .md 的幽灵条目经 _drop_ghosts 摘除;dry_run 为真只返回计划。
|
|
722
|
-
def prune_dangling(cg, *, only_roots=None, dry_run: bool = False,
|
|
723
|
-
max_nodes: int = MAX_CHECK, reason: str = "") -> dict:
|
|
724
|
-
"""清退**悬空**节点:ref 指向的源文件已删除,回读必然失败。
|
|
725
|
-
|
|
726
|
-
`check_refs` 只报告不处置(其原话是「悬空需人工处置」),本函数就是那个
|
|
727
|
-
出口——已删脚本、被搬走的文档留下的残留节点一次清掉,而不是逐条手工
|
|
728
|
-
`forget`。判定与巡检共用 `probe_ref` 的唯一实现,口径不会打架。
|
|
729
|
-
|
|
730
|
-
另清**幽灵条目**(ghosts):索引有条目、节点文件却不存在。它们是历史
|
|
731
|
-
「删除只摘内存索引、不落盘」的遗留——`cg.get` 取不回 → 悬空清退够不着它,
|
|
732
|
-
而 `check_refs` 走 ledger 会一直报 → dangling 永不归零。判据只用唯一真源
|
|
733
|
-
(节点 .md 不存在即脏索引),与「索引是派生物」的宣言一致;`only_roots`
|
|
734
|
-
非空时跳过(幽灵条目无 ref,无法归因到某个源大域)。
|
|
735
|
-
"""
|
|
736
|
-
nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
|
|
737
|
-
ghosts = []
|
|
738
|
-
if not only_roots:
|
|
739
|
-
for nid, e in list(nodes.items()):
|
|
740
|
-
tags = (e or {}).get("tags") or []
|
|
741
|
-
if not any(t in ("code", "doc") for t in tags):
|
|
742
|
-
continue
|
|
743
|
-
path = (e or {}).get("path") or ""
|
|
744
|
-
if path and not os.path.exists(os.path.join(cg.root, path)):
|
|
745
|
-
ghosts.append(nid)
|
|
746
|
-
# 同 prune_orphans:ref 只在节点 frontmatter 里,索引条目里没有,
|
|
747
|
-
# 必须 cg.get 取回节点再 ref_of(否则恒空、静默不删)。
|
|
748
|
-
todo = []
|
|
749
|
-
for nid, e in nodes.items():
|
|
750
|
-
tags = (e or {}).get("tags") or []
|
|
751
|
-
if not any(t in ("code", "doc") for t in tags):
|
|
752
|
-
continue
|
|
753
|
-
try:
|
|
754
|
-
node = cg.get(nid)
|
|
755
|
-
except Exception:
|
|
756
|
-
continue
|
|
757
|
-
if not node:
|
|
758
|
-
continue
|
|
759
|
-
_k, ref = ref_of(node)
|
|
760
|
-
if not ref or not ref.get("root"):
|
|
761
|
-
continue
|
|
762
|
-
if only_roots and not any(_same_root(ref.get("root"), r) for r in only_roots):
|
|
763
|
-
continue
|
|
764
|
-
todo.append((nid, ref))
|
|
765
|
-
truncated = len(todo) > max_nodes or len(ghosts) > max_nodes
|
|
766
|
-
plan = []
|
|
767
|
-
for nid, ref in todo[:max_nodes]:
|
|
768
|
-
try:
|
|
769
|
-
if probe_ref(ref).get("status") == "dangling":
|
|
770
|
-
plan.append(nid)
|
|
771
|
-
except Exception: # 探测失败不算悬空(宁可不删)
|
|
772
|
-
continue
|
|
773
|
-
|
|
774
|
-
why = reason or ("源文件已删除,索引节点悬空(回读必然失败),"
|
|
775
|
-
"清退以消除永不消失的 dangling")
|
|
776
|
-
ghost_plan = sorted(ghosts)[:max_nodes]
|
|
777
|
-
base = {"scanned": len(nodes), "candidates": len(plan),
|
|
778
|
-
"ghosts": len(ghost_plan),
|
|
779
|
-
"dry_run": bool(dry_run), "truncated": truncated,
|
|
780
|
-
"max_nodes": max_nodes}
|
|
781
|
-
if dry_run:
|
|
782
|
-
return {**base, "ok": True, "count": len(plan), "pruned": sorted(plan)[:50],
|
|
783
|
-
"ghost_pruned": ghost_plan[:50],
|
|
784
|
-
"skipped_protected": [], "reason": why}
|
|
785
|
-
done, blocked = _forget_many(cg, plan, why)
|
|
786
|
-
dropped = _drop_ghosts(cg, ghost_plan)
|
|
787
|
-
return {**base, "ok": True, "count": len(done), "pruned": done[:50],
|
|
788
|
-
"ghost_pruned": dropped[:50],
|
|
789
|
-
"skipped_protected": blocked[:20], "reason": why}
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
# 生效条件:cg 节点按 ref['root'] 与 kind 分组后逐组以 index_dir(incremental=False, ledger=ledger) 重切、再以 add_items(layer_of=原 layer) 重建,返回 {'ok','roots','groups','indexed','errors','truncated'};only_roots 非 None 时只处理其中列出的 root。
|
|
793
|
-
def rebuild(cg, *, ledger: "Ledger" = None, only_roots=None, max_files: int = 500,
|
|
794
|
-
max_items: int = 2000) -> dict:
|
|
795
|
-
"""按 ref 记录的 root 重建索引(sustain.heal 的修复动作)。
|
|
796
|
-
|
|
797
|
-
只重跑出了问题的 root(`only_roots`),逐节点**保留原 layer**;
|
|
798
|
-
doc 密级用默认策略重算(默认只可能更严,不会放松)。
|
|
799
|
-
"""
|
|
800
|
-
nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
|
|
801
|
-
groups = {}
|
|
802
|
-
for nid in nodes:
|
|
803
|
-
try:
|
|
804
|
-
node = cg.get(nid)
|
|
805
|
-
except Exception:
|
|
806
|
-
continue
|
|
807
|
-
kind, ref = ref_of(node)
|
|
808
|
-
if not ref or not ref.get("root"):
|
|
809
|
-
continue
|
|
810
|
-
r = ref["root"]
|
|
811
|
-
if only_roots is not None and r not in only_roots:
|
|
812
|
-
continue
|
|
813
|
-
groups.setdefault((r, kind), 0)
|
|
814
|
-
groups[(r, kind)] += 1
|
|
815
|
-
|
|
816
|
-
out = {"ok": True, "roots": sorted({r for r, _ in groups}),
|
|
817
|
-
"groups": len(groups), "indexed": 0, "errors": [], "truncated": False}
|
|
818
|
-
for (root, kind) in sorted(groups):
|
|
819
|
-
try:
|
|
820
|
-
items, errors, stats = index_dir(
|
|
821
|
-
root, kind=kind, max_files=max_files, max_items=max_items,
|
|
822
|
-
incremental=False, ledger=ledger)
|
|
823
|
-
ids, _sens = add_items(
|
|
824
|
-
cg, items, kind=kind, root=root,
|
|
825
|
-
layer_of=lambda nid: (nodes.get(nid) or {}).get("layer"))
|
|
826
|
-
out["indexed"] += len(ids)
|
|
827
|
-
out["errors"].extend(errors)
|
|
828
|
-
out["truncated"] = out["truncated"] or bool(stats.get("truncated"))
|
|
829
|
-
except Exception as exc: # 自愈不抛
|
|
830
|
-
out["ok"] = False
|
|
831
|
-
out["errors"].append(f"{root} [{kind}]:{exc}")
|
|
832
|
-
if ledger is not None:
|
|
833
|
-
ledger.save()
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""md_cg · 统一 ref 协议 + 索引水位(增量)+ 漂移/悬空巡检
|
|
3
|
+
|
|
4
|
+
对照 `docs/mdcg/认知图_索引与工程规范化_计划_v0.1.md` 的 R3(修 D + 修 F):
|
|
5
|
+
|
|
6
|
+
**D 漂移 / 悬空检测(本模块 `check_refs`)**
|
|
7
|
+
扫描带 `code_ref` / `doc_ref` 的节点,回两类问题:
|
|
8
|
+
· `stale` —— 源文件被改(区间哈希不再匹配)
|
|
9
|
+
· `dangling` —— 源文件被删(索引指向不存在的文件)
|
|
10
|
+
巡检**只读**、**不抛**、**不改源文件**;修复动作是「重跑 index_code / index_doc」,
|
|
11
|
+
因为索引是派生物(对齐 sustain.heal 的既有边界)。
|
|
12
|
+
|
|
13
|
+
**F 全量重扫 + 静默截断(本模块 `Ledger` + `index_dir`)**
|
|
14
|
+
`<root>/_refindex.json` 是 ref 索引水位(抄 `sources.Ingestor` 的 `_sources.json` 范式),
|
|
15
|
+
以**源文件绝对路径**为键,记每个源文件的 (size, mtime) 与节点区间 + 它所属的**源大域
|
|
16
|
+
root**;`incremental=True` 时未变文件**不再读盘重切**,直接跳过(`skipped_unchanged`)
|
|
17
|
+
——这就是「不全量重扫」。
|
|
18
|
+
键用绝对路径、且逐文件记 root,是因为一份认知图可以索引多个大域:只按 rel 记会在同名
|
|
19
|
+
文件上互相覆盖,巡检时若拿认知图根去拼路径则会把一切都误判成 dangling。
|
|
20
|
+
截断(`max_files` / `max_items`)由 codeindex / docindex 显式上报,本模块把
|
|
21
|
+
「最近一次索引被截断」写进水位,交给 `sustain.diagnose` 巡检看见(不再静默)。
|
|
22
|
+
|
|
23
|
+
**为什么回读要收进本模块**
|
|
24
|
+
`op=ref` 的回读与 `check_refs` 的判定**必须共用同一实现**,否则会出现
|
|
25
|
+
「回读说没漂、巡检说有漂」。与 `region_hash` 的教训同源:区间哈希只允许一份实现,
|
|
26
|
+
这里连「怎么判定 ok / stale / dangling」也只允许一份。
|
|
27
|
+
|
|
28
|
+
零第三方依赖。
|
|
29
|
+
"""
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import json
|
|
33
|
+
import os
|
|
34
|
+
import time
|
|
35
|
+
|
|
36
|
+
from .fsutil import atomic_write
|
|
37
|
+
|
|
38
|
+
SCHEMA = 2 # v2:水位以「源文件绝对路径」为键(v1 按 rel 会跨大域撞名)
|
|
39
|
+
LEDGER_FILE = "_refindex.json"
|
|
40
|
+
REF_KEYS = ("code_ref", "doc_ref")
|
|
41
|
+
MAX_CHECK = 2000 # 巡检节点上限(超出报 truncated,不静默截断)
|
|
42
|
+
STATUSES = ("ok", "stale", "dangling", "unresolved", "error")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
# 生效条件:给定 fp 时返回 os.path.abspath(fp or '')(fp 为空/None 则返回当前目录的绝对路径),作为水位键以绝对路径保证不同 root 下同名文件不互相覆盖。
|
|
46
|
+
def _src_key(fp: str) -> str:
|
|
47
|
+
"""水位的键 = 源文件绝对路径。
|
|
48
|
+
|
|
49
|
+
不能用 rel:一份认知图可以索引多个大域(不同 root),只按 rel 记会在
|
|
50
|
+
`alpha.py` 这种同名文件上互相覆盖——水位被静默丢掉,巡检就漏报。
|
|
51
|
+
"""
|
|
52
|
+
return os.path.abspath(fp or "")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
# 生效条件:无 required 形参,任何调用都返回 round(time.time(), 1),把时间戳压到 1 位小数以稳定 `_refindex.json` 字节数。
|
|
56
|
+
def _now() -> float:
|
|
57
|
+
"""时间戳压到 1 位小数:让 `_refindex.json` 字节数稳定(重跑不涨),
|
|
58
|
+
同时保留足够的「多久以前」信息(float 的最短 repr 保证小数位固定为 1)。"""
|
|
59
|
+
return round(time.time(), 1)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
# --------------------------------------------------------------------------
|
|
63
|
+
# 提取器注册表(统一调度:调用方只说 kind,不说「用哪个模块」)
|
|
64
|
+
# --------------------------------------------------------------------------
|
|
65
|
+
|
|
66
|
+
# 生效条件:kind == 'code_ref' 返回 codeindex、kind == 'doc_ref' 返回 docindex,其他 kind 抛 ValueError(提示支持 REF_KEYS)。
|
|
67
|
+
def _mod(kind: str):
|
|
68
|
+
from . import codeindex, docindex
|
|
69
|
+
if kind == "code_ref":
|
|
70
|
+
return codeindex
|
|
71
|
+
if kind == "doc_ref":
|
|
72
|
+
return docindex
|
|
73
|
+
raise ValueError(f"未知 ref kind:{kind!r}(支持 {REF_KEYS})")
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
# 生效条件:无 required 形参,调用即返回 {'code_ref': {'suffixes': tuple(codeindex.SUFFIX)}, 'doc_ref': {'suffixes': tuple(docindex.SUFFIX)}}。
|
|
77
|
+
def registry() -> dict:
|
|
78
|
+
"""后缀 → kind 的注册表(code / doc 各一份提取器)。"""
|
|
79
|
+
from . import codeindex, docindex
|
|
80
|
+
return {
|
|
81
|
+
"code_ref": {"suffixes": tuple(codeindex.SUFFIX)},
|
|
82
|
+
"doc_ref": {"suffixes": tuple(docindex.SUFFIX)},
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
# 生效条件:path 的小写后缀在 codeindex.EXTRACTORS 中返回 'code_ref',在 docindex.SUFFIX 中返回 'doc_ref',无后缀或均不匹配返回 ''。
|
|
87
|
+
def kind_of_path(path: str) -> str:
|
|
88
|
+
"""按后缀判 kind;无提取器返回 ''(由调用方决定是报错还是跳过)。"""
|
|
89
|
+
from . import codeindex, docindex
|
|
90
|
+
ext = os.path.splitext(path or "")[1].lower()
|
|
91
|
+
if not ext:
|
|
92
|
+
return ""
|
|
93
|
+
if ext in codeindex.EXTRACTORS:
|
|
94
|
+
return "code_ref"
|
|
95
|
+
if ext in docindex.SUFFIX:
|
|
96
|
+
return "doc_ref"
|
|
97
|
+
return ""
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
# 生效条件:source 为待提取文本,kind 非空或 path 后缀能推出 kind 时返回 _mod(k).extract(source, path),推不出 kind 时抛 ValueError。
|
|
101
|
+
def extract(source: str, path: str = "", kind: str = ""):
|
|
102
|
+
"""统一提取入口:按 kind(或从 path 推断)分发到对应 extractor。"""
|
|
103
|
+
k = kind or kind_of_path(path)
|
|
104
|
+
if not k:
|
|
105
|
+
ext = os.path.splitext(path or "")[1] or "<none>"
|
|
106
|
+
raise ValueError(f"无索引提取器(suffix={ext})")
|
|
107
|
+
return _mod(k).extract(source, path)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
# 生效条件:kind 为 'code_ref'/'doc_ref' 时返回 _mod(kind).node_id(item),其他 kind 由 _mod 抛 ValueError。
|
|
111
|
+
def node_id_of(item: dict, kind: str) -> str:
|
|
112
|
+
return _mod(kind).node_id(item)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
# 生效条件:kind 为 'code_ref'/'doc_ref' 时返回 _mod(kind).render(item),其他 kind 由 _mod 抛 ValueError。
|
|
116
|
+
def render_of(item: dict, kind: str) -> str:
|
|
117
|
+
return _mod(kind).render(item)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
# 生效条件:node 的 frontmatter 中 REF_KEYS 命中且值为非空 dict 时返回 {'ref': ref, 'ref_kind': kind},否则返回 {'ref': None, 'ref_kind': ''}。
|
|
121
|
+
def ref_fields(node) -> dict:
|
|
122
|
+
"""节点 → 检索结果要带的两字段(读侧只加字段,不改召回逻辑)。"""
|
|
123
|
+
kind, ref = ref_of(node)
|
|
124
|
+
return {"ref": ref, "ref_kind": kind} if ref else {"ref": None, "ref_kind": ""}
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
# 生效条件:node 的 frontmatter 按 REF_KEYS 顺序取到第一个非空 dict 时返回 (k, r),否则返回 ('', None)。
|
|
128
|
+
def ref_of(node) -> tuple:
|
|
129
|
+
"""从节点 frontmatter 取 ref:返回 (kind, ref) 或 ('', None)。"""
|
|
130
|
+
fm = (node or {}).get("frontmatter") or {}
|
|
131
|
+
for k in REF_KEYS:
|
|
132
|
+
r = fm.get(k)
|
|
133
|
+
if isinstance(r, dict) and r:
|
|
134
|
+
return k, r
|
|
135
|
+
return "", None
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
# --------------------------------------------------------------------------
|
|
139
|
+
# 索引水位(_refindex.json):增量 + 截断留痕
|
|
140
|
+
# --------------------------------------------------------------------------
|
|
141
|
+
|
|
142
|
+
# 生效条件:以 root 为必填实参构造,实例化即置 self.root=root、self.path=os.path.join(root, LEDGER_FILE)、self._d=None;
|
|
143
|
+
class Ledger:
|
|
144
|
+
"""`<root>/_refindex.json`:每个源文件的 (size, mtime) 水位 + 节点区间。"""
|
|
145
|
+
|
|
146
|
+
# 生效条件:当传入 root 时,self.root 取该 root,self.path 为 os.path.join(root, LEDGER_FILE),self._d 置为 None;
|
|
147
|
+
def __init__(self, root: str):
|
|
148
|
+
self.root = root
|
|
149
|
+
self.path = os.path.join(root, LEDGER_FILE)
|
|
150
|
+
self._d = None
|
|
151
|
+
|
|
152
|
+
# 生效条件:当 self._d is not None 时直接返回 self._d;否则读取 self.path 的 JSON,仅当 obj 是 dict 且 obj.get("schema") == SCHEMA 且 obj.get("files") 是 dict 时用 obj,否则(含 OSError/ValueError、结构不符)回落为 {"schema": SCHEMA, "updated_at": 0.0, "files": {}} 并缓存返回;
|
|
153
|
+
def load(self) -> dict:
|
|
154
|
+
if self._d is not None:
|
|
155
|
+
return self._d
|
|
156
|
+
d = None
|
|
157
|
+
try:
|
|
158
|
+
with open(self.path, "r", encoding="utf-8") as f:
|
|
159
|
+
obj = json.load(f)
|
|
160
|
+
if isinstance(obj, dict) and obj.get("schema") == SCHEMA \
|
|
161
|
+
and isinstance(obj.get("files"), dict):
|
|
162
|
+
d = obj
|
|
163
|
+
except (OSError, ValueError):
|
|
164
|
+
d = None
|
|
165
|
+
self._d = d or {"schema": SCHEMA, "updated_at": 0.0, "files": {}}
|
|
166
|
+
return self._d
|
|
167
|
+
|
|
168
|
+
# 生效条件:传入 rel、fp 时,若 self.load()["files"].get(_src_key(fp)) 缺失或为假值、或 os.stat(fp) 抛 OSError、或条目 e.get("size") != st.st_size,则返回 False;否则返回 abs(float(e.get("mtime") or 0.0) - st.st_mtime) < 1e-6(mtime 缺失或假值时按 0.0);
|
|
169
|
+
def is_fresh(self, rel: str, fp: str) -> bool:
|
|
170
|
+
"""源文件自上次索引后未变(size + mtime 双等)→ 可跳过不重切。"""
|
|
171
|
+
e = self.load()["files"].get(_src_key(fp))
|
|
172
|
+
if not e:
|
|
173
|
+
return False
|
|
174
|
+
try:
|
|
175
|
+
st = os.stat(fp)
|
|
176
|
+
except OSError:
|
|
177
|
+
return False
|
|
178
|
+
if e.get("size") != st.st_size:
|
|
179
|
+
return False
|
|
180
|
+
return abs(float(e.get("mtime") or 0.0) - st.st_mtime) < 1e-6
|
|
181
|
+
|
|
182
|
+
# 生效条件:当 rel、fp、kind、nodes 传入且 os.stat(fp) 成功时,向 self.load()["files"][_src_key(fp)] 写条目,其中 root 为 root if root else os.path.dirname(key)、path 为 rel、kind 为 kind、size/mtime 取 st、nodes 为每项 n.get("id")/n.get("lineno")/n.get("end")/n.get("hash");os.stat(fp) 抛 OSError 时不写入;
|
|
183
|
+
def record(self, rel: str, fp: str, kind: str, nodes,
|
|
184
|
+
root: str = None) -> None:
|
|
185
|
+
"""记一个源文件的水位(节点区间用于判 stale)。
|
|
186
|
+
|
|
187
|
+
`root` 是**源**大域的根(≠ 认知图根):巡检要拿它拼 `root/rel` 才能
|
|
188
|
+
找到源文件,缺了它就会把「源在别处」误判成 dangling。
|
|
189
|
+
"""
|
|
190
|
+
key = _src_key(fp)
|
|
191
|
+
try:
|
|
192
|
+
st = os.stat(fp)
|
|
193
|
+
except OSError:
|
|
194
|
+
return
|
|
195
|
+
self.load()["files"][key] = {
|
|
196
|
+
"root": root if root else os.path.dirname(key),
|
|
197
|
+
"path": rel,
|
|
198
|
+
"size": st.st_size,
|
|
199
|
+
"mtime": st.st_mtime,
|
|
200
|
+
"kind": kind,
|
|
201
|
+
"nodes": [
|
|
202
|
+
{"id": n.get("id"), "lineno": n.get("lineno"),
|
|
203
|
+
"end": n.get("end"), "hash": n.get("hash")}
|
|
204
|
+
for n in nodes
|
|
205
|
+
],
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
# 生效条件:当传入 fp 时,self.load()["files"].pop(_src_key(fp), None),即删除对应键(不存在也静默);
|
|
209
|
+
def drop(self, fp: str) -> None:
|
|
210
|
+
self.load()["files"].pop(_src_key(fp), None)
|
|
211
|
+
|
|
212
|
+
# 生效条件:当传入 root、kind、seen 时,对 self.load()["files"] 中满足 os.path.abspath(e.get("root") or "") == os.path.abspath(root) 且 e.get("kind") == kind 且键 k 不在 seen 的条目删除,返回删除数量;
|
|
213
|
+
def reconcile(self, root: str, kind: str, seen) -> int:
|
|
214
|
+
"""一次**完整**索引后对账:本 (root, kind) 下没被扫到的旧条目剪掉。
|
|
215
|
+
|
|
216
|
+
否则「源文件被删 → 索引悬空 → heal 重建」之后条目还在,巡检就永远报
|
|
217
|
+
dangling,heal 是治不好的。`seen` 是本次真正走过(提取成功或判定未变)
|
|
218
|
+
的源文件键集合。**截断的索引不能对账**——没扫完不等于剩下的都消失了。
|
|
219
|
+
"""
|
|
220
|
+
r = os.path.abspath(root)
|
|
221
|
+
files = self.load()["files"]
|
|
222
|
+
dead = [k for k, e in files.items()
|
|
223
|
+
if os.path.abspath(e.get("root") or "") == r
|
|
224
|
+
and e.get("kind") == kind and k not in seen]
|
|
225
|
+
for k in dead:
|
|
226
|
+
files.pop(k, None)
|
|
227
|
+
return len(dead)
|
|
228
|
+
|
|
229
|
+
# 生效条件:对 load()["files"] 中「条目 root(为假值时用 os.path.dirname(键) 兜底)不是目录」的条目逐一 pop 并返回删除条数,无匹配时返回 0。
|
|
230
|
+
def prune(self) -> int:
|
|
231
|
+
"""剪掉「源大域已不存在」的条目(整个目录被搬走/删除)。
|
|
232
|
+
|
|
233
|
+
这类条目已不可能再被任何大域索引到,留着只会在巡检里报永不消失的
|
|
234
|
+
dangling;而节点自带的 ref 仍会兜底探测,所以剪掉不会漏报真实悬空。
|
|
235
|
+
"""
|
|
236
|
+
files = self.load()["files"]
|
|
237
|
+
dead = [k for k, e in files.items()
|
|
238
|
+
if not os.path.isdir(e.get("root") or os.path.dirname(k))]
|
|
239
|
+
for k in dead:
|
|
240
|
+
files.pop(k, None)
|
|
241
|
+
return len(dead)
|
|
242
|
+
|
|
243
|
+
# 生效条件:当 kind、root、files、indexed、truncated 传入时,self.load()["last_index"] 被设为含 ts=_now()、kind、root、files、indexed、truncated=bool(truncated)、truncated_reason=reason or "" 的字典;reason 为假值(默认 ""/None)时 truncated_reason 回落 "";
|
|
244
|
+
def note_index(self, *, kind: str, root: str, files: int, indexed: int,
|
|
245
|
+
truncated: bool, reason: str = "") -> None:
|
|
246
|
+
"""记「最近一次索引」结果——截断在这里留痕,供 diagnose 看见。"""
|
|
247
|
+
self.load()["last_index"] = {
|
|
248
|
+
"ts": _now(), "kind": kind, "root": root, "files": files,
|
|
249
|
+
"indexed": indexed, "truncated": bool(truncated),
|
|
250
|
+
"truncated_reason": reason or "",
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
# 生效条件:无参数调用即生效,取 self.load() 结果把 updated_at 置为 _now(),再以 atomic_write 把 json.dumps(..., ensure_ascii=False, indent=1, sort_keys=True) 写入 self.path,无返回值。
|
|
254
|
+
def save(self) -> None:
|
|
255
|
+
d = self.load()
|
|
256
|
+
d["updated_at"] = _now()
|
|
257
|
+
atomic_write(self.path, json.dumps(d, ensure_ascii=False,
|
|
258
|
+
indent=1, sort_keys=True))
|
|
259
|
+
|
|
260
|
+
# 生效条件:无参数调用即生效,返回含 path、schema、load()["files"] 条目数、nodes 总数(各条目 nodes 列表长度之和)、updated_at、exists=os.path.isfile(self.path) 的 out;age_s 在 updated_at 为假值(0.0)时为 None,否则为 max(0.0, time.time()-up);仅当 load() 的 last_index 为 dict 时才并入 out["last_index"]。
|
|
261
|
+
def summary(self) -> dict:
|
|
262
|
+
d = self.load()
|
|
263
|
+
files = d.get("files") or {}
|
|
264
|
+
nodes = sum(len(e.get("nodes") or []) for e in files.values())
|
|
265
|
+
up = float(d.get("updated_at") or 0.0)
|
|
266
|
+
out = {
|
|
267
|
+
"path": self.path,
|
|
268
|
+
"schema": d.get("schema"),
|
|
269
|
+
"files": len(files),
|
|
270
|
+
"nodes": nodes,
|
|
271
|
+
"updated_at": up,
|
|
272
|
+
"age_s": None if not up else max(0.0, time.time() - up),
|
|
273
|
+
"exists": os.path.isfile(self.path),
|
|
274
|
+
}
|
|
275
|
+
if isinstance(d.get("last_index"), dict):
|
|
276
|
+
out["last_index"] = d["last_index"]
|
|
277
|
+
return out
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
# --------------------------------------------------------------------------
|
|
281
|
+
# 统一 index_dir:调度 + 水位 + 落盘(供 op=index_code / op=index_doc / heal 共用)
|
|
282
|
+
# --------------------------------------------------------------------------
|
|
283
|
+
|
|
284
|
+
# 生效条件:root 为源大域根、kind 为 'code_ref'/'doc_ref' 时经 _mod(kind) 调度底层 index_dir 并返回 (items, errors, stats);ledger 非空时逐文件 record,incremental 为真时跳过 ledger.is_fresh 为真的文件,且 stats 未截断时执行 reconcile。
|
|
285
|
+
def index_dir(root: str, *, kind: str, patterns=None, max_files: int = 500,
|
|
286
|
+
max_items: int = 2000, incremental: bool = False,
|
|
287
|
+
ledger: "Ledger" = None, skip_dirs=None):
|
|
288
|
+
"""按 kind 调度 codeindex / docindex 的全量(或增量)索引。
|
|
289
|
+
|
|
290
|
+
incremental=True 且给了 ledger 时:未变文件跳过(`skipped_unchanged`)。
|
|
291
|
+
返回 (items, errors, stats),与底层 index_dir 的返回一致(多一个
|
|
292
|
+
`skipped_unchanged`)。
|
|
293
|
+
|
|
294
|
+
`skip_dirs` 透传给底层:**追加**排除、只增不减(内置 `.git`/`.venv`/
|
|
295
|
+
`node_modules` 等不可被关闭),见 `codeindex.skip_matcher`。实际排掉了哪些目录
|
|
296
|
+
由 `stats["skipped_dirs"]` 回报,仍不静默。
|
|
297
|
+
"""
|
|
298
|
+
mod = _mod(kind)
|
|
299
|
+
fresh = None
|
|
300
|
+
on_file = None
|
|
301
|
+
seen = set() # 本次真正走过的源文件(用于对账)
|
|
302
|
+
if ledger is not None:
|
|
303
|
+
if incremental:
|
|
304
|
+
# 生效条件:当 rel、fp 传入时,ok = ledger.is_fresh(rel, fp);若 ok 为真则将 _src_key(fp) 加入 seen 并返回 ok,若 ok 为假则直接返回 False;
|
|
305
|
+
def fresh(rel, fp): # noqa: E306
|
|
306
|
+
ok = ledger.is_fresh(rel, fp)
|
|
307
|
+
if ok:
|
|
308
|
+
seen.add(_src_key(fp))
|
|
309
|
+
return ok
|
|
310
|
+
|
|
311
|
+
# 生效条件:当 rel、fp、got 传入时,将 _src_key(fp) 加入 seen,并以 root=root 调用 ledger.record(rel, fp, kind, [{"id": node_id_of(it, kind), "lineno": it.get("lineno"), "end": it.get("end"), "hash": it.get("hash")} for it in got]);
|
|
312
|
+
def on_file(rel, fp, got): # noqa: E306
|
|
313
|
+
seen.add(_src_key(fp))
|
|
314
|
+
ledger.record(rel, fp, kind,
|
|
315
|
+
[{"id": node_id_of(it, kind), "lineno": it.get("lineno"),
|
|
316
|
+
"end": it.get("end"), "hash": it.get("hash")} for it in got],
|
|
317
|
+
root=root)
|
|
318
|
+
|
|
319
|
+
items, errors, stats = mod.index_dir(
|
|
320
|
+
root, patterns=patterns, max_files=max_files, max_items=max_items,
|
|
321
|
+
fresh=fresh, on_file=on_file, skip_dirs=skip_dirs,
|
|
322
|
+
)
|
|
323
|
+
if ledger is not None:
|
|
324
|
+
ledger.prune()
|
|
325
|
+
if not stats.get("truncated"):
|
|
326
|
+
# 没扫完就不能对账:截断时「没见到」不等于「源已消失」。
|
|
327
|
+
ledger.reconcile(root, kind, seen)
|
|
328
|
+
ledger.note_index(kind=kind, root=root, files=stats.get("files", 0),
|
|
329
|
+
indexed=len(items), truncated=bool(stats.get("truncated")),
|
|
330
|
+
reason=stats.get("truncated_reason") or "")
|
|
331
|
+
ledger.save()
|
|
332
|
+
return items, errors, stats
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
# 生效条件:it['path'] 非空时返回其首段 path.split('/')[0] 作为 domain 键,path 为空返回 'orphan'。
|
|
336
|
+
def _domain_of(it: dict) -> str:
|
|
337
|
+
"""条目 → 路由域键(供 `tags` 的 `domain:` 显式声明)。
|
|
338
|
+
|
|
339
|
+
与 `observation_position` **分开**:position 是给人读的条件文本(「本地
|
|
340
|
+
源码仓(大域=md_cg)」),domain 是给 `routing.route_key` 直取的短键。
|
|
341
|
+
两者混成一个字段就会重演普查里的退化:实例名嵌进条件字段 → 3037 桶 /
|
|
342
|
+
3048 节点(99.9% 单例桶),路由等于失效。
|
|
343
|
+
|
|
344
|
+
取 path 首段,与改造前 `normalize_domain(observation_position)` 的产物
|
|
345
|
+
**逐字相同**,故本次加标签不改变任何既有节点的分桶结果。
|
|
346
|
+
"""
|
|
347
|
+
path = it.get("path") or ""
|
|
348
|
+
return path.split("/")[0] or "orphan"
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
# 生效条件:kind == 'code_ref' 时按 codeindex.node_id/render 写入 cg(tags 含 'code'、code_ref=_code_ref(it, root)),kind == 'doc_ref' 时按 docindex 写入(tags 含 'doc'、doc_ref=_doc_ref(it, root)、密级取自 docindex.sensitivity_for(it['path'], sensitivity)),其他 kind 抛 ValueError,返回 (ids, sens)。
|
|
352
|
+
def add_items(cg, items, *, kind: str, root: str, layer=None, sensitivity=None,
|
|
353
|
+
layer_of=None):
|
|
354
|
+
"""把索引条目写进认知图(code / doc 的落盘细节收在这里,唯一实现)。
|
|
355
|
+
|
|
356
|
+
- `layer=None` → 默认 `knowledge`(与代码节点同层,保证进默认召回)。
|
|
357
|
+
- `layer_of(nid)` 可逐节点覆盖 layer(heal 重建时保留原层)。
|
|
358
|
+
- doc 节点:密级走 `docindex.sensitivity_for`(只可能更严);返回密级分布。
|
|
359
|
+
- `condition_space` 走 `codeindex/docindex.condition_space`,与正文的
|
|
360
|
+
`# 生效条件:` 行**同源**——改造前此处只写 `observation_position` 单槽,
|
|
361
|
+
而单槽不是生效条件,于是 frontmatter 的条件空间形同未声明。
|
|
362
|
+
返回 (ids, sens_counts)。
|
|
363
|
+
"""
|
|
364
|
+
from . import codeindex, docindex
|
|
365
|
+
ids, sens = [], {}
|
|
366
|
+
for it in items:
|
|
367
|
+
if kind == "code_ref":
|
|
368
|
+
nid = codeindex.node_id(it)
|
|
369
|
+
cg.add(
|
|
370
|
+
nid, codeindex.render(it),
|
|
371
|
+
layer=(layer_of(nid) if layer_of else None) or layer or "knowledge",
|
|
372
|
+
tags=["code", "code:" + it.get("kind", ""),
|
|
373
|
+
"domain:" + _domain_of(it)],
|
|
374
|
+
condition_space=codeindex.condition_space(it),
|
|
375
|
+
verification_basis=it.get("basis") or "compiler",
|
|
376
|
+
code_ref=_code_ref(it, root),
|
|
377
|
+
)
|
|
378
|
+
elif kind == "doc_ref":
|
|
379
|
+
nid = docindex.node_id(it)
|
|
380
|
+
level = it.get("level")
|
|
381
|
+
s, _basis = docindex.sensitivity_for(it.get("path") or "", sensitivity)
|
|
382
|
+
sens[s] = sens.get(s, 0) + 1
|
|
383
|
+
cg.add(
|
|
384
|
+
nid, docindex.render(it),
|
|
385
|
+
layer=(layer_of(nid) if layer_of else None) or layer or "knowledge",
|
|
386
|
+
tags=["doc", "doc:md", f"level:{level}",
|
|
387
|
+
"domain:" + _domain_of(it)],
|
|
388
|
+
condition_space=docindex.condition_space(it),
|
|
389
|
+
verification_basis="data",
|
|
390
|
+
sensitivity=s,
|
|
391
|
+
doc_ref=_doc_ref(it, root),
|
|
392
|
+
)
|
|
393
|
+
else:
|
|
394
|
+
raise ValueError(f"未知 ref kind:{kind!r}")
|
|
395
|
+
ids.append(nid)
|
|
396
|
+
return ids, sens
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
# 生效条件:把入参 root 原样写入返回 dict 的 'root',path/name/kind/lineno/end/lang/hash 按 it.get 取值(缺省 None),precise 取 bool(it.get('precise', True)),render_version 取传入值(传入 None 时延迟 import codeindex 取 codeindex.RENDER_VERSION,保证与 render 契约**同源**、无第二处硬编码)。
|
|
400
|
+
def _code_ref(it: dict, root: str, render_version=None) -> dict:
|
|
401
|
+
if render_version is None: # 直接调用点的兜底:与 render 产物同源
|
|
402
|
+
from . import codeindex
|
|
403
|
+
render_version = codeindex.RENDER_VERSION
|
|
404
|
+
return {
|
|
405
|
+
"path": it.get("path"), "name": it.get("name"),
|
|
406
|
+
"kind": it.get("kind"), "lineno": it.get("lineno"), "end": it.get("end"),
|
|
407
|
+
"lang": it.get("lang"), "precise": bool(it.get("precise", True)),
|
|
408
|
+
"hash": it.get("hash"), "root": root,
|
|
409
|
+
"render_version": render_version,
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
# 生效条件:把入参 root 原样写入返回 dict 的 'root',path/heading/heading_path/level/lineno/end/anchor/hash/lang 按 it.get 取值(缺省 None),precise 取 bool(it.get('precise', True))。
|
|
414
|
+
def _doc_ref(it: dict, root: str) -> dict:
|
|
415
|
+
return {
|
|
416
|
+
"path": it.get("path"), "heading": it.get("heading"),
|
|
417
|
+
"heading_path": it.get("heading_path"), "level": it.get("level"),
|
|
418
|
+
"lineno": it.get("lineno"), "end": it.get("end"),
|
|
419
|
+
"anchor": it.get("anchor"), "hash": it.get("hash"), "lang": it.get("lang"),
|
|
420
|
+
"precise": bool(it.get("precise", True)), "root": root,
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
# --------------------------------------------------------------------------
|
|
425
|
+
# 回读(唯一实现:op=ref 与 check_refs 共用)
|
|
426
|
+
# --------------------------------------------------------------------------
|
|
427
|
+
|
|
428
|
+
# 生效条件:传入 ref 为假值(如 None/{})时按 {} 处理,rel 取 ref.get("path") or "";root 与 ref.get("root") 均为假值时返回含 ref/path/status:"unresolved"/ok:False/error:"ref 未记录 root..." 的 out;否则用 root or ref.get("root") 与 rel 拼 fp,os.path.isfile(fp) 为假时返回 status:"dangling"、stale:True,读取抛 OSError/UnicodeDecodeError 时返回 status:"error";读取成功时 lineno 取 int(ref.get("lineno") or 1)(假值回落 1)、end 取 int(ref.get("end") or lineno)(假值回落 lineno),ref.get("hash") 为 None 时 match=None、ok=True、status:"ok",ref.get("hash") 为真值且等于 region_hash 时 ok=True/status:"ok"、不等时 ok=False/status:"stale",ref.get("hash") 为假值但非 None(如 ""/0/False)时 ok=False/status:"stale";with_text 为真时 out["text"] 取 lines[max(0,lineno-1):max(max(0,lineno-1),end)] 的 join;
|
|
429
|
+
def probe_ref(ref: dict, *, root: str = None, with_text: bool = False) -> dict:
|
|
430
|
+
"""只读探测单个 ref 的状态(不回读整篇,除非 with_text)。"""
|
|
431
|
+
from . import codeindex
|
|
432
|
+
ref = ref or {}
|
|
433
|
+
rel = ref.get("path") or ""
|
|
434
|
+
r = root or ref.get("root") or ""
|
|
435
|
+
base = {"ref": ref, "path": rel, "status": "unresolved", "ok": False}
|
|
436
|
+
if not r:
|
|
437
|
+
return {**base, "error": "ref 未记录 root,请显式传 root 参数"
|
|
438
|
+
"(索引里存的是相对 root 的 path)"}
|
|
439
|
+
fp = os.path.join(r, rel)
|
|
440
|
+
base["root"] = r
|
|
441
|
+
if not os.path.isfile(fp):
|
|
442
|
+
return {**base, "status": "dangling", "abspath": fp, "stale": True,
|
|
443
|
+
"error": f"源文件不存在(索引已悬空):{fp}"}
|
|
444
|
+
try:
|
|
445
|
+
with open(fp, "r", encoding="utf-8") as f:
|
|
446
|
+
lines = f.read().split("\n")
|
|
447
|
+
except (OSError, UnicodeDecodeError) as exc:
|
|
448
|
+
return {**base, "status": "error", "abspath": fp,
|
|
449
|
+
"error": f"读取失败:{exc}"}
|
|
450
|
+
total = len(lines)
|
|
451
|
+
lineno = int(ref.get("lineno") or 1)
|
|
452
|
+
end = int(ref.get("end") or lineno)
|
|
453
|
+
got = codeindex.region_hash(lines, lineno, end)
|
|
454
|
+
expect = ref.get("hash")
|
|
455
|
+
match = (got == expect) if expect else None
|
|
456
|
+
out = {
|
|
457
|
+
**base, "abspath": fp, "total_lines": total,
|
|
458
|
+
"hash": got, "hash_expected": expect, "hash_match": match,
|
|
459
|
+
"stale": bool(expect) and not match,
|
|
460
|
+
"ok": expect is None or bool(match),
|
|
461
|
+
"status": "ok" if (expect is None or match) else "stale",
|
|
462
|
+
}
|
|
463
|
+
if with_text:
|
|
464
|
+
lo = max(0, lineno - 1)
|
|
465
|
+
out["text"] = "\n".join(lines[lo:max(lo, end)])
|
|
466
|
+
return out
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
# 生效条件:ref 经 probe_ref(root=root, with_text=True) 后 status 为 'ok'/'stale' 时返回 ok=True 及 text/total_lines/hash/hash_match/stale/precise,status 为 'unresolved'/'error'/'dangling' 时返回 ok=False 与 error。
|
|
470
|
+
def read_ref(ref: dict, *, root: str = None, ref_kind: str = "ref") -> dict:
|
|
471
|
+
"""按 ref 回读源区间——`op=ref` 与 `check_refs` 的唯一实现。"""
|
|
472
|
+
p = probe_ref(ref, root=root, with_text=True)
|
|
473
|
+
base = {"ref": ref, "ref_kind": ref_kind}
|
|
474
|
+
if p["status"] in ("unresolved", "error", "dangling"):
|
|
475
|
+
out = {**base, "ok": False, "error": p["error"]}
|
|
476
|
+
if p["status"] == "dangling":
|
|
477
|
+
out["stale"] = True
|
|
478
|
+
return out
|
|
479
|
+
return {
|
|
480
|
+
**base, "ok": True, "text": p["text"], "total_lines": p["total_lines"],
|
|
481
|
+
"hash": p["hash"], "hash_expected": p["hash_expected"],
|
|
482
|
+
"hash_match": p["hash_match"], "stale": p["stale"],
|
|
483
|
+
"precise": bool((ref or {}).get("precise", True)),
|
|
484
|
+
"note": "按 ref 区间回读;hash_match=False 说明源已改动,"
|
|
485
|
+
"重跑 index_code / index_doc 重建",
|
|
486
|
+
}
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
# 生效条件:cg 的 index['nodes'] 非空时汇总 stale/dangling/unresolved/errors 并返回 ok =(无 stale 且无 dangling);only_tagged 为真时只探测 tags 含 'code'/'doc' 的节点,ledger 非空时先走 (size, mtime) 快路径。
|
|
490
|
+
def check_refs(cg, *, ledger: "Ledger" = None, max_nodes: int = MAX_CHECK,
|
|
491
|
+
only_tagged: bool = True) -> dict:
|
|
492
|
+
"""漂移 / 悬空巡检(只读、不抛)。
|
|
493
|
+
|
|
494
|
+
优先走 ledger 的 (size, mtime) 快路径:未变文件**不读盘**直接判 ok;
|
|
495
|
+
变了的文件读一次、按记录区间重算哈希判 stale。
|
|
496
|
+
ledger 覆盖不到的节点(早期索引 / 未开增量)再回退逐节点探测。
|
|
497
|
+
|
|
498
|
+
`only_tagged=True`(默认)只探测带 `code` / `doc` 标签的节点——ref 只由
|
|
499
|
+
`index_code` / `index_doc` 产生,两者都会打这两个标签;这样巡检不必为每条
|
|
500
|
+
记忆节点都读一次盘。需要穷举(含手写 ref)时传 False。
|
|
501
|
+
"""
|
|
502
|
+
nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
|
|
503
|
+
stale, dangling, unresolved, errors = [], [], [], []
|
|
504
|
+
covered = set()
|
|
505
|
+
|
|
506
|
+
# 生效条件:当 nid、ref、kind、rel 传入时,p = probe_ref(ref) 后按 p["status"] 分派:为 "dangling" 时把含 node_id/ref_kind/path/lineno/end/error 的 row 加入 dangling,为 "stale" 时补 hash_expected/hash 加入 stale,为 "unresolved" 时加入 unresolved,为 "error" 时加入 errors;其他状态不加入;
|
|
507
|
+
def _probe_one(nid, ref, kind, rel):
|
|
508
|
+
p = probe_ref(ref)
|
|
509
|
+
row = {"node_id": nid, "ref_kind": kind, "path": rel,
|
|
510
|
+
"lineno": ref.get("lineno"), "end": ref.get("end"),
|
|
511
|
+
"error": p.get("error", "")}
|
|
512
|
+
if p["status"] == "dangling":
|
|
513
|
+
dangling.append(row)
|
|
514
|
+
elif p["status"] == "stale":
|
|
515
|
+
row["hash_expected"] = p.get("hash_expected")
|
|
516
|
+
row["hash"] = p.get("hash")
|
|
517
|
+
stale.append(row)
|
|
518
|
+
elif p["status"] == "unresolved":
|
|
519
|
+
unresolved.append(row)
|
|
520
|
+
elif p["status"] == "error":
|
|
521
|
+
errors.append(row)
|
|
522
|
+
|
|
523
|
+
# 快路径:ledger 记录的文件(键 = 源文件绝对路径,天然跨大域不撞名)
|
|
524
|
+
if ledger is not None:
|
|
525
|
+
for key, e in sorted((ledger.load().get("files") or {}).items()):
|
|
526
|
+
rec_nodes = [n for n in (e.get("nodes") or [])
|
|
527
|
+
if n.get("id") in nodes]
|
|
528
|
+
if not rec_nodes:
|
|
529
|
+
continue
|
|
530
|
+
covered.update(n.get("id") for n in rec_nodes)
|
|
531
|
+
rel = e.get("path") or ""
|
|
532
|
+
# 必须用**每个源文件自己的 root**(索引时的源大域),不能用 ledger.root:
|
|
533
|
+
# 后者是认知图根,拿它拼路径会指向不存在的位置、把一切都误判成 dangling。
|
|
534
|
+
src_root = e.get("root") or os.path.dirname(key)
|
|
535
|
+
try:
|
|
536
|
+
st = os.stat(key)
|
|
537
|
+
unchanged = (e.get("size") == st.st_size
|
|
538
|
+
and abs(float(e.get("mtime") or 0.0) - st.st_mtime) < 1e-6)
|
|
539
|
+
except OSError:
|
|
540
|
+
unchanged = False
|
|
541
|
+
if not unchanged:
|
|
542
|
+
kind = e.get("kind") or kind_of_path(rel)
|
|
543
|
+
for n in rec_nodes:
|
|
544
|
+
_probe_one(n.get("id"), {**n, "path": rel, "root": src_root},
|
|
545
|
+
kind, rel)
|
|
546
|
+
|
|
547
|
+
# 回退:ledger 未覆盖的索引节点
|
|
548
|
+
# 生效条件:当 nid 传入时,若 only_tagged 为假值立即返回 True;否则取 (nodes.get(nid) or {}).get("tags") or [],仅当其中存在 "code" 或 "doc" 返回 True,否则返回 False;
|
|
549
|
+
def _candidate(nid):
|
|
550
|
+
if not only_tagged:
|
|
551
|
+
return True
|
|
552
|
+
tags = (nodes.get(nid) or {}).get("tags") or []
|
|
553
|
+
return any(t in ("code", "doc") for t in tags)
|
|
554
|
+
|
|
555
|
+
todo = [nid for nid in nodes if nid not in covered and _candidate(nid)]
|
|
556
|
+
truncated = len(todo) > max_nodes
|
|
557
|
+
checked = 0
|
|
558
|
+
for nid in todo[:max_nodes]:
|
|
559
|
+
try:
|
|
560
|
+
node = cg.get(nid)
|
|
561
|
+
except Exception:
|
|
562
|
+
continue
|
|
563
|
+
if not node:
|
|
564
|
+
continue
|
|
565
|
+
kind, ref = ref_of(node)
|
|
566
|
+
if not ref:
|
|
567
|
+
continue
|
|
568
|
+
checked += 1
|
|
569
|
+
try:
|
|
570
|
+
_probe_one(nid, ref, kind, ref.get("path") or "")
|
|
571
|
+
except Exception as exc: # 巡检不抛
|
|
572
|
+
errors.append({"node_id": nid, "ref_kind": kind, "error": str(exc)})
|
|
573
|
+
|
|
574
|
+
return {
|
|
575
|
+
"ok": not (stale or dangling),
|
|
576
|
+
"status": "ok" if not (stale or dangling) else ("dangling" if dangling else "stale"),
|
|
577
|
+
"checked": checked + len(covered),
|
|
578
|
+
"ledger_files": len((ledger.load().get("files") or {})) if ledger else 0,
|
|
579
|
+
"stale": stale, "dangling": dangling,
|
|
580
|
+
"unresolved": unresolved, "errors": errors,
|
|
581
|
+
"truncated": truncated, "max_nodes": max_nodes,
|
|
582
|
+
}
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
# --------------------------------------------------------------------------
|
|
586
|
+
# 节点级对账:孤儿清退 + 悬空清退
|
|
587
|
+
#
|
|
588
|
+
# 水位层的 `Ledger.reconcile` 只剪**水位条目**、`prune` 只剪「源大域已消失」
|
|
589
|
+
# 的条目,两者都不碰**节点**。于是节点层的两类残留无人处置:
|
|
590
|
+
# · 孤儿(同一文档的过期代)——标题路径一变 id 全量重算,旧代与新代并存;
|
|
591
|
+
# · 悬空(源文件已删)——回读必然失败,巡检永远报 dangling。
|
|
592
|
+
# 本节的唯一实现同时供 `op=index_code / index_doc`(孤儿)与
|
|
593
|
+
# `op=ref action=prune`(悬空)使用,避免两处各写一套口径。
|
|
594
|
+
# --------------------------------------------------------------------------
|
|
595
|
+
|
|
596
|
+
# 生效条件:返回 os.path.normcase(os.path.abspath(str(p or ''))),即 p 为 None/空串时返回当前目录的归一绝对路径。
|
|
597
|
+
def _norm_root(p) -> str:
|
|
598
|
+
"""root 归一:同一目录的大小写/分隔符差异不得影响「同一大域」判定。"""
|
|
599
|
+
return os.path.normcase(os.path.abspath(str(p or "")))
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
# 生效条件:a 与 b 都非空且 _norm_root(a) == _norm_root(b) 时返回 True,否则(含 TypeError/ValueError)返回 False。
|
|
603
|
+
def _same_root(a, b) -> bool:
|
|
604
|
+
try:
|
|
605
|
+
return bool(a) and bool(b) and _norm_root(a) == _norm_root(b)
|
|
606
|
+
except (TypeError, ValueError):
|
|
607
|
+
return False
|
|
608
|
+
|
|
609
|
+
|
|
610
|
+
# 生效条件:返回 str(p or '').replace('\\', '/').lstrip('./'),即 p 为 None/空串时返回 ''。
|
|
611
|
+
def _norm_rel(p) -> str:
|
|
612
|
+
return str(p or "").replace("\\", "/").lstrip("./")
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
# 生效条件:cg 具备可调用的 forget 方法时对 plan 中每个 nid 调 cg.forget(nid, why),返回 (成功 id 列表, 被拦下/失败的 {node_id, error} 列表);cg 无 forget 时返回 ([], plan 中每 nid 一条错误)。
|
|
616
|
+
def _forget_many(cg, plan, why: str) -> tuple:
|
|
617
|
+
"""逐条软删(进 trash/、写删除清单、可 restore);受保护节点拦下不删。
|
|
618
|
+
|
|
619
|
+
`forget` 属 **MdCGOS 层**能力(保护裁决 + 回收站 + 删除清单),基础层
|
|
620
|
+
`MdCG` 没有任何删除原语。缺能力时**明确报错、不静默跳过**——否则
|
|
621
|
+
「清退了 N 条」看着成功、实际一条没删(P27 §10 实测过这个坑)。
|
|
622
|
+
"""
|
|
623
|
+
fn = getattr(cg, "forget", None)
|
|
624
|
+
if not callable(fn):
|
|
625
|
+
err = (f"{type(cg).__name__} 无 forget 能力(对账须能删除节点);"
|
|
626
|
+
f"生产路径是 MdCGOS,测试请用 MdCGOS")
|
|
627
|
+
return [], [{"node_id": nid, "error": err} for nid in sorted(plan)]
|
|
628
|
+
done, blocked = [], []
|
|
629
|
+
for nid in sorted(plan):
|
|
630
|
+
try:
|
|
631
|
+
res = fn(nid, why) or {}
|
|
632
|
+
except Exception as exc: # ProtectionError 等 → 拦下,不越权
|
|
633
|
+
blocked.append({"node_id": nid, "error": str(exc)[:120]})
|
|
634
|
+
continue
|
|
635
|
+
if res.get("ok"):
|
|
636
|
+
done.append(nid)
|
|
637
|
+
else:
|
|
638
|
+
blocked.append({"node_id": nid, "error": str(res.get("error"))[:120]})
|
|
639
|
+
return done, blocked
|
|
640
|
+
|
|
641
|
+
|
|
642
|
+
# 生效条件:cg 具备可调用的 _unstage 时对 ghosts 逐个调用并收集成功 id(单条异常跳过),cg 无该能力时返回 [](不假装成功)。
|
|
643
|
+
def _drop_ghosts(cg, ghosts) -> list:
|
|
644
|
+
"""摘除幽灵条目的索引记录(节点文件已不存在,没有可软删的实体)。
|
|
645
|
+
|
|
646
|
+
走 `MdCG._unstage`:它顺带落删除记录,保证幽灵不会再次从分片日志里复活。
|
|
647
|
+
基础层没有该能力时如实返回空列表(不假装成功)。
|
|
648
|
+
"""
|
|
649
|
+
drop = getattr(cg, "_unstage", None)
|
|
650
|
+
if not callable(drop):
|
|
651
|
+
return []
|
|
652
|
+
out = []
|
|
653
|
+
for nid in ghosts:
|
|
654
|
+
try:
|
|
655
|
+
drop(nid)
|
|
656
|
+
out.append(nid)
|
|
657
|
+
except Exception: # 单条失败不拖垮整批
|
|
658
|
+
continue
|
|
659
|
+
return out
|
|
660
|
+
|
|
661
|
+
|
|
662
|
+
# 生效条件:items 中同 kind 的节点若其 ref['path'] 命中本次 items 的文件、id 不在本次产出内且 ref['root'] 与入参 root 同一(_same_root),则列入清退计划;dry_run 为真只返回计划,否则经 _forget_many 软删;items 为空时返回 count 0。
|
|
663
|
+
def prune_orphans(cg, *, kind: str, root: str, items, dry_run: bool = False,
|
|
664
|
+
reason: str = "") -> dict:
|
|
665
|
+
"""清退「同 root + 同 path,但已不在本次产出里」的**过期代**节点。
|
|
666
|
+
|
|
667
|
+
为什么必须有:`node_id = sha1(相对path + "#" + heading_path)`,而
|
|
668
|
+
`add_items` 只做**同 id 幂等 upsert**——文档标题结构一变,整篇 id 全量
|
|
669
|
+
重算,旧代节点无人清退,与新代并存(同一文档召回两份,且旧代引用的区间
|
|
670
|
+
已失效)。截断的索引由调用方负责不调用本函数(没扫完 ≠ 剩下的都过期)。
|
|
671
|
+
|
|
672
|
+
范围**只限本次真正重切过的文件**(`items` 的 path):增量索引跳过的未变
|
|
673
|
+
文件不在 items 里,其节点不进判定——否则会把完好的节点整片误删。
|
|
674
|
+
"""
|
|
675
|
+
nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
|
|
676
|
+
touched: dict = {}
|
|
677
|
+
for it in items or []:
|
|
678
|
+
rel = _norm_rel(it.get("path"))
|
|
679
|
+
if rel:
|
|
680
|
+
touched.setdefault(rel, set()).add(node_id_of(it, kind))
|
|
681
|
+
base = {"scanned": len(nodes), "touched_files": len(touched),
|
|
682
|
+
"dry_run": bool(dry_run)}
|
|
683
|
+
if not touched:
|
|
684
|
+
return {**base, "ok": True, "count": 0, "pruned": [],
|
|
685
|
+
"skipped_protected": []}
|
|
686
|
+
|
|
687
|
+
# ⚠ ref 只存在于**节点 frontmatter**里;`cg.index['nodes']` 是元数据快照
|
|
688
|
+
# (path/layer/tags/…,见 mdcg._scan_nodes),**不含 ref**。因此必须
|
|
689
|
+
# `cg.get(nid)` 取回节点再 ref_of —— 否则 ref_of 恒返回 ('', None)、
|
|
690
|
+
# 整个对账静默失效(P27 §10 实测过这个坑)。标签预筛与 check_refs 同口径,
|
|
691
|
+
# 免得为全库每条记忆都读一次盘。
|
|
692
|
+
tag = "doc" if kind == "doc_ref" else "code"
|
|
693
|
+
plan = []
|
|
694
|
+
for nid, e in nodes.items():
|
|
695
|
+
if tag not in ((e or {}).get("tags") or []):
|
|
696
|
+
continue
|
|
697
|
+
try:
|
|
698
|
+
node = cg.get(nid)
|
|
699
|
+
except Exception:
|
|
700
|
+
continue
|
|
701
|
+
if not node:
|
|
702
|
+
continue
|
|
703
|
+
k, ref = ref_of(node)
|
|
704
|
+
if k != kind or not isinstance(ref, dict):
|
|
705
|
+
continue
|
|
706
|
+
keep = touched.get(_norm_rel(ref.get("path")))
|
|
707
|
+
if keep is None or nid in keep or not _same_root(ref.get("root"), root):
|
|
708
|
+
continue
|
|
709
|
+
plan.append(nid)
|
|
710
|
+
|
|
711
|
+
why = reason or ("索引重建:本节点已不在同文档新代产出中(标题路径变更致 "
|
|
712
|
+
"node_id 重算),清退过期代以消除重复召回")
|
|
713
|
+
if dry_run:
|
|
714
|
+
return {**base, "ok": True, "count": len(plan), "pruned": sorted(plan)[:50],
|
|
715
|
+
"skipped_protected": [], "reason": why}
|
|
716
|
+
done, blocked = _forget_many(cg, plan, why)
|
|
717
|
+
return {**base, "ok": True, "count": len(done), "pruned": done[:50],
|
|
718
|
+
"skipped_protected": blocked[:20], "reason": why}
|
|
719
|
+
|
|
720
|
+
|
|
721
|
+
# 生效条件:cg 中带 'code'/'doc' 标签且 ref 带 root 的节点经 probe_ref 判为 'dangling' 时列入清退计划;only_roots 为空时另收集 cg.root 下取不到对应节点 .md 的幽灵条目经 _drop_ghosts 摘除;dry_run 为真只返回计划。
|
|
722
|
+
def prune_dangling(cg, *, only_roots=None, dry_run: bool = False,
|
|
723
|
+
max_nodes: int = MAX_CHECK, reason: str = "") -> dict:
|
|
724
|
+
"""清退**悬空**节点:ref 指向的源文件已删除,回读必然失败。
|
|
725
|
+
|
|
726
|
+
`check_refs` 只报告不处置(其原话是「悬空需人工处置」),本函数就是那个
|
|
727
|
+
出口——已删脚本、被搬走的文档留下的残留节点一次清掉,而不是逐条手工
|
|
728
|
+
`forget`。判定与巡检共用 `probe_ref` 的唯一实现,口径不会打架。
|
|
729
|
+
|
|
730
|
+
另清**幽灵条目**(ghosts):索引有条目、节点文件却不存在。它们是历史
|
|
731
|
+
「删除只摘内存索引、不落盘」的遗留——`cg.get` 取不回 → 悬空清退够不着它,
|
|
732
|
+
而 `check_refs` 走 ledger 会一直报 → dangling 永不归零。判据只用唯一真源
|
|
733
|
+
(节点 .md 不存在即脏索引),与「索引是派生物」的宣言一致;`only_roots`
|
|
734
|
+
非空时跳过(幽灵条目无 ref,无法归因到某个源大域)。
|
|
735
|
+
"""
|
|
736
|
+
nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
|
|
737
|
+
ghosts = []
|
|
738
|
+
if not only_roots:
|
|
739
|
+
for nid, e in list(nodes.items()):
|
|
740
|
+
tags = (e or {}).get("tags") or []
|
|
741
|
+
if not any(t in ("code", "doc") for t in tags):
|
|
742
|
+
continue
|
|
743
|
+
path = (e or {}).get("path") or ""
|
|
744
|
+
if path and not os.path.exists(os.path.join(cg.root, path)):
|
|
745
|
+
ghosts.append(nid)
|
|
746
|
+
# 同 prune_orphans:ref 只在节点 frontmatter 里,索引条目里没有,
|
|
747
|
+
# 必须 cg.get 取回节点再 ref_of(否则恒空、静默不删)。
|
|
748
|
+
todo = []
|
|
749
|
+
for nid, e in nodes.items():
|
|
750
|
+
tags = (e or {}).get("tags") or []
|
|
751
|
+
if not any(t in ("code", "doc") for t in tags):
|
|
752
|
+
continue
|
|
753
|
+
try:
|
|
754
|
+
node = cg.get(nid)
|
|
755
|
+
except Exception:
|
|
756
|
+
continue
|
|
757
|
+
if not node:
|
|
758
|
+
continue
|
|
759
|
+
_k, ref = ref_of(node)
|
|
760
|
+
if not ref or not ref.get("root"):
|
|
761
|
+
continue
|
|
762
|
+
if only_roots and not any(_same_root(ref.get("root"), r) for r in only_roots):
|
|
763
|
+
continue
|
|
764
|
+
todo.append((nid, ref))
|
|
765
|
+
truncated = len(todo) > max_nodes or len(ghosts) > max_nodes
|
|
766
|
+
plan = []
|
|
767
|
+
for nid, ref in todo[:max_nodes]:
|
|
768
|
+
try:
|
|
769
|
+
if probe_ref(ref).get("status") == "dangling":
|
|
770
|
+
plan.append(nid)
|
|
771
|
+
except Exception: # 探测失败不算悬空(宁可不删)
|
|
772
|
+
continue
|
|
773
|
+
|
|
774
|
+
why = reason or ("源文件已删除,索引节点悬空(回读必然失败),"
|
|
775
|
+
"清退以消除永不消失的 dangling")
|
|
776
|
+
ghost_plan = sorted(ghosts)[:max_nodes]
|
|
777
|
+
base = {"scanned": len(nodes), "candidates": len(plan),
|
|
778
|
+
"ghosts": len(ghost_plan),
|
|
779
|
+
"dry_run": bool(dry_run), "truncated": truncated,
|
|
780
|
+
"max_nodes": max_nodes}
|
|
781
|
+
if dry_run:
|
|
782
|
+
return {**base, "ok": True, "count": len(plan), "pruned": sorted(plan)[:50],
|
|
783
|
+
"ghost_pruned": ghost_plan[:50],
|
|
784
|
+
"skipped_protected": [], "reason": why}
|
|
785
|
+
done, blocked = _forget_many(cg, plan, why)
|
|
786
|
+
dropped = _drop_ghosts(cg, ghost_plan)
|
|
787
|
+
return {**base, "ok": True, "count": len(done), "pruned": done[:50],
|
|
788
|
+
"ghost_pruned": dropped[:50],
|
|
789
|
+
"skipped_protected": blocked[:20], "reason": why}
|
|
790
|
+
|
|
791
|
+
|
|
792
|
+
# 生效条件:cg 节点按 ref['root'] 与 kind 分组后逐组以 index_dir(incremental=False, ledger=ledger) 重切、再以 add_items(layer_of=原 layer) 重建,返回 {'ok','roots','groups','indexed','errors','truncated'};only_roots 非 None 时只处理其中列出的 root。
|
|
793
|
+
def rebuild(cg, *, ledger: "Ledger" = None, only_roots=None, max_files: int = 500,
|
|
794
|
+
max_items: int = 2000) -> dict:
|
|
795
|
+
"""按 ref 记录的 root 重建索引(sustain.heal 的修复动作)。
|
|
796
|
+
|
|
797
|
+
只重跑出了问题的 root(`only_roots`),逐节点**保留原 layer**;
|
|
798
|
+
doc 密级用默认策略重算(默认只可能更严,不会放松)。
|
|
799
|
+
"""
|
|
800
|
+
nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
|
|
801
|
+
groups = {}
|
|
802
|
+
for nid in nodes:
|
|
803
|
+
try:
|
|
804
|
+
node = cg.get(nid)
|
|
805
|
+
except Exception:
|
|
806
|
+
continue
|
|
807
|
+
kind, ref = ref_of(node)
|
|
808
|
+
if not ref or not ref.get("root"):
|
|
809
|
+
continue
|
|
810
|
+
r = ref["root"]
|
|
811
|
+
if only_roots is not None and r not in only_roots:
|
|
812
|
+
continue
|
|
813
|
+
groups.setdefault((r, kind), 0)
|
|
814
|
+
groups[(r, kind)] += 1
|
|
815
|
+
|
|
816
|
+
out = {"ok": True, "roots": sorted({r for r, _ in groups}),
|
|
817
|
+
"groups": len(groups), "indexed": 0, "errors": [], "truncated": False}
|
|
818
|
+
for (root, kind) in sorted(groups):
|
|
819
|
+
try:
|
|
820
|
+
items, errors, stats = index_dir(
|
|
821
|
+
root, kind=kind, max_files=max_files, max_items=max_items,
|
|
822
|
+
incremental=False, ledger=ledger)
|
|
823
|
+
ids, _sens = add_items(
|
|
824
|
+
cg, items, kind=kind, root=root,
|
|
825
|
+
layer_of=lambda nid: (nodes.get(nid) or {}).get("layer"))
|
|
826
|
+
out["indexed"] += len(ids)
|
|
827
|
+
out["errors"].extend(errors)
|
|
828
|
+
out["truncated"] = out["truncated"] or bool(stats.get("truncated"))
|
|
829
|
+
except Exception as exc: # 自愈不抛
|
|
830
|
+
out["ok"] = False
|
|
831
|
+
out["errors"].append(f"{root} [{kind}]:{exc}")
|
|
832
|
+
if ledger is not None:
|
|
833
|
+
ledger.save()
|
|
834
834
|
return out
|