@furongjun1999/dsh-memory 0.4.11 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +552 -465
- package/codebuddy/CODEBUDDY.md +11 -3
- package/codebuddy/README.md +92 -90
- package/codebuddy/mcp.json +27 -27
- package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
- package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
- package/docs/README.md +143 -111
- package/docs/discipline/harnesses.yaml +244 -226
- package/docs/discipline/templates/full.md.tmpl +61 -61
- package/docs/discipline/templates/rules.mdc.tmpl +68 -0
- package/docs/discipline/templates/skill.md.tmpl +23 -23
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
- package/docs/eval/bench_lingshu_self/bench_self.py +140 -0
- package/docs/eval/bench_lingshu_self/self_bench_result.json +404 -0
- package/docs/eval/bench_lingshu_self//347/201/265/346/236/242/350/207/252/345/272/223/347/253/257/345/210/260/347/253/257/346/243/200/347/264/242/345/256/236/346/265/213_v1.0.md +38 -0
- package/docs/eval//344/270/215/345/217/257/351/235/240/346/200/247/350/220/275/345/234/260_P0_v1.0.md +185 -0
- package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
- package/docs/eval//346/225/205/351/232/234/346/263/250/345/205/245/345/256/236/346/265/213_v1.0.md +422 -0
- package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
- package/docs/eval//347/253/257/345/210/260/347/253/257LoCoMoQA/345/220/214/345/217/243/345/276/204/345/257/271/347/205/247_v1.0.md +100 -0
- package/docs/eval//347/253/257/345/210/260/347/253/257/345/271/262/346/211/260/346/261/240/350/257/204/346/265/213_/347/241/256/345/256/232/346/200/247/350/243/201/345/206/263vsLLM_judge_v1.1.md +197 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v1.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v10.md +210 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v11.md +227 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v12.md +203 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v13.md +233 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v14.md +191 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v15.md +213 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v16.md +214 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v17.md +199 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v2.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v3.md +152 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v4.md +128 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v5.md +114 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v6.md +192 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v7.md +187 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v8.md +207 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v9.md +203 -0
- package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
- package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
- package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
- package/docs/hive//345/256/211/345/205/250/345/256/241/350/256/241/345/256/236/351/224/232_v0.1.md +202 -0
- package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
- package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
- package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
- package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +519 -503
- package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
- package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
- package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
- package/docs/images/lingshu-moonlight-covenant-poster-preview.jpg +0 -0
- package/docs/images/lingshu-moonlight-covenant-poster.png +0 -0
- package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
- package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
- package/docs/mdcg/lingshu_tutorial.html +14449 -14449
- package/docs/mdcg/release_v0.4.11.md +49 -0
- package/docs/mdcg/release_v0.4.5.md +55 -55
- package/docs/mdcg/tool_table_v0.3.0.md +117 -117
- package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
- package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
- package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
- package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
- package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
- package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
- package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
- package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
- package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
- package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
- package/docs/theory//344/270/215/345/217/257/351/235/240/345/256/232/347/220/206/344/270/216/345/244/261/346/225/210/344/274/230/345/205/210/346/241/206/346/236/266_v0.3.md +365 -0
- package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
- package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
- package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
- package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
- package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
- package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
- package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
- package/docs//344/270/215/345/217/257/351/235/240/346/200/247/347/220/206/350/256/272_v0.1.md +343 -0
- package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
- package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
- package/dsh/README.md +82 -82
- package/dsh/cordis.yml.example +139 -139
- package/dsh/update-lingshu.bat +11 -11
- package/lib/bridge.d.ts +9 -0
- package/lib/bridge.js +35 -0
- package/lib/hooks.js +36 -2
- package/lib/index.js +7 -1
- package/lib/lib/roleplay_web.js +116 -29
- package/lib/lib/token_store.d.ts +7 -1
- package/lib/lib/token_store.js +12 -3
- package/md_cg/__init__.py +7 -7
- package/md_cg/audit.py +379 -368
- package/md_cg/autonomy.py +287 -287
- package/md_cg/backfill.py +1328 -1327
- package/md_cg/backfill_bigdomain.py +34 -34
- package/md_cg/backfill_bucket_zh.py +35 -0
- package/md_cg/bench6_arms.py +410 -410
- package/md_cg/bench6_common.py +230 -230
- package/md_cg/bench6_competitors.py +212 -212
- package/md_cg/bench_axis_domain.py +257 -257
- package/md_cg/bench_blind_comp.py +308 -308
- package/md_cg/bench_e2e_judge.py +532 -0
- package/md_cg/bench_e2e_locomo_qa.py +368 -0
- package/md_cg/bench_e2e_qa.py +256 -0
- package/md_cg/bench_en_atoms_public.py +230 -230
- package/md_cg/bench_governance.py +348 -348
- package/md_cg/bench_lme_zh.py +410 -410
- package/md_cg/bench_locomo.py +121 -121
- package/md_cg/bench_locomo_zh.py +450 -450
- package/md_cg/bench_locomo_zh_public.py +147 -147
- package/md_cg/bench_longmem.py +112 -112
- package/md_cg/bench_membench.py +632 -632
- package/md_cg/bench_p0.py +149 -149
- package/md_cg/bench_progressive.py +287 -287
- package/md_cg/bench_role_views.py +238 -238
- package/md_cg/bench_task_ab.py +243 -243
- package/md_cg/bench_task_ab_llm.py +408 -408
- package/md_cg/bench_unified_en.py +204 -204
- package/md_cg/bench_zh_mad.py +601 -601
- package/md_cg/blindspot_tickets.py +123 -123
- package/md_cg/branches.py +301 -285
- package/md_cg/build_postings.py +73 -73
- package/md_cg/ccgc.py +1006 -948
- package/md_cg/census.py +132 -132
- package/md_cg/chain.py +315 -300
- package/md_cg/codeindex.py +531 -531
- package/md_cg/coldverify.py +292 -292
- package/md_cg/comment_gate.py +337 -337
- package/md_cg/cond_compose.py +190 -190
- package/md_cg/cond_facts.py +154 -154
- package/md_cg/cond_template.json +106 -106
- package/md_cg/condition_anchor.py +142 -142
- package/md_cg/conformance.py +726 -726
- package/md_cg/consistency.py +717 -717
- package/md_cg/consolidate.py +1537 -1439
- package/md_cg/corpus.py +110 -110
- package/md_cg/crosscheck.py +1098 -1097
- package/md_cg/crypto.py +3 -1
- package/md_cg/d_meta.py +310 -310
- package/md_cg/datapath.py +78 -18
- package/md_cg/docindex.py +473 -473
- package/md_cg/eval_common.py +575 -575
- package/md_cg/evidence.py +4 -2
- package/md_cg/evolution.py +477 -477
- package/md_cg/export.py +222 -220
- package/md_cg/forgetting.py +581 -581
- package/md_cg/fsutil.py +377 -329
- package/md_cg/hotcache.py +48 -7
- package/md_cg/hyperedge.py +251 -251
- package/md_cg/identity.py +390 -390
- package/md_cg/insight.py +500 -500
- package/md_cg/interop.py +338 -0
- package/md_cg/judgment_manifest.py +177 -0
- package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
- package/md_cg/lexicon/build_standard_en.py +171 -171
- package/md_cg/lexicon/expand_en_zh.py +211 -211
- package/md_cg/lifecycle.py +272 -272
- package/md_cg/linkref.py +280 -280
- package/md_cg/links.py +140 -107
- package/md_cg/mcp_server.py +403 -55
- package/md_cg/md_whitebox.py +345 -345
- package/md_cg/mdcg.py +783 -233
- package/md_cg/mdcos.py +500 -70
- package/md_cg/metacognition.py +591 -591
- package/md_cg/migrate.py +119 -119
- package/md_cg/migrate_aeis.py +221 -221
- package/md_cg/migrate_roleplay.py +293 -293
- package/md_cg/migrate_wisdom_graph.py +360 -360
- package/md_cg/mreview/__init__.py +25 -25
- package/md_cg/mreview/__main__.py +110 -110
- package/md_cg/mreview/bundle.py +178 -178
- package/md_cg/mreview/candidates.py +262 -262
- package/md_cg/mreview/govern.py +694 -693
- package/md_cg/mreview/locate.py +939 -939
- package/md_cg/mreview/pipeline.py +728 -728
- package/md_cg/mreview/rules/duplication.json +21 -21
- package/md_cg/mreview/rules/field_coverage.json +54 -54
- package/md_cg/mreview/rules/source_license.json +21 -21
- package/md_cg/mreview/rules/template_flow.json +21 -21
- package/md_cg/mreview/ruleset.py +252 -252
- package/md_cg/nodefile.py +575 -575
- package/md_cg/pooling.py +484 -472
- package/md_cg/postings.py +300 -298
- package/md_cg/predict.py +1100 -1100
- package/md_cg/progressive.py +123 -123
- package/md_cg/protect.py +272 -272
- package/md_cg/protocol/md_cg_gate.proto +33 -33
- package/md_cg/protocol.py +372 -372
- package/md_cg/provenance.py +582 -582
- package/md_cg/reach.py +453 -453
- package/md_cg/readcache.py +143 -0
- package/md_cg/reconcile.py +228 -0
- package/md_cg/refindex.py +833 -833
- package/md_cg/refine.py +604 -604
- package/md_cg/review_cli.py +170 -0
- package/md_cg/roleviews.py +89 -89
- package/md_cg/routing.py +393 -365
- package/md_cg/run_tests.py +211 -0
- package/md_cg/scrub.py +13 -3
- package/md_cg/security.py +128 -18
- package/md_cg/self_state.py +1029 -1029
- package/md_cg/selfreport.py +152 -151
- package/md_cg/semantic/__init__.py +10 -10
- package/md_cg/semantic/canonical.py +122 -122
- package/md_cg/semantic/en_normalizer.py +364 -364
- package/md_cg/semantic/en_zh_map.json +28694 -0
- package/md_cg/semantic/export_en_zh_map.py +64 -0
- package/md_cg/semantic/unify.py +45 -0
- package/md_cg/semantic/zh_en_atoms.py +139 -139
- package/md_cg/signer.py +7 -4
- package/md_cg/sources.py +816 -582
- package/md_cg/statushdr.py +179 -179
- package/md_cg/stg.py +54 -37
- package/md_cg/subgraph.py +729 -729
- package/md_cg/sustain.py +35 -5
- package/md_cg/tasks.py +470 -470
- package/md_cg/test_access_hints.py +147 -0
- package/md_cg/test_action_derive.py +203 -203
- package/md_cg/test_audit_rotate.py +270 -270
- package/md_cg/test_autonomy.py +143 -143
- package/md_cg/test_bench_governance.py +102 -102
- package/md_cg/test_blindspot_tickets.py +166 -166
- package/md_cg/test_branch_discard_tombstone.py +136 -0
- package/md_cg/test_branches.py +13 -3
- package/md_cg/test_ccg_perturb.py +184 -184
- package/md_cg/test_ccgc.py +433 -433
- package/md_cg/test_census_prune.py +81 -81
- package/md_cg/test_chain_read_isolate.py +168 -0
- package/md_cg/test_cond_compose_anchors.py +76 -76
- package/md_cg/test_cond_match.py +165 -165
- package/md_cg/test_condition_anchor.py +81 -81
- package/md_cg/test_d_meta.py +412 -412
- package/md_cg/test_datapath_device_name.py +203 -0
- package/md_cg/test_datapath_root.py +199 -199
- package/md_cg/test_emit_negtail_cache.py +156 -0
- package/md_cg/test_en_pipeline.py +22 -2
- package/md_cg/test_gain_gate.py +212 -212
- package/md_cg/test_govern_directread.py +421 -0
- package/md_cg/test_health_scale.py +173 -173
- package/md_cg/test_hive_ingest.py +285 -0
- package/md_cg/test_hot_cold.py +215 -215
- package/md_cg/test_hyperedge.py +245 -245
- package/md_cg/test_i26_empty_first_write.py +116 -0
- package/md_cg/test_i27_e041_identity.py +128 -0
- package/md_cg/test_i28_hotcache_prodpath.py +122 -0
- package/md_cg/test_i32_hotcache_env_key.py +218 -0
- package/md_cg/test_identity_attribution.py +96 -15
- package/md_cg/test_index_durability.py +17 -3
- package/md_cg/test_interop.py +95 -0
- package/md_cg/test_interop_judgment.py +228 -0
- package/md_cg/test_issue39_utf8_stdio.py +273 -0
- package/md_cg/test_lifecycle.py +309 -309
- package/md_cg/test_linkref.py +306 -306
- package/md_cg/test_links_concurrent_write.py +188 -0
- package/md_cg/test_lock.py +43 -43
- package/md_cg/test_md_access_parity.py +255 -255
- package/md_cg/test_md_writepath.py +345 -345
- package/md_cg/test_mdstore_search_parity.py +160 -0
- package/md_cg/test_merge_upsert.py +168 -0
- package/md_cg/test_mr_m2.py +587 -587
- package/md_cg/test_mr_m3.py +710 -710
- package/md_cg/test_mr_m4.py +485 -485
- package/md_cg/test_n123_derive_expiry_chain.py +205 -0
- package/md_cg/test_n130_verify_falsified_protect.py +185 -0
- package/md_cg/test_n131_merge_gate.py +205 -0
- package/md_cg/test_p0.py +250 -250
- package/md_cg/test_p1.py +316 -316
- package/md_cg/test_p10_identity.py +173 -173
- package/md_cg/test_p11_consistency.py +233 -233
- package/md_cg/test_p12_metacognition.py +212 -212
- package/md_cg/test_p13_encryption.py +241 -241
- package/md_cg/test_p14_sustain.py +249 -249
- package/md_cg/test_p15_scrub.py +280 -280
- package/md_cg/test_p16_self_state.py +301 -301
- package/md_cg/test_p17_predict.py +354 -354
- package/md_cg/test_p18_whitebox.py +171 -171
- package/md_cg/test_p19_migrate_roleplay.py +149 -149
- package/md_cg/test_p1x_ref_root.py +160 -0
- package/md_cg/test_p20_evolution.py +315 -315
- package/md_cg/test_p21_tokens.py +293 -270
- package/md_cg/test_p22_theory.py +175 -175
- package/md_cg/test_p23_links.py +311 -311
- package/md_cg/test_p24_evidence.py +227 -227
- package/md_cg/test_p25_weights.py +156 -156
- package/md_cg/test_p26_refindex.py +416 -416
- package/md_cg/test_p27_docindex.py +16 -7
- package/md_cg/test_p28_refcheck.py +305 -305
- package/md_cg/test_p29_session_ingest_export.py +354 -333
- package/md_cg/test_p2_mcp.py +3 -0
- package/md_cg/test_p3.py +11 -2
- package/md_cg/test_p30_maintain.py +330 -330
- package/md_cg/test_p31_insight.py +534 -534
- package/md_cg/test_p32_backfill.py +7 -1
- package/md_cg/test_p33_ccg_wiring.py +293 -293
- package/md_cg/test_p34_crosscheck.py +331 -331
- package/md_cg/test_p35_conditioned_claim.py +252 -252
- package/md_cg/test_p36_kp_align.py +230 -230
- package/md_cg/test_p37_condition_space.py +248 -248
- package/md_cg/test_p38_concurrent_flush.py +102 -0
- package/md_cg/test_p38_contextualize.py +273 -273
- package/md_cg/test_p39_verify_flow.py +153 -0
- package/md_cg/test_p39_vision_evidence.py +369 -369
- package/md_cg/test_p40_refine_worklist.py +241 -241
- package/md_cg/test_p41_evolve_patrol.py +224 -224
- package/md_cg/test_p42_provenance.py +269 -269
- package/md_cg/test_p43_pooling.py +412 -398
- package/md_cg/test_p44_md_whitebox.py +231 -231
- package/md_cg/test_p45_session_identity.py +219 -219
- package/md_cg/test_p46_unit_scope.py +272 -272
- package/md_cg/test_p47_session_view.py +316 -0
- package/md_cg/test_p4_fuzzy.py +223 -223
- package/md_cg/test_p5_semantic.py +226 -226
- package/md_cg/test_p6_consolidate.py +440 -387
- package/md_cg/test_p7_goals_recent.py +202 -202
- package/md_cg/test_p8_subgraph_chain.py +200 -200
- package/md_cg/test_p9_forget_protect.py +231 -231
- package/md_cg/test_predict_beta.py +135 -135
- package/md_cg/test_preflight_failclosed.py +100 -100
- package/md_cg/test_progressive.py +146 -146
- package/md_cg/test_propose_tail_index.py +157 -0
- package/md_cg/test_protocol.py +243 -243
- package/md_cg/test_reach.py +378 -378
- package/md_cg/test_reach_keys.py +201 -201
- package/md_cg/test_read_clip.py +141 -141
- package/md_cg/test_read_scope_b27.py +277 -0
- package/md_cg/test_readcache_default_on.py +168 -0
- package/md_cg/test_readcache_precise_inval.py +270 -0
- package/md_cg/test_readcache_prodpath.py +203 -0
- package/md_cg/test_reconcile_v0.py +294 -0
- package/md_cg/test_retr_gates_prodpath.py +140 -0
- package/md_cg/test_retr_s1.py +344 -340
- package/md_cg/test_retr_s1b.py +276 -209
- package/md_cg/test_retr_s3.py +194 -194
- package/md_cg/test_retr_s4.py +163 -163
- package/md_cg/test_retr_s5.py +200 -200
- package/md_cg/test_retr_s6.py +157 -157
- package/md_cg/test_retr_s7.py +392 -384
- package/md_cg/test_retr_s8_time.py +369 -316
- package/md_cg/test_retr_s9_edges.py +286 -286
- package/md_cg/test_retr_s9_entity_ctx.py +9 -3
- package/md_cg/test_retr_score_once.py +208 -0
- package/md_cg/test_review_conformance.py +367 -367
- package/md_cg/test_review_onepass.py +170 -0
- package/md_cg/test_role_views.py +354 -354
- package/md_cg/test_rrf_graph_seed_cache.py +154 -0
- package/md_cg/test_security_audit.py +155 -0
- package/md_cg/test_security_audit_b26.py +161 -0
- package/md_cg/test_security_audit_v21.py +250 -0
- package/md_cg/test_sem_noise.py +242 -242
- package/md_cg/test_semantic_canonical.py +16 -2
- package/md_cg/test_session_isolation.py +168 -0
- package/md_cg/test_snapshot_autoclose.py +187 -0
- package/md_cg/test_subproc_encoding.py +192 -192
- package/md_cg/test_sustain_mutual.py +153 -153
- package/md_cg/test_tail_watermark_race.py +208 -0
- package/md_cg/test_tasks.py +409 -409
- package/md_cg/test_tenant_env_override_warn.py +139 -0
- package/md_cg/test_tenant_registry_corrupt_warn.py +151 -0
- package/md_cg/test_tool_face.py +189 -189
- package/md_cg/test_transfer.py +180 -180
- package/md_cg/test_trust.py +361 -361
- package/md_cg/test_twophase.py +286 -286
- package/md_cg/test_v14_fixes.py +38 -20
- package/md_cg/test_validity_filter.py +280 -280
- package/md_cg/test_verify_answer.py +138 -138
- package/md_cg/test_verify_dirty_reconcile.py +157 -0
- package/md_cg/test_wisdom_md_store.py +292 -292
- package/md_cg/test_writelimit.py +197 -197
- package/md_cg/test_writepipe.py +214 -214
- package/md_cg/theory.py +6 -3
- package/md_cg/tokens.py +85 -14
- package/md_cg/tool_face.py +260 -260
- package/md_cg/trust.py +986 -950
- package/md_cg/twophase.py +231 -231
- package/md_cg/units.py +668 -667
- package/md_cg/vision_evidence.py +667 -666
- package/md_cg/weights.py +624 -624
- package/md_cg/whitebox.py +527 -527
- package/md_cg/whitebox_kb/__init__.py +37 -37
- package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
- package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
- package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
- package/md_cg/whitebox_kb/engine.py +310 -310
- package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
- package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
- package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
- package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
- package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
- package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
- package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
- package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
- package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
- package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
- package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
- package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
- package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
- package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
- package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
- package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
- package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
- package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
- package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
- package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
- package/md_cg/whitebox_kb/wisdom/multilang_ir.py +128 -124
- package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
- package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
- package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
- package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
- package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
- package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
- package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
- package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
- package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
- package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
- package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
- package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
- package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
- package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
- package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
- package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
- package/md_cg/writelimit.py +356 -356
- package/md_cg/writepipe.py +20 -8
- package/package.json +101 -96
- package/skills/plugin.json +54 -54
- package/skills/skills/designer-perspective/SKILL.md +158 -158
- package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
- package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
- package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
- package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
- package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
- package/skills/skills/designer-perspective/scripts/designer.py +545 -545
- package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
- package/skills/skills/designer-perspective/tests/selftest.py +61 -61
- package/skills/skills/lingshu-browser/SKILL.md +60 -60
- package/skills/skills/lingshu-compiler/SKILL.md +56 -56
- package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
- package/skills/skills/lingshu-graph/SKILL.md +63 -63
- package/skills/skills/lingshu-net/SKILL.md +48 -48
- package/skills/skills/lingshu-os/SKILL.md +64 -64
- package/skills/skills/lingshu-pylang/SKILL.md +71 -71
- package/src/bridge.ts +33 -0
- package/src/hooks.ts +38 -2
- package/src/index.ts +526 -518
- package/src/lib/datapath.ts +326 -326
- package/src/lib/mdcg_client.ts +413 -413
- package/src/lib/mutual.ts +428 -428
- package/src/lib/prompt_safety.ts +62 -62
- package/src/lib/python_path.ts +71 -71
- package/src/lib/roleplay_web.ts +116 -29
- package/src/lib/token_store.ts +13 -3
- package/src/tools.ts +212 -212
- package/zcode/AGENTS.md +11 -3
- package/zcode/README.md +41 -41
- /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
package/md_cg/mreview/locate.py
CHANGED
|
@@ -1,939 +1,939 @@
|
|
|
1
|
-
# -*- coding: utf-8 -*-
|
|
2
|
-
"""记忆评审流水线 · M3 定位模块(D1 字段级定位,确定性、零写入)。
|
|
3
|
-
|
|
4
|
-
真源:docs/记忆评审系统_立项设计与施工交接_20260915.md §5 D1。
|
|
5
|
-
|
|
6
|
-
D1 契约:
|
|
7
|
-
|
|
8
|
-
locate(node_id, issue_hint) -> [{"field", "span", "issue_kind", "evidence"}, ...]
|
|
9
|
-
|
|
10
|
-
* **field** —— frontmatter 字段名,或 `"content"`(正文)
|
|
11
|
-
* **span** —— 命中处的字符区间 `[start, end)`(半开),**相对正文 content**;
|
|
12
|
-
纯字段级缺失无字符区间可指 → `None`
|
|
13
|
-
* **issue_kind** —— dup / missing_field / stale / contradiction / weak_source / template_flow
|
|
14
|
-
(`ISSUE_KINDS`=问题面);另有 `ADVISORY_KINDS`(观测面,可显式定位
|
|
15
|
-
但**不进默认全量、不进评审告警面**,见下「stale 与 observation_aged」)
|
|
16
|
-
* **evidence** —— 白箱判据:为什么算问题(人可复核、机器可断言)
|
|
17
|
-
|
|
18
|
-
定位的职责边界:M2 的意见说「这一条有问题」,D1 回答「问题在这一条的哪个字段/哪一段」。
|
|
19
|
-
故本模块只做**确定性可判定**的事——语义级矛盾(正文自相矛盾)**不猜**,如实返回
|
|
20
|
-
BLINDSPOT 项交回 LLM 意见层(`status="blindspot"`,见 `locate()` 返回值)。
|
|
21
|
-
|
|
22
|
-
口径全部引用既有唯一真源,**不在本模块另立一份**(防「术语双写法」漂移):
|
|
23
|
-
|
|
24
|
-
=============== ==================================================================
|
|
25
|
-
字段/正文结构 ``nodefile``(CCG_MARKS / CONDITION_SLOTS / VERIFICATION_BASIS /
|
|
26
|
-
loads / content_hash / is_placeholder_text / time_window_text)
|
|
27
|
-
模板骨架 ``writelimit._skeleton`` / ``MIN_SKELETON``
|
|
28
|
-
赛道×基底相容 ``crosscheck``(与 M1 `ruleset._chk_basis_licensed` 同源)
|
|
29
|
-
字段层门限 ``ruleset``(即 `rules/*.json` 的 `matcher.layer`,见
|
|
30
|
-
`field_layer_scope`——**派生**,不在此另立一份)
|
|
31
|
-
索引快照 ``conformance.load_index``
|
|
32
|
-
=============== ==================================================================
|
|
33
|
-
|
|
34
|
-
与 M1(`ruleset` 机械层)的用词归并:D1 说 `dup`,M1 的 `dup_hash_group` 说
|
|
35
|
-
`dup_content`——同一件事两种写法,由 `KIND_ALIASES` 单一归口(`canonical_kind`)。
|
|
36
|
-
|
|
37
|
-
超出 D1 最小契约的**增强键**(供人工核对直接跳文件/跳行,不影响四键契约):
|
|
38
|
-
`line`(节点文件原文 1-based 行号)、`sentence`(正文句索引,0 基)、
|
|
39
|
-
`snippet`(命中片段)、`rule`(触发定位的判据名,审计用)、`peer`(同组对照节点)、
|
|
40
|
-
`cause`(`contradiction` 的**成因**判定,见 `hash_mismatch_cause`——同一条命中
|
|
41
|
-
可能是「真不一致」也可能是「索引快照滞后」,两种成因不分会让复核者误判为数据损坏)。
|
|
42
|
-
|
|
43
|
-
**`stale` 与 `observation_aged` 的判据源分工(2026-09-16 修正)**:
|
|
44
|
-
|
|
45
|
-
* `stale`(**问题面**)—— **依赖存在性**:`code_ref`/`doc_ref` 指的源文件已不存在
|
|
46
|
-
(悬空)或已漂移(区间哈希不符)。判据直接调 `refindex.probe_ref`,**不另立一份**
|
|
47
|
-
(防「术语双写法」漂移)。代码知识的真值挂在源文件上,源没了/改了才是适用边界越出。
|
|
48
|
-
* `observation_aged`(**观测面**,`ADVISORY_KINDS`,**不进默认全量**)——
|
|
49
|
-
`condition_space.time_window` 已过。这是**观测时刻**不是失效声明:写入端
|
|
50
|
-
(`mdcg.add`,见其 `OBSERVATION_WINDOW_SEC`)在未给 time_window 时以**写入时刻**
|
|
51
|
-
自动填 1 小时窗,故真库中「已过期」绝大多数是「写入超过 1 小时」,与知识是否失效
|
|
52
|
-
无关——把它当问题投进评审告警面就是系统性误报(实测:全库 1453 条过期里 1434 条
|
|
53
|
-
属默认窗填充)。需要时效判定用 `stale`(依赖存在性),不是这里。
|
|
54
|
-
"""
|
|
55
|
-
from __future__ import annotations
|
|
56
|
-
|
|
57
|
-
import argparse
|
|
58
|
-
import json
|
|
59
|
-
import os
|
|
60
|
-
import re
|
|
61
|
-
import sys
|
|
62
|
-
import time
|
|
63
|
-
|
|
64
|
-
from .. import conformance as CF
|
|
65
|
-
from .. import crosscheck as CC
|
|
66
|
-
from .. import nodefile as NF
|
|
67
|
-
from .. import refindex as RI
|
|
68
|
-
from .. import writelimit as WL
|
|
69
|
-
from . import ruleset as RS
|
|
70
|
-
|
|
71
|
-
__all__ = ["ISSUE_KINDS", "ADVISORY_KINDS", "KIND_ALIASES", "canonical_kind", "locate",
|
|
72
|
-
"locate_many", "locate_package", "sentence_spans", "mark_spans", "load_node",
|
|
73
|
-
"main", "field_layer_scope", "hash_mismatch_cause", "HASH_CAUSES"]
|
|
74
|
-
|
|
75
|
-
#: D1 的问题类别(真源:立项文档 §5 D1)——**问题面**:命中即「知识有问题」
|
|
76
|
-
ISSUE_KINDS = ("dup", "missing_field", "stale", "contradiction",
|
|
77
|
-
"weak_source", "template_flow")
|
|
78
|
-
|
|
79
|
-
#: **观测面**(咨询级):可显式定位,但**不进默认全量定位、不进评审告警面**。
|
|
80
|
-
#: 判据:本类命中说的是「观测手段的副产品」(如「这条记忆写入超过 1 小时」),
|
|
81
|
-
#: 不是「知识有问题」——把它混进 ISSUE_KINDS 会让每一次写入在 1 小时后自动变成
|
|
82
|
-
#: 待评审问题(系统性误报)。故与问题面分表,只有显式点名才产出。
|
|
83
|
-
ADVISORY_KINDS = ("observation_aged",)
|
|
84
|
-
|
|
85
|
-
#: M1(ruleset 机械层)与 D1 的用词归并——两处说的是同一件事,不许各说各话
|
|
86
|
-
KIND_ALIASES = {"dup_content": "dup", # M1 `dup_hash_group`
|
|
87
|
-
"missing": "missing_field",
|
|
88
|
-
"flow": "template_flow",
|
|
89
|
-
"template_flow_digits_only": "template_flow"}
|
|
90
|
-
|
|
91
|
-
#: 扫描 frontmatter 的字段清单(真源字段名取自 nodefile 口径,不自由发明)
|
|
92
|
-
FM_SCAN_FIELDS = ("role", "tags", "importance", "evidence_count",
|
|
93
|
-
"verification_basis", "lifecycle_state", "condition_space")
|
|
94
|
-
|
|
95
|
-
#: 判为「同模板流水」所需的最少同骨架句数——单句偶合不算流水(防误报)
|
|
96
|
-
MIN_FLOW_SENTENCES = 2
|
|
97
|
-
|
|
98
|
-
#: 一句话摘要的最大长度
|
|
99
|
-
SNIPPET_MAX = 120
|
|
100
|
-
|
|
101
|
-
#: 索引快照与节点文件的 mtime 差**小于**此值 → 视为同一批写入,不判「快照滞后」
|
|
102
|
-
MTIME_TOLERANCE = 1.0
|
|
103
|
-
|
|
104
|
-
#: `content_hash` 不一致的**成因**(确定性判据,不猜;见 `hash_mismatch_cause`)
|
|
105
|
-
HASH_CAUSES = ("index_lag", "true_mismatch", "unknown")
|
|
106
|
-
|
|
107
|
-
#: `refindex.probe_ref` 的五态里,哪些构成「依赖已失效」(`stale` 判据,见 `_loc_stale`):
|
|
108
|
-
#: 只有这两态说明**载体真的没了/改了**;`unresolved`(拿不到 root)与 `error`(读盘失败)
|
|
109
|
-
#: 是观测手段不足,按「不猜」纪律不判。
|
|
110
|
-
_REF_DEAD_STATUSES = ("dangling", "stale")
|
|
111
|
-
|
|
112
|
-
#: 字段层门限缓存(规则库是包内数据文件,进程内不变 → 惰性算一次)
|
|
113
|
-
_FIELD_LAYER_CACHE = None
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
# 生效条件:kind 经 str(kind or "").strip() 得 k(None/空串等假值 → 空串 ""),k 命中 KIND_ALIASES 时返回其规范名,否则原样返回 k。
|
|
117
|
-
def canonical_kind(kind) -> str:
|
|
118
|
-
"""M1/D1 用词 → D1 规范名(未知原样返回,不假装认路)。"""
|
|
119
|
-
k = str(kind or "").strip()
|
|
120
|
-
return KIND_ALIASES.get(k, k)
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
# 生效条件:仅当 rules 与 rules_dir 均为 None 且模块级 _FIELD_LAYER_CACHE 非 None 时直接返回该缓存;否则遍历 rules(dict 取其 "rules" 键的列表、非 dict 直接 list(rules))或 rules_dir 经 RS.load_rules 取得的 rules 列表,从限了 matcher.layer 的 mechanical 规则中收集 field_absent 的 spec["field"] 与 evidence_zero 的 evidence_count,返回 {字段: 排序列},且只在 rules 与 rules_dir 均为 None 时写回 _FIELD_LAYER_CACHE。
|
|
124
|
-
def field_layer_scope(rules=None, rules_dir=None) -> dict:
|
|
125
|
-
"""字段 → 适用层清单(**派生**自 M1 规则库,不在此另立一份)。
|
|
126
|
-
|
|
127
|
-
真源 = `md_cg/mreview/rules/*.json` 的 `matcher.layer`:只收录**限了层**的
|
|
128
|
-
机械规则字段(`field_absent` 取 `spec["field"]`、`evidence_zero` 取
|
|
129
|
-
`evidence_count`);未限层的规则 = 全层适用,不入表。
|
|
130
|
-
|
|
131
|
-
理由:字段范围两处各写一份 ⇒ 必然漂移(同「术语双写法」)。派生使 M1 改规则
|
|
132
|
-
时 M3 自动跟随。规则库损坏 → `load_rules` 抛错(fail-closed,不静默降级成全层)。
|
|
133
|
-
"""
|
|
134
|
-
global _FIELD_LAYER_CACHE
|
|
135
|
-
if rules is None and rules_dir is None and _FIELD_LAYER_CACHE is not None:
|
|
136
|
-
return _FIELD_LAYER_CACHE
|
|
137
|
-
if isinstance(rules, dict):
|
|
138
|
-
rl = list(rules.get("rules") or [])
|
|
139
|
-
elif rules is not None:
|
|
140
|
-
rl = list(rules)
|
|
141
|
-
else:
|
|
142
|
-
rl = list(RS.load_rules(rules_dir)["rules"])
|
|
143
|
-
scope = {}
|
|
144
|
-
for r in rl:
|
|
145
|
-
layers = set((r.get("matcher") or {}).get("layer") or [])
|
|
146
|
-
if not layers:
|
|
147
|
-
continue
|
|
148
|
-
for spec in r.get("mechanical") or []:
|
|
149
|
-
chk = spec.get("check")
|
|
150
|
-
if chk == "field_absent" and spec.get("field"):
|
|
151
|
-
scope.setdefault(str(spec["field"]), set()).update(layers)
|
|
152
|
-
elif chk == "evidence_zero":
|
|
153
|
-
scope.setdefault("evidence_count", set()).update(layers)
|
|
154
|
-
out = {k: sorted(v) for k, v in sorted(scope.items())}
|
|
155
|
-
if rules is None and rules_dir is None:
|
|
156
|
-
_FIELD_LAYER_CACHE = out
|
|
157
|
-
return out
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
# 生效条件:(scope or {}).get(field) 为假值(scope 为 None、空 dict 或 field 不在其中)→ 返回 True;want 为真值时仅当 str((meta or {}).get("layer")) 在 set(want) 内返回 True,否则 False。
|
|
161
|
-
def _layer_ok(field, scope, meta) -> bool:
|
|
162
|
-
"""字段层门限:**限层的字段只在该层的节点上检查**。
|
|
163
|
-
|
|
164
|
-
口径与 M1 `ruleset._scope`(`str(e.get("layer")) in want`)逐字一致——缺层
|
|
165
|
-
(`None`)不在任何层清单内 ⇒ 不检查。两处判据同源,防「越层误报」:
|
|
166
|
-
M1 不在某层报的问题,M3 也不该报。
|
|
167
|
-
"""
|
|
168
|
-
want = (scope or {}).get(field)
|
|
169
|
-
if not want:
|
|
170
|
-
return True
|
|
171
|
-
return str((meta or {}).get("layer")) in set(want)
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
# 生效条件:os.path.getmtime(path) 成功则返回该 mtime;抛 OSError 或 TypeError → 返回 None。
|
|
175
|
-
def _mtime(path):
|
|
176
|
-
"""文件 mtime(不可读 → `None`,不假装知道)。"""
|
|
177
|
-
try:
|
|
178
|
-
return os.path.getmtime(path)
|
|
179
|
-
except (OSError, TypeError):
|
|
180
|
-
return None
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
# 生效条件:root 或 rel(path 为真值时取 path,否则取 meta 的 "path")为空 → 返回 ("unknown", "无 root/path 可用,取不到盘上 mtime,成因未判定");两者都有时取节点文件与 CF.INDEX_FILE 的 mtime,任一为 None → 返回 ("unknown", "节点文件或索引快照不可读,成因未判定"),节点 mtime 减索引 mtime 之差 > MTIME_TOLERANCE → 返回 ("index_lag", …),否则返回 ("true_mismatch", …)。
|
|
184
|
-
def hash_mismatch_cause(meta, *, root=None, path=None) -> tuple:
|
|
185
|
-
"""`content_hash` 声明值 ≠ 正文实算值 → **成因**判定(靠 mtime 证据,不猜)。
|
|
186
|
-
|
|
187
|
-
→ `(cause, note)`;cause ∈ `HASH_CAUSES`;取不到盘上证据 → `"unknown"`。
|
|
188
|
-
|
|
189
|
-
写路径真源(`md_cg/mdcg.py`):写走 `_stage()` 落**分片日志**,`rebuild_index()`
|
|
190
|
-
才写 `_index.json` 快照——故「节点文件比索引快照新」是**正常写路径现象**
|
|
191
|
-
(快照滞后),不是文件被篡改;把两种成因合并成一句「索引与文件不一致」会让
|
|
192
|
-
复核者误判为数据损坏(实测滞后可达 14s 量级)。
|
|
193
|
-
"""
|
|
194
|
-
rel = path if path else (meta or {}).get("path")
|
|
195
|
-
if not root or not rel:
|
|
196
|
-
return "unknown", "无 root/path 可用,取不到盘上 mtime,成因未判定"
|
|
197
|
-
fp = str(rel) if os.path.isabs(str(rel)) else os.path.join(str(root), str(rel))
|
|
198
|
-
fm_t = _mtime(fp)
|
|
199
|
-
idx_t = _mtime(os.path.join(str(root), CF.INDEX_FILE))
|
|
200
|
-
if fm_t is None or idx_t is None:
|
|
201
|
-
return "unknown", "节点文件或索引快照不可读,成因未判定"
|
|
202
|
-
delta = fm_t - idx_t
|
|
203
|
-
if delta > MTIME_TOLERANCE:
|
|
204
|
-
return "index_lag", ("节点文件比索引快照新 %.2fs——分片日志已写、_index.json "
|
|
205
|
-
"未 rebuild(快照滞后属正常写路径现象)" % delta)
|
|
206
|
-
return "true_mismatch", ("节点文件不晚于索引快照(差 %.2fs)——非快照滞后,"
|
|
207
|
-
"索引声明与文件真源相抵触" % delta)
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
# ---------------------------- 基础工具(纯函数) ----------------------------
|
|
211
|
-
|
|
212
|
-
# 生效条件:v 为 list/tuple/dict 时返回 not v(空容器 → True);其余类型返回 v is None 或 str(v).strip() 为 "" 或 "None"。
|
|
213
|
-
def _blank(v) -> bool:
|
|
214
|
-
if isinstance(v, (list, tuple, dict)):
|
|
215
|
-
return not v
|
|
216
|
-
return v is None or str(v).strip() in ("", "None")
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
# 生效条件:int(float(v)) 可算(含数字字符串)则返回该整数;抛 TypeError 或 ValueError(含 v 为 None、非数字串)→ 返回 None。
|
|
220
|
-
def _as_int(v):
|
|
221
|
-
try:
|
|
222
|
-
return int(float(v))
|
|
223
|
-
except (TypeError, ValueError):
|
|
224
|
-
return None
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
# 生效条件:float(v) 可算则返回该浮点数;抛 TypeError 或 ValueError(含 v 为 None、非数字串)→ 返回 None。
|
|
228
|
-
def _as_float(v):
|
|
229
|
-
try:
|
|
230
|
-
return float(v)
|
|
231
|
-
except (TypeError, ValueError):
|
|
232
|
-
return None
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
_SENT_END = "。!?;!?;\n"
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
# 生效条件:content 为假值(None/空串)按 "" 处理,逐字符命中 _SENT_END 切出且 seg.strip() 非空的段以 len(out) 为序号追加 (索引, start, i+1, 原文),末尾 tail.strip() 非空亦追加;无合格段 → 返回空列表。
|
|
239
|
-
def sentence_spans(content: str) -> list:
|
|
240
|
-
"""正文 → `[(句索引, start, end, 原文)]`;空句不编号(索引连续,确定性)。"""
|
|
241
|
-
text, out, start = content or "", [], 0
|
|
242
|
-
for i, ch in enumerate(text):
|
|
243
|
-
if ch in _SENT_END:
|
|
244
|
-
seg = text[start:i + 1]
|
|
245
|
-
if seg.strip():
|
|
246
|
-
out.append((len(out), start, i + 1, seg))
|
|
247
|
-
start = i + 1
|
|
248
|
-
tail = text[start:]
|
|
249
|
-
if tail.strip():
|
|
250
|
-
out.append((len(out), start, len(text), tail))
|
|
251
|
-
return out
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
#: 正文要素行:`# 功能名:…` / `# 生效条件: …`(间隔与分隔符两形态都收)
|
|
255
|
-
_RE_MARK = re.compile(r"^#[ \t]*(?P<mark>[^::\n]{1,16})[::][^\n]*", re.M)
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
# 生效条件:在 content(假值 → "")上用模块级 _RE_MARK 迭代匹配,以 m.group("mark").strip() 为键 setdefault 记下首次出现的 [m.start(), m.end());无匹配(含 content 为假值)→ 返回空 dict。
|
|
259
|
-
def mark_spans(content: str) -> dict:
|
|
260
|
-
"""正文 CCG 要素行 → `{要素名: [start, end)}`;同要素取首次出现(确定性)。"""
|
|
261
|
-
out = {}
|
|
262
|
-
for m in _RE_MARK.finditer(content or ""):
|
|
263
|
-
out.setdefault(m.group("mark").strip(), [m.start(), m.end()])
|
|
264
|
-
return out
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
# 生效条件:(text or "").split("\n") 的每行 strip 后非空、不以 "#" 开头且含 ":" 时,取首个冒号前的键 k,k 非空且尚未入表则记 (1-based 行号, [该行起始偏移, 起始偏移+len(line)]);text 为假值或无合格行 → 返回空 dict。
|
|
268
|
-
def key_line_spans(text: str) -> dict:
|
|
269
|
-
"""节点文件原文 → `{键: (1-based 行号, [start, end))}`。
|
|
270
|
-
|
|
271
|
-
只认**无 `#` 前缀**且含 `:` 的行——即 frontmatter 行;`---` 分隔线无冒号,
|
|
272
|
-
自然被排除。同键取首次出现。
|
|
273
|
-
"""
|
|
274
|
-
out, pos = {}, 0
|
|
275
|
-
for i, line in enumerate((text or "").split("\n")):
|
|
276
|
-
s = line.strip()
|
|
277
|
-
if s and not s.startswith("#") and ":" in s:
|
|
278
|
-
k = s.split(":", 1)[0].strip()
|
|
279
|
-
if k and k not in out:
|
|
280
|
-
out[k] = (i + 1, [pos, pos + len(line)])
|
|
281
|
-
pos += len(line) + 1
|
|
282
|
-
return out
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
# 生效条件:index 为 dict 时其 "nodes" 为 dict 则返回 index["nodes"],否则返回 index 本身;index 为 None/非 dict 时调 CF.load_index(root),抛 OSError 或 ValueError → 返回 {},返回值为 dict 且其 "nodes" 为 dict → 返回该 "nodes",是 dict → 原样返回,否则返回 {}。
|
|
286
|
-
def _index(root, index=None) -> dict:
|
|
287
|
-
"""索引节点表 `{node_id: meta}`(兼容 load_index 的 `{"nodes": …}` 形态)。"""
|
|
288
|
-
if isinstance(index, dict):
|
|
289
|
-
return index.get("nodes") if isinstance(index.get("nodes"), dict) else index
|
|
290
|
-
try:
|
|
291
|
-
idx = CF.load_index(root)
|
|
292
|
-
except (OSError, ValueError):
|
|
293
|
-
return {}
|
|
294
|
-
if isinstance(idx, dict) and isinstance(idx.get("nodes"), dict):
|
|
295
|
-
return idx["nodes"]
|
|
296
|
-
return idx if isinstance(idx, dict) else {}
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
# 生效条件:_index(root, index) 中 node_id 对应值非 dict → 返回 None;否则用 meta.get("path")(绝对路径直接用,否则 join(root, str(rel or "")))读文件——OSError 时返回 content/text 为 None、fm 为 {} 的 dict,成功则把 NF.loads(text) 得到的 content 与 fm(假值 → {})连同 meta/path/text 一并返回。
|
|
300
|
-
def load_node(node_id, root, *, index=None) -> dict:
|
|
301
|
-
"""读一个节点(索引 meta + 文件 frontmatter + 正文原文)。
|
|
302
|
-
|
|
303
|
-
→ `{"node_id","meta","fm","content","text","path"}`;索引无此项 → `None`。
|
|
304
|
-
`meta` 是**索引快照**(含索引侧 content_hash,矛盾判定要用),`fm` 是**文件真源**。
|
|
305
|
-
"""
|
|
306
|
-
nodes = _index(root, index)
|
|
307
|
-
meta = nodes.get(node_id)
|
|
308
|
-
if not isinstance(meta, dict):
|
|
309
|
-
return None
|
|
310
|
-
rel = meta.get("path")
|
|
311
|
-
fp = rel if (rel and os.path.isabs(str(rel))) else os.path.join(root, str(rel or ""))
|
|
312
|
-
try:
|
|
313
|
-
with open(fp, encoding="utf-8") as f:
|
|
314
|
-
text = f.read()
|
|
315
|
-
except OSError:
|
|
316
|
-
return {"node_id": node_id, "meta": dict(meta), "fm": {}, "content": None,
|
|
317
|
-
"text": None, "path": fp}
|
|
318
|
-
fm, content = NF.loads(text)
|
|
319
|
-
return {"node_id": node_id, "meta": dict(meta), "fm": fm or {}, "content": content,
|
|
320
|
-
"text": text, "path": fp}
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
# 生效条件:span 为 None → 返回 "";否则取 (text or "")[span[0]:span[1]] 去空白并把换行替换为 "⏎",长度超 SNIPPET_MAX 时截断并追加 "…"。
|
|
324
|
-
def _snippet(text, span) -> str:
|
|
325
|
-
"""命中片段(供人工核对肉眼确认「指的是不是这一句」)。"""
|
|
326
|
-
if span is None:
|
|
327
|
-
return ""
|
|
328
|
-
seg = (text or "")[span[0]:span[1]].strip().replace("\n", "⏎")
|
|
329
|
-
return seg[:SNIPPET_MAX] + ("…" if len(seg) > SNIPPET_MAX else "")
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
# 生效条件:任意 node_id/kind/field/span/evidence 均原样写入返回 dict 的 node_id/issue_kind/field/span/evidence 键,可选 rule/line/sentence/snippet/peer/cause/severity 未传时为 None、status 未传时为 "located",不做任何校验。
|
|
333
|
-
def _hit(node_id, kind, field, span, evidence, *, rule=None, line=None,
|
|
334
|
-
sentence=None, snippet=None, peer=None, cause=None, status="located",
|
|
335
|
-
severity=None) -> dict:
|
|
336
|
-
"""命中行(四键契约 + 增强键)。
|
|
337
|
-
|
|
338
|
-
`severity` 是**观测面专用**的等级标注(`ADVISORY_KINDS` 命中带 `"info"`):
|
|
339
|
-
问题面命中不带(问题即问题,无等级可降);本键让下游一眼分清「这是观测
|
|
340
|
-
副产品」与「这是待修的问题」,不必靠 issue_kind 名字去猜。
|
|
341
|
-
"""
|
|
342
|
-
return {"node_id": node_id, "field": field, "span": span, "issue_kind": kind,
|
|
343
|
-
"evidence": evidence, "rule": rule, "line": line, "sentence": sentence,
|
|
344
|
-
"snippet": snippet, "peer": peer, "cause": cause, "status": status,
|
|
345
|
-
"severity": severity}
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
# 生效条件:text 为真值、span 与 content 均非 None 且 text.find(content)>=0 时,返回 text.count("\n",0,min(off+span[0],len(text)))+1 的 1-based 行号;text 假值或 span/content 为 None 或 content 未找到时返回 None。
|
|
349
|
-
def _line_of(text, content, span):
|
|
350
|
-
"""正文区间 → 节点文件原文的 1-based 行号(供人工核对直接跳文件)。"""
|
|
351
|
-
if not text or span is None or content is None:
|
|
352
|
-
return None
|
|
353
|
-
off = text.find(content)
|
|
354
|
-
if off < 0:
|
|
355
|
-
return None
|
|
356
|
-
return text.count("\n", 0, min(off + span[0], len(text))) + 1
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
# ---------------------------- 定位器注册表 ----------------------------
|
|
360
|
-
#
|
|
361
|
-
# 每个定位器是**纯函数**:只吃 (node_id, meta, content, ctx),只吐 hits。
|
|
362
|
-
# 新增判据 = 加函数 + 注册,不改 locate 主流程(与 M1「引擎冻结、规则可增删」同构)。
|
|
363
|
-
# ctx = {"fm", "text", "root", "index", "peers", "now"}
|
|
364
|
-
|
|
365
|
-
# 生效条件:在 node_id/meta/content/ctx 下,对 FM_SCAN_FIELDS 中除 verification_basis、condition_space 外且 _layer_ok(f,scope,meta) 为真的字段,若 meta.get(f) 与 ctx.get("fm") 或 {} 中的同名字段皆 _blank 则记 field_absent;对过 _layer_ok 且 _as_int(meta.get("evidence_count"))==0 记 evidence_zero;对 meta.get("importance") 非 None 且 _as_float 为 None 或不在 [0.0,1.0] 记 field_invalid;对 condition_space 经 meta 或回落 ctx.get("fm") 后 NF.condition_space_missing 非空记 condition_slots;对正文缺 NF.CCG_MARKS 行记 ccg_incomplete,返回这些命中列表。
|
|
366
|
-
def _loc_missing_field(node_id, meta, content, ctx):
|
|
367
|
-
"""frontmatter/正文结构字段为空(判据与 M1 `field_absent`/`evidence_zero` 同源)。
|
|
368
|
-
|
|
369
|
-
**层门限**:限层的字段(如 `role`/`evidence_count` 限 `knowledge`)只在
|
|
370
|
-
该层节点上检查——与 M1 `_scope` 同口径,否则 M1 不报的问题会被 M3 越层报出。
|
|
371
|
-
"""
|
|
372
|
-
fm = ctx.get("fm") or {}
|
|
373
|
-
scope = ctx.get("field_layers")
|
|
374
|
-
if scope is None:
|
|
375
|
-
scope = field_layer_scope()
|
|
376
|
-
lines = key_line_spans(ctx.get("text") or "")
|
|
377
|
-
marks = mark_spans(content or "")
|
|
378
|
-
hits = []
|
|
379
|
-
|
|
380
|
-
for f in FM_SCAN_FIELDS:
|
|
381
|
-
if f in ("verification_basis", "condition_space"):
|
|
382
|
-
continue # 基底空属 weak_source;四槽缺失单独报(要点名缺哪几槽)
|
|
383
|
-
if not _layer_ok(f, scope, meta):
|
|
384
|
-
continue # 越层不报(层门限真源 = M1 规则库 matcher.layer)
|
|
385
|
-
if _blank(meta.get(f)) and _blank(fm.get(f)):
|
|
386
|
-
hits.append(_hit(node_id, "missing_field", f, None,
|
|
387
|
-
"字段 %s 为空(索引与文件两处皆空)" % f,
|
|
388
|
-
rule="field_absent",
|
|
389
|
-
line=(lines.get(f) or (None, None))[0]))
|
|
390
|
-
|
|
391
|
-
if _layer_ok("evidence_count", scope, meta) \
|
|
392
|
-
and _as_int(meta.get("evidence_count")) == 0:
|
|
393
|
-
hits.append(_hit(node_id, "missing_field", "evidence_count", None,
|
|
394
|
-
"evidence_count=0(无验证证据计数)", rule="evidence_zero",
|
|
395
|
-
line=(lines.get("evidence_count") or (None, None))[0]))
|
|
396
|
-
|
|
397
|
-
imp = meta.get("importance")
|
|
398
|
-
if imp is not None:
|
|
399
|
-
fv = _as_float(imp)
|
|
400
|
-
if fv is None or not 0.0 <= fv <= 1.0:
|
|
401
|
-
hits.append(_hit(node_id, "missing_field", "importance", None,
|
|
402
|
-
"importance=%r 不在 [0,1](字段存在但不可用)"
|
|
403
|
-
% (imp,), rule="field_invalid",
|
|
404
|
-
line=(lines.get("importance") or (None, None))[0]))
|
|
405
|
-
|
|
406
|
-
cs = meta.get("condition_space")
|
|
407
|
-
if not isinstance(cs, dict):
|
|
408
|
-
cs = fm.get("condition_space")
|
|
409
|
-
miss = NF.condition_space_missing(cs)
|
|
410
|
-
if miss:
|
|
411
|
-
span = marks.get("生效条件")
|
|
412
|
-
hits.append(_hit(node_id, "missing_field", "condition_space", span,
|
|
413
|
-
"条件空间缺槽 %s(%d/4 已声明)——四槽不全不构成生效条件"
|
|
414
|
-
% ("/".join(miss), len(NF.CONDITION_SLOTS) - len(miss)),
|
|
415
|
-
rule="condition_slots",
|
|
416
|
-
snippet=_snippet(content, span)))
|
|
417
|
-
|
|
418
|
-
# 正文 CCG 要素缺行:行不存在 → 无字符区间可指(span=None),点名缺哪几行
|
|
419
|
-
missing_marks = [m for m in NF.CCG_MARKS if m not in marks]
|
|
420
|
-
if missing_marks:
|
|
421
|
-
hits.append(_hit(node_id, "missing_field", "content", None,
|
|
422
|
-
"正文缺 CCG 要素行 %s(# 要素名:值)——缺声明即缺证据"
|
|
423
|
-
% "/".join(missing_marks), rule="ccg_incomplete"))
|
|
424
|
-
return hits
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
# 生效条件:meta.get("verification_basis") 为空时回落 ctx.get("fm") 或 {} 的 verification_basis,若两者皆 _blank 返回 basis_absent 命中;非空但 str(basis) 不在 NF.VERIFICATION_BASIS 返回 basis_enum 命中;在枚举内时按 CC.classify_track 依 meta.get("layer")、meta.get("tags") 与 content 判赛道,若 CC.basis_licensed 为假返回 basis_licensed 命中;否则返回 []。
|
|
428
|
-
def _loc_weak_source(node_id, meta, content, ctx):
|
|
429
|
-
"""验证基底缺失/越枚举/与赛道不相容(与 M1 `basis_licensed` 判据逐字同源)。"""
|
|
430
|
-
fm = ctx.get("fm") or {}
|
|
431
|
-
basis = meta.get("verification_basis")
|
|
432
|
-
if _blank(basis):
|
|
433
|
-
basis = fm.get("verification_basis")
|
|
434
|
-
_ln = key_line_spans(ctx.get("text") or "").get("verification_basis")
|
|
435
|
-
ln = (_ln or (None, None))[0]
|
|
436
|
-
|
|
437
|
-
if _blank(basis):
|
|
438
|
-
return [_hit(node_id, "weak_source", "verification_basis", None,
|
|
439
|
-
"verification_basis 为空——无验证基底的断言不可裁决",
|
|
440
|
-
rule="basis_absent", line=ln)]
|
|
441
|
-
if str(basis) not in NF.VERIFICATION_BASIS:
|
|
442
|
-
return [_hit(node_id, "weak_source", "verification_basis", None,
|
|
443
|
-
"基底 %s 不在合法枚举(%s)"
|
|
444
|
-
% (basis, "/".join(NF.VERIFICATION_BASIS)),
|
|
445
|
-
rule="basis_enum", line=ln)]
|
|
446
|
-
track = CC.classify_track({"layer": meta.get("layer"),
|
|
447
|
-
"tags": meta.get("tags") or []}, content or "")
|
|
448
|
-
if not CC.basis_licensed(track, basis):
|
|
449
|
-
allow = "/".join(CC.allowed_basis(track)) or "—"
|
|
450
|
-
return [_hit(node_id, "weak_source", "verification_basis", None,
|
|
451
|
-
"赛道 %s(政策 %s)不允许基底 %s;允许 %s"
|
|
452
|
-
% (track, CC.source_policy(track) or "—", basis, allow),
|
|
453
|
-
rule="basis_licensed", line=ln)]
|
|
454
|
-
return []
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
# 生效条件:ctx.get("fm") 或 {} 中 code_ref 或 doc_ref 为非空 dict 且 RI.probe_ref(ref) 返回的 status 属于 _REF_DEAD_STATUSES(dangling 或 stale)时,返回对应 stale 命中(dangling 归因载体消失、stale 归因区间哈希不符);否则返回 []。
|
|
458
|
-
def _loc_stale(node_id, meta, content, ctx):
|
|
459
|
-
"""**依赖存在性**——声明的载体(源文件)已不存在 / 已漂移(真正的适用边界越出)。
|
|
460
|
-
|
|
461
|
-
知识的真值挂在它描述的对象上:代码知识挂在源文件的符号上,源没了或改了,
|
|
462
|
-
知识才真的不再成立。判据与 `refindex` **同源**(`probe_ref`,不另立一份):
|
|
463
|
-
|
|
464
|
-
============== ====================================== ==========
|
|
465
|
-
probe_ref 态 含义 本判据
|
|
466
|
-
============== ====================================== ==========
|
|
467
|
-
``ok`` 源文件在、区间哈希匹配 不判
|
|
468
|
-
``stale`` 源文件在、区间哈希不符(源被改,漂移) **stale**
|
|
469
|
-
``dangling`` 源文件不存在(载体消失) **stale**
|
|
470
|
-
``unresolved`` 拿不到 root(判不了) 不判(不猜)
|
|
471
|
-
``error`` 读盘失败 不判(不猜)
|
|
472
|
-
============== ====================================== ==========
|
|
473
|
-
|
|
474
|
-
后两态是**观测手段不足**,不是「依赖已失效」——按「不猜」纪律如实不报。
|
|
475
|
-
|
|
476
|
-
**时间窗不在此处判**(2026-09-16 修正的原缺陷):`condition_space.time_window`
|
|
477
|
-
是写入端**观测时刻**栏(`mdcg.add` 未给时自动填「写入时刻 +1h」,见其
|
|
478
|
-
`OBSERVATION_WINDOW_SEC`),不是知识的有效期——拿它判「适用边界越出」,
|
|
479
|
-
等于把「这条记忆写入超过 1 小时」冒充为「失效」。真实适用前提由知识自己
|
|
480
|
-
声明在 `code_ref`/`doc_ref` 上,故改由依赖存在性裁决;观测时刻本身不丢,
|
|
481
|
-
降级为 `observation_aged`(观测面,见 `ADVISORY_KINDS`)。
|
|
482
|
-
"""
|
|
483
|
-
fm = ctx.get("fm") or {}
|
|
484
|
-
hits = []
|
|
485
|
-
for key in ("code_ref", "doc_ref"):
|
|
486
|
-
ref = fm.get(key)
|
|
487
|
-
if not isinstance(ref, dict) or not ref:
|
|
488
|
-
continue
|
|
489
|
-
# root 基准必须取自 ref 自身(`_code_ref`/`_doc_ref` 落的是**源大域** root,
|
|
490
|
-
# path 相对它);`ctx["root"]` 是**认知图** root——拿它去拼会把每一条依赖
|
|
491
|
-
# 都误判成悬空。ref 无 root(旧节点)时 probe 返回 `unresolved` → 不判。
|
|
492
|
-
probe = RI.probe_ref(ref)
|
|
493
|
-
st = str(probe.get("status") or "")
|
|
494
|
-
if st not in _REF_DEAD_STATUSES:
|
|
495
|
-
continue
|
|
496
|
-
rel = str(ref.get("path") or "?")
|
|
497
|
-
if st == "dangling":
|
|
498
|
-
ev = ("依赖的源文件已不存在(%s:%s)——知识所指的载体消失,适用边界越出"
|
|
499
|
-
% (key, rel))
|
|
500
|
-
else:
|
|
501
|
-
ev = ("依赖的源文件已漂移(%s:%s):索引区间哈希 %s ≠ 现算 %s——"
|
|
502
|
-
"源被改动,知识所指的符号可能已不是原物"
|
|
503
|
-
% (key, rel, probe.get("hash_expected"), probe.get("hash")))
|
|
504
|
-
hits.append(_hit(node_id, "stale", key, None, ev, rule="ref_" + st))
|
|
505
|
-
return hits
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
# 生效条件:从 meta.get("condition_space") 或回落 ctx.get("fm") 的 condition_space(须为 dict)取 time_window,缺则取 meta.get("time_window"),若 NF.is_full_time_window(tw) 或 float(tw[1]) 抛 TypeError/ValueError/IndexError/KeyError 或 ctx.get("now") 为 None 或 hi>=float(now) 则返回 [];否则返回一条 severity="info" 的 observation_aged 命中。
|
|
509
|
-
def _loc_observation_aged(node_id, meta, content, ctx):
|
|
510
|
-
"""时间窗已过——**这是观测时刻,不是失效声明**(观测面,不进告警面)。
|
|
511
|
-
|
|
512
|
-
时间窗来源链与 `_loc_missing_field` 的槽检查同构:`meta.condition_space`
|
|
513
|
-
(显式注入面)→ `fm.condition_space`(文件真源)→ `meta.time_window`
|
|
514
|
-
(索引快照字段)。**索引快照只透传 `time_window`**(`mdcg.py` `_stage` 口径:
|
|
515
|
-
时空字段入快照、condition_space 整块不入),故第三跳必留。
|
|
516
|
-
|
|
517
|
-
**不进默认全量**(不在 `ISSUE_KINDS`):写入端未给 time_window 时以写入时刻
|
|
518
|
-
自动填 1 小时窗,故真库「已过期」绝大多数是「写入超过 1 小时」,与知识是否
|
|
519
|
-
失效无关。观测时刻本身仍如实报(审计有用),只是**不冒充问题**——命中带
|
|
520
|
-
`severity="info"` 标记它是咨询级。
|
|
521
|
-
"""
|
|
522
|
-
fm = ctx.get("fm") or {}
|
|
523
|
-
cs = meta.get("condition_space")
|
|
524
|
-
if not isinstance(cs, dict):
|
|
525
|
-
csf = fm.get("condition_space")
|
|
526
|
-
cs = csf if isinstance(csf, dict) else None
|
|
527
|
-
tw = (cs or {}).get("time_window")
|
|
528
|
-
if tw is None:
|
|
529
|
-
tw = meta.get("time_window")
|
|
530
|
-
if NF.is_full_time_window(tw):
|
|
531
|
-
return [] # 全时窗是**合法声明**(任意时刻成立),不判
|
|
532
|
-
try:
|
|
533
|
-
hi = float(tw[1])
|
|
534
|
-
except (TypeError, ValueError, IndexError, KeyError):
|
|
535
|
-
return []
|
|
536
|
-
now = ctx.get("now")
|
|
537
|
-
if now is None or hi >= float(now):
|
|
538
|
-
return []
|
|
539
|
-
span = mark_spans(content or "").get("生效条件")
|
|
540
|
-
return [_hit(node_id, "observation_aged", "condition_space", span,
|
|
541
|
-
"观测时间窗 %s 已过(hi=%.0f < now=%.0f)——这是**观测时刻**,"
|
|
542
|
-
"不是失效声明(知识不因观测过去而失效);时效判定见 stale"
|
|
543
|
-
% (NF.time_window_text(tw) or "?", hi, float(now)),
|
|
544
|
-
rule="observation_window_passed", severity="info",
|
|
545
|
-
line=_line_of(ctx.get("text"), content, span),
|
|
546
|
-
snippet=_snippet(content, span))]
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
# 生效条件:h 取 str(meta.get("content_hash") or "").strip(),若 _blank(h) 则 h=NF.content_hash(content 或 "");对 ctx.get("peers") 或 [] 中 node_id 不等于本节点且 ph=str(p.get("hash") or "") 或回落 NF.content_hash(p.get("content") or "") 后非空且等于 h 的 peer 计入 matched;matched 非空时返回一条 dup 命中(span=[0,len(body)]),否则返回 []。
|
|
550
|
-
def _loc_dup(node_id, meta, content, ctx):
|
|
551
|
-
"""同组同内容指纹(与 M1 `dup_hash_group` 同判据;D1 名 `dup`)。"""
|
|
552
|
-
body = content if content is not None else ""
|
|
553
|
-
h = str(meta.get("content_hash") or "").strip()
|
|
554
|
-
if _blank(h):
|
|
555
|
-
h = NF.content_hash(body) # 索引无指纹 → 用正文实算(诚实标注来源)
|
|
556
|
-
matched = []
|
|
557
|
-
for p in ctx.get("peers") or []:
|
|
558
|
-
if p.get("node_id") == node_id:
|
|
559
|
-
continue
|
|
560
|
-
ph = str(p.get("hash") or "") or NF.content_hash(p.get("content") or "")
|
|
561
|
-
if ph and ph == h:
|
|
562
|
-
matched.append(p)
|
|
563
|
-
if not matched:
|
|
564
|
-
return []
|
|
565
|
-
span = [0, len(body)]
|
|
566
|
-
return [_hit(node_id, "dup", "content_hash", span,
|
|
567
|
-
"与 %s 同内容指纹 %s(同类共 %d 条),整篇正文重复"
|
|
568
|
-
% (matched[0].get("node_id"), h, len(matched) + 1),
|
|
569
|
-
rule="dup_hash_group", peer=matched[0].get("node_id"),
|
|
570
|
-
snippet=_snippet(body, span))]
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
# 生效条件:对 ctx.get("peers") 或 [] 中每个非自身 peer,按 sentence_spans 对齐 content 与 peer content 的句子,当同一句位上去数字骨架相同(WL._skeleton(seg) 非空、长度 ≥ WL.MIN_SKELETON 且等于 peer 骨架)且 seg.strip() != ptext.strip() 的句子数达到 MIN_FLOW_SENTENCES 时,为这些句各返回一条 template_flow 命中;否则返回 []。
|
|
574
|
-
def _loc_template_flow(node_id, meta, content, ctx):
|
|
575
|
-
"""同模板流水:与同组节点逐句「去数字骨架相同、字面不同」→ 指向那些句。
|
|
576
|
-
|
|
577
|
-
与 M1 `template_flow_digits_only` 同判据(都走 `writelimit._skeleton`),
|
|
578
|
-
差别只在产物:M1 给一条条目级 issue,D1 给出**具体是哪几句**(span + 句索引)。
|
|
579
|
-
单句偶合不算流水 → 需同组至少 `MIN_FLOW_SENTENCES` 句同时成立。
|
|
580
|
-
"""
|
|
581
|
-
tgt = sentence_spans(content or "")
|
|
582
|
-
hits = []
|
|
583
|
-
for p in ctx.get("peers") or []:
|
|
584
|
-
if p.get("node_id") == node_id:
|
|
585
|
-
continue
|
|
586
|
-
pseg = sentence_spans(p.get("content") or "")
|
|
587
|
-
same = []
|
|
588
|
-
for (i, s, e, seg) in tgt:
|
|
589
|
-
if i >= len(pseg):
|
|
590
|
-
break
|
|
591
|
-
ptext = pseg[i][3]
|
|
592
|
-
sk = WL._skeleton(seg)
|
|
593
|
-
if not sk or len(sk) < WL.MIN_SKELETON or sk != WL._skeleton(ptext):
|
|
594
|
-
continue
|
|
595
|
-
if seg.strip() == ptext.strip():
|
|
596
|
-
continue # 字面全同属整篇重复(dup),不是「只换数字」的流水
|
|
597
|
-
same.append((i, s, e, seg, sk))
|
|
598
|
-
if len(same) < MIN_FLOW_SENTENCES:
|
|
599
|
-
continue
|
|
600
|
-
for (i, s, e, seg, sk) in same:
|
|
601
|
-
hits.append(_hit(node_id, "template_flow", "content", [s, e],
|
|
602
|
-
"同模板流水(与 %s 第 %d 句去数字骨架相同、仅字面数值不同):%s"
|
|
603
|
-
% (p.get("node_id"), i, sk[:40]),
|
|
604
|
-
rule="template_flow_digits_only", sentence=i,
|
|
605
|
-
peer=p.get("node_id"),
|
|
606
|
-
line=_line_of(ctx.get("text"), content, [s, e]),
|
|
607
|
-
snippet=seg.strip()[:SNIPPET_MAX]))
|
|
608
|
-
return hits
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
# 生效条件:meta.get("content_hash") 非 _blank 且 content 非 None 且 h != NF.content_hash(content) 时,返回一条 hash_declared_vs_actual 的 contradiction 命中(cause 由 hash_mismatch_cause 依 meta/ctx.get("root")/ctx.get("path") 判);ctx.get("fm") 或 {} 的 id 非 _blank 且 str(fm_id) != str(node_id) 时额外返回一条 id_declared_vs_index 命中;两者皆不成立返回 []。
|
|
612
|
-
def _loc_contradiction(node_id, meta, content, ctx):
|
|
613
|
-
"""**确定性**矛盾:声明与事实不符(指纹 / 标识)。
|
|
614
|
-
|
|
615
|
-
语义级矛盾(正文自相矛盾)不在此处——那是 LLM 的活,D1 不猜(见 `SEMANTIC_ONLY`)。
|
|
616
|
-
指纹不一致**必须带成因**(`cause`):`index_lag`(快照滞后,属正常写路径现象,
|
|
617
|
-
复核者无需修数据)与 `true_mismatch`(真不一致,需查)处置完全不同。
|
|
618
|
-
"""
|
|
619
|
-
hits = []
|
|
620
|
-
lines = key_line_spans(ctx.get("text") or "")
|
|
621
|
-
h = str(meta.get("content_hash") or "").strip()
|
|
622
|
-
if not _blank(h) and content is not None:
|
|
623
|
-
actual = NF.content_hash(content)
|
|
624
|
-
if h != actual:
|
|
625
|
-
cause, note = hash_mismatch_cause(meta, root=ctx.get("root"),
|
|
626
|
-
path=ctx.get("path"))
|
|
627
|
-
hits.append(_hit(node_id, "contradiction", "content_hash", None,
|
|
628
|
-
"索引声明指纹 %s ≠ 正文实算 %s(%s)"
|
|
629
|
-
% (h, actual, note), rule="hash_declared_vs_actual",
|
|
630
|
-
cause=cause,
|
|
631
|
-
line=(lines.get("content_hash") or (None, None))[0]))
|
|
632
|
-
fm_id = (ctx.get("fm") or {}).get("id")
|
|
633
|
-
if not _blank(fm_id) and str(fm_id) != str(node_id):
|
|
634
|
-
hits.append(_hit(node_id, "contradiction", "id", None,
|
|
635
|
-
"文件 frontmatter id=%s ≠ 索引键 %s" % (fm_id, node_id),
|
|
636
|
-
rule="id_declared_vs_index",
|
|
637
|
-
line=(lines.get("id") or (None, None))[0]))
|
|
638
|
-
return hits
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
LOCATORS = {
|
|
642
|
-
"missing_field": _loc_missing_field,
|
|
643
|
-
"weak_source": _loc_weak_source,
|
|
644
|
-
"stale": _loc_stale,
|
|
645
|
-
"observation_aged": _loc_observation_aged,
|
|
646
|
-
"dup": _loc_dup,
|
|
647
|
-
"template_flow": _loc_template_flow,
|
|
648
|
-
"contradiction": _loc_contradiction,
|
|
649
|
-
}
|
|
650
|
-
|
|
651
|
-
#: D1 **不定位**的判定:属语义层,交回 LLM 意见(不猜 = 不编造字符区间)
|
|
652
|
-
SEMANTIC_ONLY = {
|
|
653
|
-
"contradiction_semantic": "正文自相矛盾属语义判定;D1 只做确定性矛盾"
|
|
654
|
-
"(索引指纹 / 标识与事实不符)",
|
|
655
|
-
}
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
# 生效条件:始终返回 SEMANTIC_ONLY 中以 str(kind or "").strip() 为键查得的值,缺键时返回空串(空串表示可定位)。
|
|
659
|
-
def blindspot_reason(kind) -> str:
|
|
660
|
-
"""D1 拒绝定位的类别 → 理由(空串表示可定位)。"""
|
|
661
|
-
return SEMANTIC_ONLY.get(str(kind or "").strip(), "")
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
# ---------------------------- 主入口 ----------------------------
|
|
665
|
-
|
|
666
|
-
# 生效条件:items 中每项按 content=it.get("content") or ""、hash=str(it.get("hash") or "").strip() 或回落 NF.content_hash(content)、sk=WL.template_signature(content) or "" 预处理后,返回 {node_id: [同 hash 或同非空 sk 的其他 peer]};items 为空/None 返回空 dict。
|
|
667
|
-
def build_peers(items) -> dict:
|
|
668
|
-
"""`[{node_id, content, hash?, …}]` → `{node_id: [peer, …]}`。
|
|
669
|
-
|
|
670
|
-
只有**同内容指纹**或**同模板骨架**才互为对照——不是「同一批就是一组」。
|
|
671
|
-
组的作用域由调用方给定(批 / 包),与 M1 机械层「组 = 同一个包」的口径一致。
|
|
672
|
-
"""
|
|
673
|
-
prep = []
|
|
674
|
-
for it in items or []:
|
|
675
|
-
c = it.get("content") or ""
|
|
676
|
-
prep.append({"node_id": it.get("node_id"), "content": c, "meta": it.get("meta"),
|
|
677
|
-
"text": it.get("text"), "fm": it.get("fm"),
|
|
678
|
-
"hash": str(it.get("hash") or "").strip() or NF.content_hash(c),
|
|
679
|
-
"sk": WL.template_signature(c) or ""})
|
|
680
|
-
by_hash, by_sk = {}, {}
|
|
681
|
-
for p in prep:
|
|
682
|
-
by_hash.setdefault(p["hash"], []).append(p)
|
|
683
|
-
if p["sk"]:
|
|
684
|
-
by_sk.setdefault(p["sk"], []).append(p)
|
|
685
|
-
out = {}
|
|
686
|
-
for p in prep:
|
|
687
|
-
grp = {q["node_id"]: q for q in by_hash.get(p["hash"], [])
|
|
688
|
-
if q["node_id"] != p["node_id"]}
|
|
689
|
-
for q in (by_sk.get(p["sk"], []) if p["sk"] else []):
|
|
690
|
-
if q["node_id"] != p["node_id"]:
|
|
691
|
-
grp.setdefault(q["node_id"], q)
|
|
692
|
-
out[p["node_id"]] = list(grp.values())
|
|
693
|
-
return out
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
# 生效条件:对 issue_hint(None/str/dict 或其 list/tuple)逐 hint 分类,返回 (kinds, fields, blind):能 canonical_kind 到 ISSUE_KINDS 或 ADVISORY_KINDS 的入 kinds,blindspot_reason 非空的入 blind,其余非空 kind 与 field 入 fields;issue_hint 为 None 时 kinds/fields 为空集、blind 为空列表。
|
|
697
|
-
def _norm_hint(issue_hint):
|
|
698
|
-
"""issue_hint → `(kinds, fields, blindspot)`。
|
|
699
|
-
|
|
700
|
-
形态:`None`(全量)|str(issue_kind 或 field)|dict(issue_kind/field)
|
|
701
|
-
|上述的 list。
|
|
702
|
-
"""
|
|
703
|
-
kinds, fields, blind = set(), set(), []
|
|
704
|
-
hints = issue_hint if isinstance(issue_hint, (list, tuple)) else [issue_hint]
|
|
705
|
-
for h in hints:
|
|
706
|
-
if h is None:
|
|
707
|
-
continue
|
|
708
|
-
if isinstance(h, dict):
|
|
709
|
-
s = h.get("issue_kind") or h.get("kind") or ""
|
|
710
|
-
f = h.get("field")
|
|
711
|
-
else:
|
|
712
|
-
s, f = str(h).strip(), None
|
|
713
|
-
if s:
|
|
714
|
-
why = blindspot_reason(s)
|
|
715
|
-
if why:
|
|
716
|
-
blind.append("%s:%s" % (s, why))
|
|
717
|
-
else:
|
|
718
|
-
ck = canonical_kind(s)
|
|
719
|
-
if ck in ISSUE_KINDS or ck in ADVISORY_KINDS:
|
|
720
|
-
kinds.add(ck)
|
|
721
|
-
else:
|
|
722
|
-
fields.add(s) # 不是类别 → 当作字段过滤
|
|
723
|
-
if f:
|
|
724
|
-
fields.add(str(f).strip())
|
|
725
|
-
return kinds, fields, blind
|
|
726
|
-
|
|
727
|
-
|
|
728
|
-
# 生效条件:当 meta 与 content 均非 None 时不读盘;否则 root 为 None 抛 ValueError,经 load_node 读不到节点返回 {"hits": [], "blindspot": blind+["节点 %s 不在索引"%node_id], "load": None};读到时补 meta/content/text/fm/path,按 kinds 非空时只跑 sorted(kinds)、否则 fields 非空或 issue_hint 为 None 时跑 sorted(ISSUE_KINDS)、否则 run=[] 执行 LOCATORS,对命中按 fields 过滤、补 line 并排序后返回 {"hits": hits, "blindspot": blind, "load": {...}}。
|
|
729
|
-
def locate_ex(node_id, issue_hint=None, *, meta=None, content=None, fm=None, text=None,
|
|
730
|
-
root=None, index=None, peers=None, now=None, path=None,
|
|
731
|
-
field_layers=None) -> dict:
|
|
732
|
-
"""D1 定位(完整返回):`{"hits": [...], "blindspot": [...], "load": {...}}`。
|
|
733
|
-
|
|
734
|
-
`meta`/`content` 显式给出则**不读盘**(纯函数用法——测试与批量复用都走这条);
|
|
735
|
-
否则 `root` 必填,经 `load_node` 从索引+文件取。返回的 hits 已按
|
|
736
|
-
`(issue_kind, field, span起点, peer)` 排序,**确定性可复算**。
|
|
737
|
-
|
|
738
|
-
`path`(节点文件绝对路径——供 `contradiction` 判成因)、`field_layers`
|
|
739
|
-
(字段层门限,缺省 `field_layer_scope()`)可显式注入,便于纯函数测试。
|
|
740
|
-
"""
|
|
741
|
-
kinds, fields, blind = _norm_hint(issue_hint)
|
|
742
|
-
if meta is None or content is None:
|
|
743
|
-
if root is None:
|
|
744
|
-
raise ValueError("locate 需要 meta+content,或给出 root 以便读节点")
|
|
745
|
-
nd = load_node(node_id, root, index=index)
|
|
746
|
-
if nd is None:
|
|
747
|
-
return {"hits": [], "blindspot": blind + ["节点 %s 不在索引" % node_id],
|
|
748
|
-
"load": None}
|
|
749
|
-
if meta is None:
|
|
750
|
-
meta = nd["meta"]
|
|
751
|
-
if content is None:
|
|
752
|
-
content = nd["content"]
|
|
753
|
-
if text is None:
|
|
754
|
-
text = nd["text"]
|
|
755
|
-
if fm is None:
|
|
756
|
-
fm = nd["fm"]
|
|
757
|
-
if path is None:
|
|
758
|
-
path = nd.get("path")
|
|
759
|
-
meta = meta or {}
|
|
760
|
-
ctx = {"fm": fm or {}, "text": text, "root": root, "index": index,
|
|
761
|
-
"peers": peers or [], "now": time.time() if now is None else float(now),
|
|
762
|
-
"path": path, "field_layers": field_layers}
|
|
763
|
-
hits = []
|
|
764
|
-
# 定位范围:显式类别 → 只跑该类;给出字段 → 全量再由字段收窄;
|
|
765
|
-
# 无提示 → 全量。**提示全部落 blindspot 时范围为空**——不借机报别的类
|
|
766
|
-
# (否则「给了语义提示」会顺手退回全量,与「不猜」纪律相悖)。
|
|
767
|
-
if kinds:
|
|
768
|
-
run = sorted(kinds)
|
|
769
|
-
elif fields or issue_hint is None:
|
|
770
|
-
run = sorted(ISSUE_KINDS)
|
|
771
|
-
else:
|
|
772
|
-
run = []
|
|
773
|
-
for k in run:
|
|
774
|
-
fn = LOCATORS.get(k)
|
|
775
|
-
if fn is None:
|
|
776
|
-
continue
|
|
777
|
-
for h in fn(node_id, meta, content, ctx):
|
|
778
|
-
if fields and h.get("field") not in fields:
|
|
779
|
-
continue
|
|
780
|
-
if h.get("line") is None and h.get("span") is not None:
|
|
781
|
-
h["line"] = _line_of(text, content, h["span"])
|
|
782
|
-
hits.append(h)
|
|
783
|
-
hits.sort(key=lambda h: (h["issue_kind"], str(h.get("field") or ""),
|
|
784
|
-
-1 if h.get("span") is None else h["span"][0],
|
|
785
|
-
str(h.get("peer") or "")))
|
|
786
|
-
return {"hits": hits, "blindspot": blind,
|
|
787
|
-
"load": {"node_id": node_id, "meta": meta,
|
|
788
|
-
"content_len": len(content or "")}}
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
# 生效条件:给定 node_id(必填)与可选 issue_hint 及 kw 后,直接返回 locate_ex(node_id, issue_hint, **kw) 结果的 "hits" 列表。
|
|
792
|
-
def locate(node_id, issue_hint=None, **kw) -> list:
|
|
793
|
-
"""**D1 契约入口**:`locate(node_id, issue_hint) -> [{field, span, issue_kind, evidence}]`。"""
|
|
794
|
-
return locate_ex(node_id, issue_hint, **kw)["hits"]
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
# 生效条件:给定 hits 与 key 后,返回 hits 中每个 h.get(key) 字符串化取值到出现次数的字典(键升序);hits 为 None/空时返回空 dict。
|
|
798
|
-
def _counts(hits, key) -> dict:
|
|
799
|
-
out = {}
|
|
800
|
-
for h in hits or []:
|
|
801
|
-
k = h.get(key)
|
|
802
|
-
out[str(k)] = out.get(str(k), 0) + 1
|
|
803
|
-
return dict(sorted(out.items()))
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
# 生效条件:items 显式给出时不读盘;items 为 None 时 root 为 None 抛 ValueError,否则按 node_ids 逐节点 load_node 组装 items(读不到的记入 missing);随后对每个 item 调 locate_ex 汇总 hits 或 clean,并返回含 nodes/hits/by_kind/by_field/by_rule/clean/missing 的 dict。
|
|
807
|
-
def locate_many(node_ids=None, *, root=None, index=None, items=None,
|
|
808
|
-
issue_hint=None, now=None) -> dict:
|
|
809
|
-
"""批量定位:**同一批内**互为对照(`dup` / `template_flow` 的组 = 本批)。
|
|
810
|
-
|
|
811
|
-
`items` 显式给出 `[{node_id, content, meta?, text?, fm?, hash?}]` 则不再读盘。
|
|
812
|
-
返回 `{nodes, hits, by_kind, by_field, clean, missing}`;`clean` 是**无命中**的节点、
|
|
813
|
-
`missing` 是**读不到**的节点——两者分开报(「没问题」≠「没看」)。
|
|
814
|
-
"""
|
|
815
|
-
missing = []
|
|
816
|
-
if items is None:
|
|
817
|
-
if root is None:
|
|
818
|
-
raise ValueError("locate_many 需要 root(读盘)或 items(显式给出)")
|
|
819
|
-
items = []
|
|
820
|
-
for nid in node_ids or []:
|
|
821
|
-
nd = load_node(nid, root, index=index)
|
|
822
|
-
if nd is None:
|
|
823
|
-
missing.append(nid)
|
|
824
|
-
continue
|
|
825
|
-
items.append({"node_id": nid, "content": nd["content"] or "",
|
|
826
|
-
"text": nd["text"], "fm": nd["fm"], "meta": nd["meta"],
|
|
827
|
-
"path": nd.get("path"),
|
|
828
|
-
"hash": nd["meta"].get("content_hash")})
|
|
829
|
-
peers = build_peers(items)
|
|
830
|
-
now = time.time() if now is None else float(now)
|
|
831
|
-
hits, clean = [], []
|
|
832
|
-
for it in items:
|
|
833
|
-
nid = it["node_id"]
|
|
834
|
-
ex = locate_ex(nid, issue_hint, meta=it.get("meta") or {},
|
|
835
|
-
content=it.get("content") or "", fm=it.get("fm"),
|
|
836
|
-
text=it.get("text"), peers=peers.get(nid) or [], now=now,
|
|
837
|
-
root=root, path=it.get("path"))
|
|
838
|
-
if ex["hits"]:
|
|
839
|
-
hits.extend(ex["hits"])
|
|
840
|
-
else:
|
|
841
|
-
clean.append(nid)
|
|
842
|
-
hits.sort(key=lambda h: (str(h.get("node_id")), h["issue_kind"],
|
|
843
|
-
str(h.get("field") or ""),
|
|
844
|
-
-1 if h.get("span") is None else h["span"][0]))
|
|
845
|
-
return {"nodes": len(items), "hits": hits, "by_kind": _counts(hits, "issue_kind"),
|
|
846
|
-
"by_field": _counts(hits, "field"), "by_rule": _counts(hits, "rule"),
|
|
847
|
-
"clean": clean, "missing": missing}
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
# 生效条件:从 pkg.get("entries") or [] 取条目并跳过 e.get("node_id") 为空者,root 非 None 时逐个 load_node 取正文,取不到时回落 e.get("excerpt") or "",组装 items 调 locate_many 后返回其结果并附加 bundle_id 与 entries 数量;pkg 为 None 时 entries 为空列表。
|
|
851
|
-
def locate_package(pkg, *, root=None, index=None, issue_hint=None, now=None) -> dict:
|
|
852
|
-
"""M1 包 → 定位汇总(组的作用域 = 本包,与 M1 机械层同口径)。
|
|
853
|
-
|
|
854
|
-
正文优先读盘(准确);文件不可读时退回包内 `excerpt`(可能截断——宁可降级也不
|
|
855
|
-
假装读到了全文,`missing` 会如实记下读不到的节点)。
|
|
856
|
-
"""
|
|
857
|
-
entries = (pkg or {}).get("entries") or []
|
|
858
|
-
items = []
|
|
859
|
-
for e in entries:
|
|
860
|
-
nid = e.get("node_id")
|
|
861
|
-
if not nid:
|
|
862
|
-
continue
|
|
863
|
-
nd = load_node(nid, root, index=index) if root else None
|
|
864
|
-
body = (nd or {}).get("content")
|
|
865
|
-
items.append({"node_id": nid,
|
|
866
|
-
"content": body if body is not None else (e.get("excerpt") or ""),
|
|
867
|
-
"text": (nd or {}).get("text"), "fm": (nd or {}).get("fm"),
|
|
868
|
-
"meta": (nd or {}).get("meta") or e,
|
|
869
|
-
"path": (nd or {}).get("path") or e.get("path"),
|
|
870
|
-
"hash": ((nd or {}).get("meta") or {}).get("content_hash")
|
|
871
|
-
or e.get("content_hash")})
|
|
872
|
-
out = locate_many(items=items, issue_hint=issue_hint, now=now)
|
|
873
|
-
out["bundle_id"] = (pkg or {}).get("bundle_id")
|
|
874
|
-
out["entries"] = len(entries)
|
|
875
|
-
return out
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
# 生效条件:给定 hits 后返回 total=len(hits or [])、去重 node_id 数、排序后的 node_ids、以及 by_kind/by_field/by_rule 计数;hits 为 None/空时 total=0、nodes=0、node_ids=[]。
|
|
879
|
-
def summary(hits) -> dict:
|
|
880
|
-
"""命中汇总(审计留痕用)。"""
|
|
881
|
-
nodes = sorted({str(h.get("node_id")) for h in hits or []})
|
|
882
|
-
return {"total": len(hits or []), "nodes": len(nodes), "node_ids": nodes,
|
|
883
|
-
"by_kind": _counts(hits, "issue_kind"), "by_field": _counts(hits, "field"),
|
|
884
|
-
"by_rule": _counts(hits, "rule")}
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
# 生效条件:给定 hits 后逐条生成 Markdown 行并返回表头加各行:field/line/snippet 取 h.get(...) or "—"(假值回落 "—"),span 为 None 时显示 "—" 否则 "起-止",evidence 取 (h.get("evidence") or "").replace("|","\\|")(假值回落空串)。
|
|
888
|
-
def markdown_table(hits) -> str:
|
|
889
|
-
"""人工核对清单:每行一条命中,末列留空供核对者填判定(D1 验收抽样用)。"""
|
|
890
|
-
head = ("| # | node_id | issue_kind | field | span | line | 片段 | evidence | 人工判定 |\n"
|
|
891
|
-
"|---|---|---|---|---|---|---|---|---|\n")
|
|
892
|
-
rows = []
|
|
893
|
-
for i, h in enumerate(hits or [], 1):
|
|
894
|
-
sp = "—" if h.get("span") is None else "%d-%d" % (h["span"][0], h["span"][1])
|
|
895
|
-
rows.append("| %d | %s | %s | %s | %s | %s | %s | %s | |"
|
|
896
|
-
% (i, h.get("node_id"), h.get("issue_kind"), h.get("field") or "—",
|
|
897
|
-
sp, h.get("line") or "—",
|
|
898
|
-
(h.get("snippet") or "—").replace("|", "\\|"),
|
|
899
|
-
(h.get("evidence") or "").replace("|", "\\|")))
|
|
900
|
-
return head + "\n".join(rows)
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
# 生效条件:解析 argv(缺省 sys.argv)后,--root 为假值(含默认 os.environ.get("MDCG_ROOT") 为 None 或空串)时打印提示并返回 2;--node 追加列表为空时打印提示并返回 2;否则以 --root 与 --node 调 locate_many,并按 --json 或 --markdown 输出后返回 0,两者皆无则逐行打印命中与汇总后返回 0。
|
|
904
|
-
def main(argv=None) -> int:
|
|
905
|
-
ap = argparse.ArgumentParser(prog="python -m md_cg.mreview.locate",
|
|
906
|
-
description="记忆评审 M3 · D1 字段级定位")
|
|
907
|
-
ap.add_argument("--root", default=os.environ.get("MDCG_ROOT"),
|
|
908
|
-
help="真源根(缺省读环境变量 MDCG_ROOT)")
|
|
909
|
-
ap.add_argument("--node", action="append", default=[],
|
|
910
|
-
help="节点 id(可多次;同一批内互为对照)")
|
|
911
|
-
ap.add_argument("--kind", action="append", default=[], help="只定位该类问题")
|
|
912
|
-
ap.add_argument("--field", action="append", default=[], help="只保留该字段的命中")
|
|
913
|
-
ap.add_argument("--json", action="store_true", help="输出 JSON(缺省人读表)")
|
|
914
|
-
ap.add_argument("--markdown", action="store_true", help="输出人工核对清单")
|
|
915
|
-
a = ap.parse_args(argv)
|
|
916
|
-
if not a.root:
|
|
917
|
-
print("需要 --root 或环境变量 MDCG_ROOT", file=sys.stderr)
|
|
918
|
-
return 2
|
|
919
|
-
if not a.node:
|
|
920
|
-
print("需要 --node <id>(可多次)", file=sys.stderr)
|
|
921
|
-
return 2
|
|
922
|
-
rep = locate_many(a.node, root=a.root,
|
|
923
|
-
issue_hint=(list(a.kind) + list(a.field)) or None)
|
|
924
|
-
if a.json:
|
|
925
|
-
print(json.dumps(rep, ensure_ascii=False, indent=2))
|
|
926
|
-
return 0
|
|
927
|
-
if a.markdown:
|
|
928
|
-
print(markdown_table(rep["hits"]))
|
|
929
|
-
return 0
|
|
930
|
-
for h in rep["hits"]:
|
|
931
|
-
print("%-14s %-20s %-14s %s" % (h["node_id"], h["field"], h["issue_kind"],
|
|
932
|
-
h["evidence"]))
|
|
933
|
-
print("-- 命中 %d 条|干净 %d 条|缺项 %d 条"
|
|
934
|
-
% (len(rep["hits"]), len(rep["clean"]), len(rep["missing"])))
|
|
935
|
-
return 0
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
if __name__ == "__main__":
|
|
939
|
-
sys.exit(main())
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""记忆评审流水线 · M3 定位模块(D1 字段级定位,确定性、零写入)。
|
|
3
|
+
|
|
4
|
+
真源:docs/记忆评审系统_立项设计与施工交接_20260915.md §5 D1。
|
|
5
|
+
|
|
6
|
+
D1 契约:
|
|
7
|
+
|
|
8
|
+
locate(node_id, issue_hint) -> [{"field", "span", "issue_kind", "evidence"}, ...]
|
|
9
|
+
|
|
10
|
+
* **field** —— frontmatter 字段名,或 `"content"`(正文)
|
|
11
|
+
* **span** —— 命中处的字符区间 `[start, end)`(半开),**相对正文 content**;
|
|
12
|
+
纯字段级缺失无字符区间可指 → `None`
|
|
13
|
+
* **issue_kind** —— dup / missing_field / stale / contradiction / weak_source / template_flow
|
|
14
|
+
(`ISSUE_KINDS`=问题面);另有 `ADVISORY_KINDS`(观测面,可显式定位
|
|
15
|
+
但**不进默认全量、不进评审告警面**,见下「stale 与 observation_aged」)
|
|
16
|
+
* **evidence** —— 白箱判据:为什么算问题(人可复核、机器可断言)
|
|
17
|
+
|
|
18
|
+
定位的职责边界:M2 的意见说「这一条有问题」,D1 回答「问题在这一条的哪个字段/哪一段」。
|
|
19
|
+
故本模块只做**确定性可判定**的事——语义级矛盾(正文自相矛盾)**不猜**,如实返回
|
|
20
|
+
BLINDSPOT 项交回 LLM 意见层(`status="blindspot"`,见 `locate()` 返回值)。
|
|
21
|
+
|
|
22
|
+
口径全部引用既有唯一真源,**不在本模块另立一份**(防「术语双写法」漂移):
|
|
23
|
+
|
|
24
|
+
=============== ==================================================================
|
|
25
|
+
字段/正文结构 ``nodefile``(CCG_MARKS / CONDITION_SLOTS / VERIFICATION_BASIS /
|
|
26
|
+
loads / content_hash / is_placeholder_text / time_window_text)
|
|
27
|
+
模板骨架 ``writelimit._skeleton`` / ``MIN_SKELETON``
|
|
28
|
+
赛道×基底相容 ``crosscheck``(与 M1 `ruleset._chk_basis_licensed` 同源)
|
|
29
|
+
字段层门限 ``ruleset``(即 `rules/*.json` 的 `matcher.layer`,见
|
|
30
|
+
`field_layer_scope`——**派生**,不在此另立一份)
|
|
31
|
+
索引快照 ``conformance.load_index``
|
|
32
|
+
=============== ==================================================================
|
|
33
|
+
|
|
34
|
+
与 M1(`ruleset` 机械层)的用词归并:D1 说 `dup`,M1 的 `dup_hash_group` 说
|
|
35
|
+
`dup_content`——同一件事两种写法,由 `KIND_ALIASES` 单一归口(`canonical_kind`)。
|
|
36
|
+
|
|
37
|
+
超出 D1 最小契约的**增强键**(供人工核对直接跳文件/跳行,不影响四键契约):
|
|
38
|
+
`line`(节点文件原文 1-based 行号)、`sentence`(正文句索引,0 基)、
|
|
39
|
+
`snippet`(命中片段)、`rule`(触发定位的判据名,审计用)、`peer`(同组对照节点)、
|
|
40
|
+
`cause`(`contradiction` 的**成因**判定,见 `hash_mismatch_cause`——同一条命中
|
|
41
|
+
可能是「真不一致」也可能是「索引快照滞后」,两种成因不分会让复核者误判为数据损坏)。
|
|
42
|
+
|
|
43
|
+
**`stale` 与 `observation_aged` 的判据源分工(2026-09-16 修正)**:
|
|
44
|
+
|
|
45
|
+
* `stale`(**问题面**)—— **依赖存在性**:`code_ref`/`doc_ref` 指的源文件已不存在
|
|
46
|
+
(悬空)或已漂移(区间哈希不符)。判据直接调 `refindex.probe_ref`,**不另立一份**
|
|
47
|
+
(防「术语双写法」漂移)。代码知识的真值挂在源文件上,源没了/改了才是适用边界越出。
|
|
48
|
+
* `observation_aged`(**观测面**,`ADVISORY_KINDS`,**不进默认全量**)——
|
|
49
|
+
`condition_space.time_window` 已过。这是**观测时刻**不是失效声明:写入端
|
|
50
|
+
(`mdcg.add`,见其 `OBSERVATION_WINDOW_SEC`)在未给 time_window 时以**写入时刻**
|
|
51
|
+
自动填 1 小时窗,故真库中「已过期」绝大多数是「写入超过 1 小时」,与知识是否失效
|
|
52
|
+
无关——把它当问题投进评审告警面就是系统性误报(实测:全库 1453 条过期里 1434 条
|
|
53
|
+
属默认窗填充)。需要时效判定用 `stale`(依赖存在性),不是这里。
|
|
54
|
+
"""
|
|
55
|
+
from __future__ import annotations
|
|
56
|
+
|
|
57
|
+
import argparse
|
|
58
|
+
import json
|
|
59
|
+
import os
|
|
60
|
+
import re
|
|
61
|
+
import sys
|
|
62
|
+
import time
|
|
63
|
+
|
|
64
|
+
from .. import conformance as CF
|
|
65
|
+
from .. import crosscheck as CC
|
|
66
|
+
from .. import nodefile as NF
|
|
67
|
+
from .. import refindex as RI
|
|
68
|
+
from .. import writelimit as WL
|
|
69
|
+
from . import ruleset as RS
|
|
70
|
+
|
|
71
|
+
__all__ = ["ISSUE_KINDS", "ADVISORY_KINDS", "KIND_ALIASES", "canonical_kind", "locate",
|
|
72
|
+
"locate_many", "locate_package", "sentence_spans", "mark_spans", "load_node",
|
|
73
|
+
"main", "field_layer_scope", "hash_mismatch_cause", "HASH_CAUSES"]
|
|
74
|
+
|
|
75
|
+
#: D1 的问题类别(真源:立项文档 §5 D1)——**问题面**:命中即「知识有问题」
|
|
76
|
+
ISSUE_KINDS = ("dup", "missing_field", "stale", "contradiction",
|
|
77
|
+
"weak_source", "template_flow")
|
|
78
|
+
|
|
79
|
+
#: **观测面**(咨询级):可显式定位,但**不进默认全量定位、不进评审告警面**。
|
|
80
|
+
#: 判据:本类命中说的是「观测手段的副产品」(如「这条记忆写入超过 1 小时」),
|
|
81
|
+
#: 不是「知识有问题」——把它混进 ISSUE_KINDS 会让每一次写入在 1 小时后自动变成
|
|
82
|
+
#: 待评审问题(系统性误报)。故与问题面分表,只有显式点名才产出。
|
|
83
|
+
ADVISORY_KINDS = ("observation_aged",)
|
|
84
|
+
|
|
85
|
+
#: M1(ruleset 机械层)与 D1 的用词归并——两处说的是同一件事,不许各说各话
|
|
86
|
+
KIND_ALIASES = {"dup_content": "dup", # M1 `dup_hash_group`
|
|
87
|
+
"missing": "missing_field",
|
|
88
|
+
"flow": "template_flow",
|
|
89
|
+
"template_flow_digits_only": "template_flow"}
|
|
90
|
+
|
|
91
|
+
#: 扫描 frontmatter 的字段清单(真源字段名取自 nodefile 口径,不自由发明)
|
|
92
|
+
FM_SCAN_FIELDS = ("role", "tags", "importance", "evidence_count",
|
|
93
|
+
"verification_basis", "lifecycle_state", "condition_space")
|
|
94
|
+
|
|
95
|
+
#: 判为「同模板流水」所需的最少同骨架句数——单句偶合不算流水(防误报)
|
|
96
|
+
MIN_FLOW_SENTENCES = 2
|
|
97
|
+
|
|
98
|
+
#: 一句话摘要的最大长度
|
|
99
|
+
SNIPPET_MAX = 120
|
|
100
|
+
|
|
101
|
+
#: 索引快照与节点文件的 mtime 差**小于**此值 → 视为同一批写入,不判「快照滞后」
|
|
102
|
+
MTIME_TOLERANCE = 1.0
|
|
103
|
+
|
|
104
|
+
#: `content_hash` 不一致的**成因**(确定性判据,不猜;见 `hash_mismatch_cause`)
|
|
105
|
+
HASH_CAUSES = ("index_lag", "true_mismatch", "unknown")
|
|
106
|
+
|
|
107
|
+
#: `refindex.probe_ref` 的五态里,哪些构成「依赖已失效」(`stale` 判据,见 `_loc_stale`):
|
|
108
|
+
#: 只有这两态说明**载体真的没了/改了**;`unresolved`(拿不到 root)与 `error`(读盘失败)
|
|
109
|
+
#: 是观测手段不足,按「不猜」纪律不判。
|
|
110
|
+
_REF_DEAD_STATUSES = ("dangling", "stale")
|
|
111
|
+
|
|
112
|
+
#: 字段层门限缓存(规则库是包内数据文件,进程内不变 → 惰性算一次)
|
|
113
|
+
_FIELD_LAYER_CACHE = None
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
# 生效条件:kind 经 str(kind or "").strip() 得 k(None/空串等假值 → 空串 ""),k 命中 KIND_ALIASES 时返回其规范名,否则原样返回 k。
|
|
117
|
+
def canonical_kind(kind) -> str:
|
|
118
|
+
"""M1/D1 用词 → D1 规范名(未知原样返回,不假装认路)。"""
|
|
119
|
+
k = str(kind or "").strip()
|
|
120
|
+
return KIND_ALIASES.get(k, k)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
# 生效条件:仅当 rules 与 rules_dir 均为 None 且模块级 _FIELD_LAYER_CACHE 非 None 时直接返回该缓存;否则遍历 rules(dict 取其 "rules" 键的列表、非 dict 直接 list(rules))或 rules_dir 经 RS.load_rules 取得的 rules 列表,从限了 matcher.layer 的 mechanical 规则中收集 field_absent 的 spec["field"] 与 evidence_zero 的 evidence_count,返回 {字段: 排序列},且只在 rules 与 rules_dir 均为 None 时写回 _FIELD_LAYER_CACHE。
|
|
124
|
+
def field_layer_scope(rules=None, rules_dir=None) -> dict:
|
|
125
|
+
"""字段 → 适用层清单(**派生**自 M1 规则库,不在此另立一份)。
|
|
126
|
+
|
|
127
|
+
真源 = `md_cg/mreview/rules/*.json` 的 `matcher.layer`:只收录**限了层**的
|
|
128
|
+
机械规则字段(`field_absent` 取 `spec["field"]`、`evidence_zero` 取
|
|
129
|
+
`evidence_count`);未限层的规则 = 全层适用,不入表。
|
|
130
|
+
|
|
131
|
+
理由:字段范围两处各写一份 ⇒ 必然漂移(同「术语双写法」)。派生使 M1 改规则
|
|
132
|
+
时 M3 自动跟随。规则库损坏 → `load_rules` 抛错(fail-closed,不静默降级成全层)。
|
|
133
|
+
"""
|
|
134
|
+
global _FIELD_LAYER_CACHE
|
|
135
|
+
if rules is None and rules_dir is None and _FIELD_LAYER_CACHE is not None:
|
|
136
|
+
return _FIELD_LAYER_CACHE
|
|
137
|
+
if isinstance(rules, dict):
|
|
138
|
+
rl = list(rules.get("rules") or [])
|
|
139
|
+
elif rules is not None:
|
|
140
|
+
rl = list(rules)
|
|
141
|
+
else:
|
|
142
|
+
rl = list(RS.load_rules(rules_dir)["rules"])
|
|
143
|
+
scope = {}
|
|
144
|
+
for r in rl:
|
|
145
|
+
layers = set((r.get("matcher") or {}).get("layer") or [])
|
|
146
|
+
if not layers:
|
|
147
|
+
continue
|
|
148
|
+
for spec in r.get("mechanical") or []:
|
|
149
|
+
chk = spec.get("check")
|
|
150
|
+
if chk == "field_absent" and spec.get("field"):
|
|
151
|
+
scope.setdefault(str(spec["field"]), set()).update(layers)
|
|
152
|
+
elif chk == "evidence_zero":
|
|
153
|
+
scope.setdefault("evidence_count", set()).update(layers)
|
|
154
|
+
out = {k: sorted(v) for k, v in sorted(scope.items())}
|
|
155
|
+
if rules is None and rules_dir is None:
|
|
156
|
+
_FIELD_LAYER_CACHE = out
|
|
157
|
+
return out
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
# 生效条件:(scope or {}).get(field) 为假值(scope 为 None、空 dict 或 field 不在其中)→ 返回 True;want 为真值时仅当 str((meta or {}).get("layer")) 在 set(want) 内返回 True,否则 False。
|
|
161
|
+
def _layer_ok(field, scope, meta) -> bool:
|
|
162
|
+
"""字段层门限:**限层的字段只在该层的节点上检查**。
|
|
163
|
+
|
|
164
|
+
口径与 M1 `ruleset._scope`(`str(e.get("layer")) in want`)逐字一致——缺层
|
|
165
|
+
(`None`)不在任何层清单内 ⇒ 不检查。两处判据同源,防「越层误报」:
|
|
166
|
+
M1 不在某层报的问题,M3 也不该报。
|
|
167
|
+
"""
|
|
168
|
+
want = (scope or {}).get(field)
|
|
169
|
+
if not want:
|
|
170
|
+
return True
|
|
171
|
+
return str((meta or {}).get("layer")) in set(want)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
# 生效条件:os.path.getmtime(path) 成功则返回该 mtime;抛 OSError 或 TypeError → 返回 None。
|
|
175
|
+
def _mtime(path):
|
|
176
|
+
"""文件 mtime(不可读 → `None`,不假装知道)。"""
|
|
177
|
+
try:
|
|
178
|
+
return os.path.getmtime(path)
|
|
179
|
+
except (OSError, TypeError):
|
|
180
|
+
return None
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
# 生效条件:root 或 rel(path 为真值时取 path,否则取 meta 的 "path")为空 → 返回 ("unknown", "无 root/path 可用,取不到盘上 mtime,成因未判定");两者都有时取节点文件与 CF.INDEX_FILE 的 mtime,任一为 None → 返回 ("unknown", "节点文件或索引快照不可读,成因未判定"),节点 mtime 减索引 mtime 之差 > MTIME_TOLERANCE → 返回 ("index_lag", …),否则返回 ("true_mismatch", …)。
|
|
184
|
+
def hash_mismatch_cause(meta, *, root=None, path=None) -> tuple:
|
|
185
|
+
"""`content_hash` 声明值 ≠ 正文实算值 → **成因**判定(靠 mtime 证据,不猜)。
|
|
186
|
+
|
|
187
|
+
→ `(cause, note)`;cause ∈ `HASH_CAUSES`;取不到盘上证据 → `"unknown"`。
|
|
188
|
+
|
|
189
|
+
写路径真源(`md_cg/mdcg.py`):写走 `_stage()` 落**分片日志**,`rebuild_index()`
|
|
190
|
+
才写 `_index.json` 快照——故「节点文件比索引快照新」是**正常写路径现象**
|
|
191
|
+
(快照滞后),不是文件被篡改;把两种成因合并成一句「索引与文件不一致」会让
|
|
192
|
+
复核者误判为数据损坏(实测滞后可达 14s 量级)。
|
|
193
|
+
"""
|
|
194
|
+
rel = path if path else (meta or {}).get("path")
|
|
195
|
+
if not root or not rel:
|
|
196
|
+
return "unknown", "无 root/path 可用,取不到盘上 mtime,成因未判定"
|
|
197
|
+
fp = str(rel) if os.path.isabs(str(rel)) else os.path.join(str(root), str(rel))
|
|
198
|
+
fm_t = _mtime(fp)
|
|
199
|
+
idx_t = _mtime(os.path.join(str(root), CF.INDEX_FILE))
|
|
200
|
+
if fm_t is None or idx_t is None:
|
|
201
|
+
return "unknown", "节点文件或索引快照不可读,成因未判定"
|
|
202
|
+
delta = fm_t - idx_t
|
|
203
|
+
if delta > MTIME_TOLERANCE:
|
|
204
|
+
return "index_lag", ("节点文件比索引快照新 %.2fs——分片日志已写、_index.json "
|
|
205
|
+
"未 rebuild(快照滞后属正常写路径现象)" % delta)
|
|
206
|
+
return "true_mismatch", ("节点文件不晚于索引快照(差 %.2fs)——非快照滞后,"
|
|
207
|
+
"索引声明与文件真源相抵触" % delta)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
# ---------------------------- 基础工具(纯函数) ----------------------------
|
|
211
|
+
|
|
212
|
+
# 生效条件:v 为 list/tuple/dict 时返回 not v(空容器 → True);其余类型返回 v is None 或 str(v).strip() 为 "" 或 "None"。
|
|
213
|
+
def _blank(v) -> bool:
|
|
214
|
+
if isinstance(v, (list, tuple, dict)):
|
|
215
|
+
return not v
|
|
216
|
+
return v is None or str(v).strip() in ("", "None")
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
# 生效条件:int(float(v)) 可算(含数字字符串)则返回该整数;抛 TypeError 或 ValueError(含 v 为 None、非数字串)→ 返回 None。
|
|
220
|
+
def _as_int(v):
|
|
221
|
+
try:
|
|
222
|
+
return int(float(v))
|
|
223
|
+
except (TypeError, ValueError):
|
|
224
|
+
return None
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
# 生效条件:float(v) 可算则返回该浮点数;抛 TypeError 或 ValueError(含 v 为 None、非数字串)→ 返回 None。
|
|
228
|
+
def _as_float(v):
|
|
229
|
+
try:
|
|
230
|
+
return float(v)
|
|
231
|
+
except (TypeError, ValueError):
|
|
232
|
+
return None
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
_SENT_END = "。!?;!?;\n"
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
# 生效条件:content 为假值(None/空串)按 "" 处理,逐字符命中 _SENT_END 切出且 seg.strip() 非空的段以 len(out) 为序号追加 (索引, start, i+1, 原文),末尾 tail.strip() 非空亦追加;无合格段 → 返回空列表。
|
|
239
|
+
def sentence_spans(content: str) -> list:
|
|
240
|
+
"""正文 → `[(句索引, start, end, 原文)]`;空句不编号(索引连续,确定性)。"""
|
|
241
|
+
text, out, start = content or "", [], 0
|
|
242
|
+
for i, ch in enumerate(text):
|
|
243
|
+
if ch in _SENT_END:
|
|
244
|
+
seg = text[start:i + 1]
|
|
245
|
+
if seg.strip():
|
|
246
|
+
out.append((len(out), start, i + 1, seg))
|
|
247
|
+
start = i + 1
|
|
248
|
+
tail = text[start:]
|
|
249
|
+
if tail.strip():
|
|
250
|
+
out.append((len(out), start, len(text), tail))
|
|
251
|
+
return out
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
#: 正文要素行:`# 功能名:…` / `# 生效条件: …`(间隔与分隔符两形态都收)
|
|
255
|
+
_RE_MARK = re.compile(r"^#[ \t]*(?P<mark>[^::\n]{1,16})[::][^\n]*", re.M)
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
# 生效条件:在 content(假值 → "")上用模块级 _RE_MARK 迭代匹配,以 m.group("mark").strip() 为键 setdefault 记下首次出现的 [m.start(), m.end());无匹配(含 content 为假值)→ 返回空 dict。
|
|
259
|
+
def mark_spans(content: str) -> dict:
|
|
260
|
+
"""正文 CCG 要素行 → `{要素名: [start, end)}`;同要素取首次出现(确定性)。"""
|
|
261
|
+
out = {}
|
|
262
|
+
for m in _RE_MARK.finditer(content or ""):
|
|
263
|
+
out.setdefault(m.group("mark").strip(), [m.start(), m.end()])
|
|
264
|
+
return out
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
# 生效条件:(text or "").split("\n") 的每行 strip 后非空、不以 "#" 开头且含 ":" 时,取首个冒号前的键 k,k 非空且尚未入表则记 (1-based 行号, [该行起始偏移, 起始偏移+len(line)]);text 为假值或无合格行 → 返回空 dict。
|
|
268
|
+
def key_line_spans(text: str) -> dict:
|
|
269
|
+
"""节点文件原文 → `{键: (1-based 行号, [start, end))}`。
|
|
270
|
+
|
|
271
|
+
只认**无 `#` 前缀**且含 `:` 的行——即 frontmatter 行;`---` 分隔线无冒号,
|
|
272
|
+
自然被排除。同键取首次出现。
|
|
273
|
+
"""
|
|
274
|
+
out, pos = {}, 0
|
|
275
|
+
for i, line in enumerate((text or "").split("\n")):
|
|
276
|
+
s = line.strip()
|
|
277
|
+
if s and not s.startswith("#") and ":" in s:
|
|
278
|
+
k = s.split(":", 1)[0].strip()
|
|
279
|
+
if k and k not in out:
|
|
280
|
+
out[k] = (i + 1, [pos, pos + len(line)])
|
|
281
|
+
pos += len(line) + 1
|
|
282
|
+
return out
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
# 生效条件:index 为 dict 时其 "nodes" 为 dict 则返回 index["nodes"],否则返回 index 本身;index 为 None/非 dict 时调 CF.load_index(root),抛 OSError 或 ValueError → 返回 {},返回值为 dict 且其 "nodes" 为 dict → 返回该 "nodes",是 dict → 原样返回,否则返回 {}。
|
|
286
|
+
def _index(root, index=None) -> dict:
|
|
287
|
+
"""索引节点表 `{node_id: meta}`(兼容 load_index 的 `{"nodes": …}` 形态)。"""
|
|
288
|
+
if isinstance(index, dict):
|
|
289
|
+
return index.get("nodes") if isinstance(index.get("nodes"), dict) else index
|
|
290
|
+
try:
|
|
291
|
+
idx = CF.load_index(root)
|
|
292
|
+
except (OSError, ValueError):
|
|
293
|
+
return {}
|
|
294
|
+
if isinstance(idx, dict) and isinstance(idx.get("nodes"), dict):
|
|
295
|
+
return idx["nodes"]
|
|
296
|
+
return idx if isinstance(idx, dict) else {}
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
# 生效条件:_index(root, index) 中 node_id 对应值非 dict → 返回 None;否则用 meta.get("path")(绝对路径直接用,否则 join(root, str(rel or "")))读文件——OSError 时返回 content/text 为 None、fm 为 {} 的 dict,成功则把 NF.loads(text) 得到的 content 与 fm(假值 → {})连同 meta/path/text 一并返回。
|
|
300
|
+
def load_node(node_id, root, *, index=None) -> dict:
|
|
301
|
+
"""读一个节点(索引 meta + 文件 frontmatter + 正文原文)。
|
|
302
|
+
|
|
303
|
+
→ `{"node_id","meta","fm","content","text","path"}`;索引无此项 → `None`。
|
|
304
|
+
`meta` 是**索引快照**(含索引侧 content_hash,矛盾判定要用),`fm` 是**文件真源**。
|
|
305
|
+
"""
|
|
306
|
+
nodes = _index(root, index)
|
|
307
|
+
meta = nodes.get(node_id)
|
|
308
|
+
if not isinstance(meta, dict):
|
|
309
|
+
return None
|
|
310
|
+
rel = meta.get("path")
|
|
311
|
+
fp = rel if (rel and os.path.isabs(str(rel))) else os.path.join(root, str(rel or ""))
|
|
312
|
+
try:
|
|
313
|
+
with open(fp, encoding="utf-8") as f:
|
|
314
|
+
text = f.read()
|
|
315
|
+
except OSError:
|
|
316
|
+
return {"node_id": node_id, "meta": dict(meta), "fm": {}, "content": None,
|
|
317
|
+
"text": None, "path": fp}
|
|
318
|
+
fm, content = NF.loads(text)
|
|
319
|
+
return {"node_id": node_id, "meta": dict(meta), "fm": fm or {}, "content": content,
|
|
320
|
+
"text": text, "path": fp}
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
# 生效条件:span 为 None → 返回 "";否则取 (text or "")[span[0]:span[1]] 去空白并把换行替换为 "⏎",长度超 SNIPPET_MAX 时截断并追加 "…"。
|
|
324
|
+
def _snippet(text, span) -> str:
|
|
325
|
+
"""命中片段(供人工核对肉眼确认「指的是不是这一句」)。"""
|
|
326
|
+
if span is None:
|
|
327
|
+
return ""
|
|
328
|
+
seg = (text or "")[span[0]:span[1]].strip().replace("\n", "⏎")
|
|
329
|
+
return seg[:SNIPPET_MAX] + ("…" if len(seg) > SNIPPET_MAX else "")
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
# 生效条件:任意 node_id/kind/field/span/evidence 均原样写入返回 dict 的 node_id/issue_kind/field/span/evidence 键,可选 rule/line/sentence/snippet/peer/cause/severity 未传时为 None、status 未传时为 "located",不做任何校验。
|
|
333
|
+
def _hit(node_id, kind, field, span, evidence, *, rule=None, line=None,
|
|
334
|
+
sentence=None, snippet=None, peer=None, cause=None, status="located",
|
|
335
|
+
severity=None) -> dict:
|
|
336
|
+
"""命中行(四键契约 + 增强键)。
|
|
337
|
+
|
|
338
|
+
`severity` 是**观测面专用**的等级标注(`ADVISORY_KINDS` 命中带 `"info"`):
|
|
339
|
+
问题面命中不带(问题即问题,无等级可降);本键让下游一眼分清「这是观测
|
|
340
|
+
副产品」与「这是待修的问题」,不必靠 issue_kind 名字去猜。
|
|
341
|
+
"""
|
|
342
|
+
return {"node_id": node_id, "field": field, "span": span, "issue_kind": kind,
|
|
343
|
+
"evidence": evidence, "rule": rule, "line": line, "sentence": sentence,
|
|
344
|
+
"snippet": snippet, "peer": peer, "cause": cause, "status": status,
|
|
345
|
+
"severity": severity}
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
# 生效条件:text 为真值、span 与 content 均非 None 且 text.find(content)>=0 时,返回 text.count("\n",0,min(off+span[0],len(text)))+1 的 1-based 行号;text 假值或 span/content 为 None 或 content 未找到时返回 None。
|
|
349
|
+
def _line_of(text, content, span):
|
|
350
|
+
"""正文区间 → 节点文件原文的 1-based 行号(供人工核对直接跳文件)。"""
|
|
351
|
+
if not text or span is None or content is None:
|
|
352
|
+
return None
|
|
353
|
+
off = text.find(content)
|
|
354
|
+
if off < 0:
|
|
355
|
+
return None
|
|
356
|
+
return text.count("\n", 0, min(off + span[0], len(text))) + 1
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
# ---------------------------- 定位器注册表 ----------------------------
|
|
360
|
+
#
|
|
361
|
+
# 每个定位器是**纯函数**:只吃 (node_id, meta, content, ctx),只吐 hits。
|
|
362
|
+
# 新增判据 = 加函数 + 注册,不改 locate 主流程(与 M1「引擎冻结、规则可增删」同构)。
|
|
363
|
+
# ctx = {"fm", "text", "root", "index", "peers", "now"}
|
|
364
|
+
|
|
365
|
+
# 生效条件:在 node_id/meta/content/ctx 下,对 FM_SCAN_FIELDS 中除 verification_basis、condition_space 外且 _layer_ok(f,scope,meta) 为真的字段,若 meta.get(f) 与 ctx.get("fm") 或 {} 中的同名字段皆 _blank 则记 field_absent;对过 _layer_ok 且 _as_int(meta.get("evidence_count"))==0 记 evidence_zero;对 meta.get("importance") 非 None 且 _as_float 为 None 或不在 [0.0,1.0] 记 field_invalid;对 condition_space 经 meta 或回落 ctx.get("fm") 后 NF.condition_space_missing 非空记 condition_slots;对正文缺 NF.CCG_MARKS 行记 ccg_incomplete,返回这些命中列表。
|
|
366
|
+
def _loc_missing_field(node_id, meta, content, ctx):
|
|
367
|
+
"""frontmatter/正文结构字段为空(判据与 M1 `field_absent`/`evidence_zero` 同源)。
|
|
368
|
+
|
|
369
|
+
**层门限**:限层的字段(如 `role`/`evidence_count` 限 `knowledge`)只在
|
|
370
|
+
该层节点上检查——与 M1 `_scope` 同口径,否则 M1 不报的问题会被 M3 越层报出。
|
|
371
|
+
"""
|
|
372
|
+
fm = ctx.get("fm") or {}
|
|
373
|
+
scope = ctx.get("field_layers")
|
|
374
|
+
if scope is None:
|
|
375
|
+
scope = field_layer_scope()
|
|
376
|
+
lines = key_line_spans(ctx.get("text") or "")
|
|
377
|
+
marks = mark_spans(content or "")
|
|
378
|
+
hits = []
|
|
379
|
+
|
|
380
|
+
for f in FM_SCAN_FIELDS:
|
|
381
|
+
if f in ("verification_basis", "condition_space"):
|
|
382
|
+
continue # 基底空属 weak_source;四槽缺失单独报(要点名缺哪几槽)
|
|
383
|
+
if not _layer_ok(f, scope, meta):
|
|
384
|
+
continue # 越层不报(层门限真源 = M1 规则库 matcher.layer)
|
|
385
|
+
if _blank(meta.get(f)) and _blank(fm.get(f)):
|
|
386
|
+
hits.append(_hit(node_id, "missing_field", f, None,
|
|
387
|
+
"字段 %s 为空(索引与文件两处皆空)" % f,
|
|
388
|
+
rule="field_absent",
|
|
389
|
+
line=(lines.get(f) or (None, None))[0]))
|
|
390
|
+
|
|
391
|
+
if _layer_ok("evidence_count", scope, meta) \
|
|
392
|
+
and _as_int(meta.get("evidence_count")) == 0:
|
|
393
|
+
hits.append(_hit(node_id, "missing_field", "evidence_count", None,
|
|
394
|
+
"evidence_count=0(无验证证据计数)", rule="evidence_zero",
|
|
395
|
+
line=(lines.get("evidence_count") or (None, None))[0]))
|
|
396
|
+
|
|
397
|
+
imp = meta.get("importance")
|
|
398
|
+
if imp is not None:
|
|
399
|
+
fv = _as_float(imp)
|
|
400
|
+
if fv is None or not 0.0 <= fv <= 1.0:
|
|
401
|
+
hits.append(_hit(node_id, "missing_field", "importance", None,
|
|
402
|
+
"importance=%r 不在 [0,1](字段存在但不可用)"
|
|
403
|
+
% (imp,), rule="field_invalid",
|
|
404
|
+
line=(lines.get("importance") or (None, None))[0]))
|
|
405
|
+
|
|
406
|
+
cs = meta.get("condition_space")
|
|
407
|
+
if not isinstance(cs, dict):
|
|
408
|
+
cs = fm.get("condition_space")
|
|
409
|
+
miss = NF.condition_space_missing(cs)
|
|
410
|
+
if miss:
|
|
411
|
+
span = marks.get("生效条件")
|
|
412
|
+
hits.append(_hit(node_id, "missing_field", "condition_space", span,
|
|
413
|
+
"条件空间缺槽 %s(%d/4 已声明)——四槽不全不构成生效条件"
|
|
414
|
+
% ("/".join(miss), len(NF.CONDITION_SLOTS) - len(miss)),
|
|
415
|
+
rule="condition_slots",
|
|
416
|
+
snippet=_snippet(content, span)))
|
|
417
|
+
|
|
418
|
+
# 正文 CCG 要素缺行:行不存在 → 无字符区间可指(span=None),点名缺哪几行
|
|
419
|
+
missing_marks = [m for m in NF.CCG_MARKS if m not in marks]
|
|
420
|
+
if missing_marks:
|
|
421
|
+
hits.append(_hit(node_id, "missing_field", "content", None,
|
|
422
|
+
"正文缺 CCG 要素行 %s(# 要素名:值)——缺声明即缺证据"
|
|
423
|
+
% "/".join(missing_marks), rule="ccg_incomplete"))
|
|
424
|
+
return hits
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
# 生效条件:meta.get("verification_basis") 为空时回落 ctx.get("fm") 或 {} 的 verification_basis,若两者皆 _blank 返回 basis_absent 命中;非空但 str(basis) 不在 NF.VERIFICATION_BASIS 返回 basis_enum 命中;在枚举内时按 CC.classify_track 依 meta.get("layer")、meta.get("tags") 与 content 判赛道,若 CC.basis_licensed 为假返回 basis_licensed 命中;否则返回 []。
|
|
428
|
+
def _loc_weak_source(node_id, meta, content, ctx):
|
|
429
|
+
"""验证基底缺失/越枚举/与赛道不相容(与 M1 `basis_licensed` 判据逐字同源)。"""
|
|
430
|
+
fm = ctx.get("fm") or {}
|
|
431
|
+
basis = meta.get("verification_basis")
|
|
432
|
+
if _blank(basis):
|
|
433
|
+
basis = fm.get("verification_basis")
|
|
434
|
+
_ln = key_line_spans(ctx.get("text") or "").get("verification_basis")
|
|
435
|
+
ln = (_ln or (None, None))[0]
|
|
436
|
+
|
|
437
|
+
if _blank(basis):
|
|
438
|
+
return [_hit(node_id, "weak_source", "verification_basis", None,
|
|
439
|
+
"verification_basis 为空——无验证基底的断言不可裁决",
|
|
440
|
+
rule="basis_absent", line=ln)]
|
|
441
|
+
if str(basis) not in NF.VERIFICATION_BASIS:
|
|
442
|
+
return [_hit(node_id, "weak_source", "verification_basis", None,
|
|
443
|
+
"基底 %s 不在合法枚举(%s)"
|
|
444
|
+
% (basis, "/".join(NF.VERIFICATION_BASIS)),
|
|
445
|
+
rule="basis_enum", line=ln)]
|
|
446
|
+
track = CC.classify_track({"layer": meta.get("layer"),
|
|
447
|
+
"tags": meta.get("tags") or []}, content or "")
|
|
448
|
+
if not CC.basis_licensed(track, basis):
|
|
449
|
+
allow = "/".join(CC.allowed_basis(track)) or "—"
|
|
450
|
+
return [_hit(node_id, "weak_source", "verification_basis", None,
|
|
451
|
+
"赛道 %s(政策 %s)不允许基底 %s;允许 %s"
|
|
452
|
+
% (track, CC.source_policy(track) or "—", basis, allow),
|
|
453
|
+
rule="basis_licensed", line=ln)]
|
|
454
|
+
return []
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
# 生效条件:ctx.get("fm") 或 {} 中 code_ref 或 doc_ref 为非空 dict 且 RI.probe_ref(ref) 返回的 status 属于 _REF_DEAD_STATUSES(dangling 或 stale)时,返回对应 stale 命中(dangling 归因载体消失、stale 归因区间哈希不符);否则返回 []。
|
|
458
|
+
def _loc_stale(node_id, meta, content, ctx):
|
|
459
|
+
"""**依赖存在性**——声明的载体(源文件)已不存在 / 已漂移(真正的适用边界越出)。
|
|
460
|
+
|
|
461
|
+
知识的真值挂在它描述的对象上:代码知识挂在源文件的符号上,源没了或改了,
|
|
462
|
+
知识才真的不再成立。判据与 `refindex` **同源**(`probe_ref`,不另立一份):
|
|
463
|
+
|
|
464
|
+
============== ====================================== ==========
|
|
465
|
+
probe_ref 态 含义 本判据
|
|
466
|
+
============== ====================================== ==========
|
|
467
|
+
``ok`` 源文件在、区间哈希匹配 不判
|
|
468
|
+
``stale`` 源文件在、区间哈希不符(源被改,漂移) **stale**
|
|
469
|
+
``dangling`` 源文件不存在(载体消失) **stale**
|
|
470
|
+
``unresolved`` 拿不到 root(判不了) 不判(不猜)
|
|
471
|
+
``error`` 读盘失败 不判(不猜)
|
|
472
|
+
============== ====================================== ==========
|
|
473
|
+
|
|
474
|
+
后两态是**观测手段不足**,不是「依赖已失效」——按「不猜」纪律如实不报。
|
|
475
|
+
|
|
476
|
+
**时间窗不在此处判**(2026-09-16 修正的原缺陷):`condition_space.time_window`
|
|
477
|
+
是写入端**观测时刻**栏(`mdcg.add` 未给时自动填「写入时刻 +1h」,见其
|
|
478
|
+
`OBSERVATION_WINDOW_SEC`),不是知识的有效期——拿它判「适用边界越出」,
|
|
479
|
+
等于把「这条记忆写入超过 1 小时」冒充为「失效」。真实适用前提由知识自己
|
|
480
|
+
声明在 `code_ref`/`doc_ref` 上,故改由依赖存在性裁决;观测时刻本身不丢,
|
|
481
|
+
降级为 `observation_aged`(观测面,见 `ADVISORY_KINDS`)。
|
|
482
|
+
"""
|
|
483
|
+
fm = ctx.get("fm") or {}
|
|
484
|
+
hits = []
|
|
485
|
+
for key in ("code_ref", "doc_ref"):
|
|
486
|
+
ref = fm.get(key)
|
|
487
|
+
if not isinstance(ref, dict) or not ref:
|
|
488
|
+
continue
|
|
489
|
+
# root 基准必须取自 ref 自身(`_code_ref`/`_doc_ref` 落的是**源大域** root,
|
|
490
|
+
# path 相对它);`ctx["root"]` 是**认知图** root——拿它去拼会把每一条依赖
|
|
491
|
+
# 都误判成悬空。ref 无 root(旧节点)时 probe 返回 `unresolved` → 不判。
|
|
492
|
+
probe = RI.probe_ref(ref)
|
|
493
|
+
st = str(probe.get("status") or "")
|
|
494
|
+
if st not in _REF_DEAD_STATUSES:
|
|
495
|
+
continue
|
|
496
|
+
rel = str(ref.get("path") or "?")
|
|
497
|
+
if st == "dangling":
|
|
498
|
+
ev = ("依赖的源文件已不存在(%s:%s)——知识所指的载体消失,适用边界越出"
|
|
499
|
+
% (key, rel))
|
|
500
|
+
else:
|
|
501
|
+
ev = ("依赖的源文件已漂移(%s:%s):索引区间哈希 %s ≠ 现算 %s——"
|
|
502
|
+
"源被改动,知识所指的符号可能已不是原物"
|
|
503
|
+
% (key, rel, probe.get("hash_expected"), probe.get("hash")))
|
|
504
|
+
hits.append(_hit(node_id, "stale", key, None, ev, rule="ref_" + st))
|
|
505
|
+
return hits
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
# 生效条件:从 meta.get("condition_space") 或回落 ctx.get("fm") 的 condition_space(须为 dict)取 time_window,缺则取 meta.get("time_window"),若 NF.is_full_time_window(tw) 或 float(tw[1]) 抛 TypeError/ValueError/IndexError/KeyError 或 ctx.get("now") 为 None 或 hi>=float(now) 则返回 [];否则返回一条 severity="info" 的 observation_aged 命中。
|
|
509
|
+
def _loc_observation_aged(node_id, meta, content, ctx):
|
|
510
|
+
"""时间窗已过——**这是观测时刻,不是失效声明**(观测面,不进告警面)。
|
|
511
|
+
|
|
512
|
+
时间窗来源链与 `_loc_missing_field` 的槽检查同构:`meta.condition_space`
|
|
513
|
+
(显式注入面)→ `fm.condition_space`(文件真源)→ `meta.time_window`
|
|
514
|
+
(索引快照字段)。**索引快照只透传 `time_window`**(`mdcg.py` `_stage` 口径:
|
|
515
|
+
时空字段入快照、condition_space 整块不入),故第三跳必留。
|
|
516
|
+
|
|
517
|
+
**不进默认全量**(不在 `ISSUE_KINDS`):写入端未给 time_window 时以写入时刻
|
|
518
|
+
自动填 1 小时窗,故真库「已过期」绝大多数是「写入超过 1 小时」,与知识是否
|
|
519
|
+
失效无关。观测时刻本身仍如实报(审计有用),只是**不冒充问题**——命中带
|
|
520
|
+
`severity="info"` 标记它是咨询级。
|
|
521
|
+
"""
|
|
522
|
+
fm = ctx.get("fm") or {}
|
|
523
|
+
cs = meta.get("condition_space")
|
|
524
|
+
if not isinstance(cs, dict):
|
|
525
|
+
csf = fm.get("condition_space")
|
|
526
|
+
cs = csf if isinstance(csf, dict) else None
|
|
527
|
+
tw = (cs or {}).get("time_window")
|
|
528
|
+
if tw is None:
|
|
529
|
+
tw = meta.get("time_window")
|
|
530
|
+
if NF.is_full_time_window(tw):
|
|
531
|
+
return [] # 全时窗是**合法声明**(任意时刻成立),不判
|
|
532
|
+
try:
|
|
533
|
+
hi = float(tw[1])
|
|
534
|
+
except (TypeError, ValueError, IndexError, KeyError):
|
|
535
|
+
return []
|
|
536
|
+
now = ctx.get("now")
|
|
537
|
+
if now is None or hi >= float(now):
|
|
538
|
+
return []
|
|
539
|
+
span = mark_spans(content or "").get("生效条件")
|
|
540
|
+
return [_hit(node_id, "observation_aged", "condition_space", span,
|
|
541
|
+
"观测时间窗 %s 已过(hi=%.0f < now=%.0f)——这是**观测时刻**,"
|
|
542
|
+
"不是失效声明(知识不因观测过去而失效);时效判定见 stale"
|
|
543
|
+
% (NF.time_window_text(tw) or "?", hi, float(now)),
|
|
544
|
+
rule="observation_window_passed", severity="info",
|
|
545
|
+
line=_line_of(ctx.get("text"), content, span),
|
|
546
|
+
snippet=_snippet(content, span))]
|
|
547
|
+
|
|
548
|
+
|
|
549
|
+
# 生效条件:h 取 str(meta.get("content_hash") or "").strip(),若 _blank(h) 则 h=NF.content_hash(content 或 "");对 ctx.get("peers") 或 [] 中 node_id 不等于本节点且 ph=str(p.get("hash") or "") 或回落 NF.content_hash(p.get("content") or "") 后非空且等于 h 的 peer 计入 matched;matched 非空时返回一条 dup 命中(span=[0,len(body)]),否则返回 []。
|
|
550
|
+
def _loc_dup(node_id, meta, content, ctx):
|
|
551
|
+
"""同组同内容指纹(与 M1 `dup_hash_group` 同判据;D1 名 `dup`)。"""
|
|
552
|
+
body = content if content is not None else ""
|
|
553
|
+
h = str(meta.get("content_hash") or "").strip()
|
|
554
|
+
if _blank(h):
|
|
555
|
+
h = NF.content_hash(body) # 索引无指纹 → 用正文实算(诚实标注来源)
|
|
556
|
+
matched = []
|
|
557
|
+
for p in ctx.get("peers") or []:
|
|
558
|
+
if p.get("node_id") == node_id:
|
|
559
|
+
continue
|
|
560
|
+
ph = str(p.get("hash") or "") or NF.content_hash(p.get("content") or "")
|
|
561
|
+
if ph and ph == h:
|
|
562
|
+
matched.append(p)
|
|
563
|
+
if not matched:
|
|
564
|
+
return []
|
|
565
|
+
span = [0, len(body)]
|
|
566
|
+
return [_hit(node_id, "dup", "content_hash", span,
|
|
567
|
+
"与 %s 同内容指纹 %s(同类共 %d 条),整篇正文重复"
|
|
568
|
+
% (matched[0].get("node_id"), h, len(matched) + 1),
|
|
569
|
+
rule="dup_hash_group", peer=matched[0].get("node_id"),
|
|
570
|
+
snippet=_snippet(body, span))]
|
|
571
|
+
|
|
572
|
+
|
|
573
|
+
# 生效条件:对 ctx.get("peers") 或 [] 中每个非自身 peer,按 sentence_spans 对齐 content 与 peer content 的句子,当同一句位上去数字骨架相同(WL._skeleton(seg) 非空、长度 ≥ WL.MIN_SKELETON 且等于 peer 骨架)且 seg.strip() != ptext.strip() 的句子数达到 MIN_FLOW_SENTENCES 时,为这些句各返回一条 template_flow 命中;否则返回 []。
|
|
574
|
+
def _loc_template_flow(node_id, meta, content, ctx):
|
|
575
|
+
"""同模板流水:与同组节点逐句「去数字骨架相同、字面不同」→ 指向那些句。
|
|
576
|
+
|
|
577
|
+
与 M1 `template_flow_digits_only` 同判据(都走 `writelimit._skeleton`),
|
|
578
|
+
差别只在产物:M1 给一条条目级 issue,D1 给出**具体是哪几句**(span + 句索引)。
|
|
579
|
+
单句偶合不算流水 → 需同组至少 `MIN_FLOW_SENTENCES` 句同时成立。
|
|
580
|
+
"""
|
|
581
|
+
tgt = sentence_spans(content or "")
|
|
582
|
+
hits = []
|
|
583
|
+
for p in ctx.get("peers") or []:
|
|
584
|
+
if p.get("node_id") == node_id:
|
|
585
|
+
continue
|
|
586
|
+
pseg = sentence_spans(p.get("content") or "")
|
|
587
|
+
same = []
|
|
588
|
+
for (i, s, e, seg) in tgt:
|
|
589
|
+
if i >= len(pseg):
|
|
590
|
+
break
|
|
591
|
+
ptext = pseg[i][3]
|
|
592
|
+
sk = WL._skeleton(seg)
|
|
593
|
+
if not sk or len(sk) < WL.MIN_SKELETON or sk != WL._skeleton(ptext):
|
|
594
|
+
continue
|
|
595
|
+
if seg.strip() == ptext.strip():
|
|
596
|
+
continue # 字面全同属整篇重复(dup),不是「只换数字」的流水
|
|
597
|
+
same.append((i, s, e, seg, sk))
|
|
598
|
+
if len(same) < MIN_FLOW_SENTENCES:
|
|
599
|
+
continue
|
|
600
|
+
for (i, s, e, seg, sk) in same:
|
|
601
|
+
hits.append(_hit(node_id, "template_flow", "content", [s, e],
|
|
602
|
+
"同模板流水(与 %s 第 %d 句去数字骨架相同、仅字面数值不同):%s"
|
|
603
|
+
% (p.get("node_id"), i, sk[:40]),
|
|
604
|
+
rule="template_flow_digits_only", sentence=i,
|
|
605
|
+
peer=p.get("node_id"),
|
|
606
|
+
line=_line_of(ctx.get("text"), content, [s, e]),
|
|
607
|
+
snippet=seg.strip()[:SNIPPET_MAX]))
|
|
608
|
+
return hits
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
# 生效条件:meta.get("content_hash") 非 _blank 且 content 非 None 且 h != NF.content_hash(content) 时,返回一条 hash_declared_vs_actual 的 contradiction 命中(cause 由 hash_mismatch_cause 依 meta/ctx.get("root")/ctx.get("path") 判);ctx.get("fm") 或 {} 的 id 非 _blank 且 str(fm_id) != str(node_id) 时额外返回一条 id_declared_vs_index 命中;两者皆不成立返回 []。
|
|
612
|
+
def _loc_contradiction(node_id, meta, content, ctx):
|
|
613
|
+
"""**确定性**矛盾:声明与事实不符(指纹 / 标识)。
|
|
614
|
+
|
|
615
|
+
语义级矛盾(正文自相矛盾)不在此处——那是 LLM 的活,D1 不猜(见 `SEMANTIC_ONLY`)。
|
|
616
|
+
指纹不一致**必须带成因**(`cause`):`index_lag`(快照滞后,属正常写路径现象,
|
|
617
|
+
复核者无需修数据)与 `true_mismatch`(真不一致,需查)处置完全不同。
|
|
618
|
+
"""
|
|
619
|
+
hits = []
|
|
620
|
+
lines = key_line_spans(ctx.get("text") or "")
|
|
621
|
+
h = str(meta.get("content_hash") or "").strip()
|
|
622
|
+
if not _blank(h) and content is not None:
|
|
623
|
+
actual = NF.content_hash(content)
|
|
624
|
+
if h != actual:
|
|
625
|
+
cause, note = hash_mismatch_cause(meta, root=ctx.get("root"),
|
|
626
|
+
path=ctx.get("path"))
|
|
627
|
+
hits.append(_hit(node_id, "contradiction", "content_hash", None,
|
|
628
|
+
"索引声明指纹 %s ≠ 正文实算 %s(%s)"
|
|
629
|
+
% (h, actual, note), rule="hash_declared_vs_actual",
|
|
630
|
+
cause=cause,
|
|
631
|
+
line=(lines.get("content_hash") or (None, None))[0]))
|
|
632
|
+
fm_id = (ctx.get("fm") or {}).get("id")
|
|
633
|
+
if not _blank(fm_id) and str(fm_id) != str(node_id):
|
|
634
|
+
hits.append(_hit(node_id, "contradiction", "id", None,
|
|
635
|
+
"文件 frontmatter id=%s ≠ 索引键 %s" % (fm_id, node_id),
|
|
636
|
+
rule="id_declared_vs_index",
|
|
637
|
+
line=(lines.get("id") or (None, None))[0]))
|
|
638
|
+
return hits
|
|
639
|
+
|
|
640
|
+
|
|
641
|
+
LOCATORS = {
|
|
642
|
+
"missing_field": _loc_missing_field,
|
|
643
|
+
"weak_source": _loc_weak_source,
|
|
644
|
+
"stale": _loc_stale,
|
|
645
|
+
"observation_aged": _loc_observation_aged,
|
|
646
|
+
"dup": _loc_dup,
|
|
647
|
+
"template_flow": _loc_template_flow,
|
|
648
|
+
"contradiction": _loc_contradiction,
|
|
649
|
+
}
|
|
650
|
+
|
|
651
|
+
#: D1 **不定位**的判定:属语义层,交回 LLM 意见(不猜 = 不编造字符区间)
|
|
652
|
+
SEMANTIC_ONLY = {
|
|
653
|
+
"contradiction_semantic": "正文自相矛盾属语义判定;D1 只做确定性矛盾"
|
|
654
|
+
"(索引指纹 / 标识与事实不符)",
|
|
655
|
+
}
|
|
656
|
+
|
|
657
|
+
|
|
658
|
+
# 生效条件:始终返回 SEMANTIC_ONLY 中以 str(kind or "").strip() 为键查得的值,缺键时返回空串(空串表示可定位)。
|
|
659
|
+
def blindspot_reason(kind) -> str:
|
|
660
|
+
"""D1 拒绝定位的类别 → 理由(空串表示可定位)。"""
|
|
661
|
+
return SEMANTIC_ONLY.get(str(kind or "").strip(), "")
|
|
662
|
+
|
|
663
|
+
|
|
664
|
+
# ---------------------------- 主入口 ----------------------------
|
|
665
|
+
|
|
666
|
+
# 生效条件:items 中每项按 content=it.get("content") or ""、hash=str(it.get("hash") or "").strip() 或回落 NF.content_hash(content)、sk=WL.template_signature(content) or "" 预处理后,返回 {node_id: [同 hash 或同非空 sk 的其他 peer]};items 为空/None 返回空 dict。
|
|
667
|
+
def build_peers(items) -> dict:
|
|
668
|
+
"""`[{node_id, content, hash?, …}]` → `{node_id: [peer, …]}`。
|
|
669
|
+
|
|
670
|
+
只有**同内容指纹**或**同模板骨架**才互为对照——不是「同一批就是一组」。
|
|
671
|
+
组的作用域由调用方给定(批 / 包),与 M1 机械层「组 = 同一个包」的口径一致。
|
|
672
|
+
"""
|
|
673
|
+
prep = []
|
|
674
|
+
for it in items or []:
|
|
675
|
+
c = it.get("content") or ""
|
|
676
|
+
prep.append({"node_id": it.get("node_id"), "content": c, "meta": it.get("meta"),
|
|
677
|
+
"text": it.get("text"), "fm": it.get("fm"),
|
|
678
|
+
"hash": str(it.get("hash") or "").strip() or NF.content_hash(c),
|
|
679
|
+
"sk": WL.template_signature(c) or ""})
|
|
680
|
+
by_hash, by_sk = {}, {}
|
|
681
|
+
for p in prep:
|
|
682
|
+
by_hash.setdefault(p["hash"], []).append(p)
|
|
683
|
+
if p["sk"]:
|
|
684
|
+
by_sk.setdefault(p["sk"], []).append(p)
|
|
685
|
+
out = {}
|
|
686
|
+
for p in prep:
|
|
687
|
+
grp = {q["node_id"]: q for q in by_hash.get(p["hash"], [])
|
|
688
|
+
if q["node_id"] != p["node_id"]}
|
|
689
|
+
for q in (by_sk.get(p["sk"], []) if p["sk"] else []):
|
|
690
|
+
if q["node_id"] != p["node_id"]:
|
|
691
|
+
grp.setdefault(q["node_id"], q)
|
|
692
|
+
out[p["node_id"]] = list(grp.values())
|
|
693
|
+
return out
|
|
694
|
+
|
|
695
|
+
|
|
696
|
+
# 生效条件:对 issue_hint(None/str/dict 或其 list/tuple)逐 hint 分类,返回 (kinds, fields, blind):能 canonical_kind 到 ISSUE_KINDS 或 ADVISORY_KINDS 的入 kinds,blindspot_reason 非空的入 blind,其余非空 kind 与 field 入 fields;issue_hint 为 None 时 kinds/fields 为空集、blind 为空列表。
|
|
697
|
+
def _norm_hint(issue_hint):
|
|
698
|
+
"""issue_hint → `(kinds, fields, blindspot)`。
|
|
699
|
+
|
|
700
|
+
形态:`None`(全量)|str(issue_kind 或 field)|dict(issue_kind/field)
|
|
701
|
+
|上述的 list。
|
|
702
|
+
"""
|
|
703
|
+
kinds, fields, blind = set(), set(), []
|
|
704
|
+
hints = issue_hint if isinstance(issue_hint, (list, tuple)) else [issue_hint]
|
|
705
|
+
for h in hints:
|
|
706
|
+
if h is None:
|
|
707
|
+
continue
|
|
708
|
+
if isinstance(h, dict):
|
|
709
|
+
s = h.get("issue_kind") or h.get("kind") or ""
|
|
710
|
+
f = h.get("field")
|
|
711
|
+
else:
|
|
712
|
+
s, f = str(h).strip(), None
|
|
713
|
+
if s:
|
|
714
|
+
why = blindspot_reason(s)
|
|
715
|
+
if why:
|
|
716
|
+
blind.append("%s:%s" % (s, why))
|
|
717
|
+
else:
|
|
718
|
+
ck = canonical_kind(s)
|
|
719
|
+
if ck in ISSUE_KINDS or ck in ADVISORY_KINDS:
|
|
720
|
+
kinds.add(ck)
|
|
721
|
+
else:
|
|
722
|
+
fields.add(s) # 不是类别 → 当作字段过滤
|
|
723
|
+
if f:
|
|
724
|
+
fields.add(str(f).strip())
|
|
725
|
+
return kinds, fields, blind
|
|
726
|
+
|
|
727
|
+
|
|
728
|
+
# 生效条件:当 meta 与 content 均非 None 时不读盘;否则 root 为 None 抛 ValueError,经 load_node 读不到节点返回 {"hits": [], "blindspot": blind+["节点 %s 不在索引"%node_id], "load": None};读到时补 meta/content/text/fm/path,按 kinds 非空时只跑 sorted(kinds)、否则 fields 非空或 issue_hint 为 None 时跑 sorted(ISSUE_KINDS)、否则 run=[] 执行 LOCATORS,对命中按 fields 过滤、补 line 并排序后返回 {"hits": hits, "blindspot": blind, "load": {...}}。
|
|
729
|
+
def locate_ex(node_id, issue_hint=None, *, meta=None, content=None, fm=None, text=None,
|
|
730
|
+
root=None, index=None, peers=None, now=None, path=None,
|
|
731
|
+
field_layers=None) -> dict:
|
|
732
|
+
"""D1 定位(完整返回):`{"hits": [...], "blindspot": [...], "load": {...}}`。
|
|
733
|
+
|
|
734
|
+
`meta`/`content` 显式给出则**不读盘**(纯函数用法——测试与批量复用都走这条);
|
|
735
|
+
否则 `root` 必填,经 `load_node` 从索引+文件取。返回的 hits 已按
|
|
736
|
+
`(issue_kind, field, span起点, peer)` 排序,**确定性可复算**。
|
|
737
|
+
|
|
738
|
+
`path`(节点文件绝对路径——供 `contradiction` 判成因)、`field_layers`
|
|
739
|
+
(字段层门限,缺省 `field_layer_scope()`)可显式注入,便于纯函数测试。
|
|
740
|
+
"""
|
|
741
|
+
kinds, fields, blind = _norm_hint(issue_hint)
|
|
742
|
+
if meta is None or content is None:
|
|
743
|
+
if root is None:
|
|
744
|
+
raise ValueError("locate 需要 meta+content,或给出 root 以便读节点")
|
|
745
|
+
nd = load_node(node_id, root, index=index)
|
|
746
|
+
if nd is None:
|
|
747
|
+
return {"hits": [], "blindspot": blind + ["节点 %s 不在索引" % node_id],
|
|
748
|
+
"load": None}
|
|
749
|
+
if meta is None:
|
|
750
|
+
meta = nd["meta"]
|
|
751
|
+
if content is None:
|
|
752
|
+
content = nd["content"]
|
|
753
|
+
if text is None:
|
|
754
|
+
text = nd["text"]
|
|
755
|
+
if fm is None:
|
|
756
|
+
fm = nd["fm"]
|
|
757
|
+
if path is None:
|
|
758
|
+
path = nd.get("path")
|
|
759
|
+
meta = meta or {}
|
|
760
|
+
ctx = {"fm": fm or {}, "text": text, "root": root, "index": index,
|
|
761
|
+
"peers": peers or [], "now": time.time() if now is None else float(now),
|
|
762
|
+
"path": path, "field_layers": field_layers}
|
|
763
|
+
hits = []
|
|
764
|
+
# 定位范围:显式类别 → 只跑该类;给出字段 → 全量再由字段收窄;
|
|
765
|
+
# 无提示 → 全量。**提示全部落 blindspot 时范围为空**——不借机报别的类
|
|
766
|
+
# (否则「给了语义提示」会顺手退回全量,与「不猜」纪律相悖)。
|
|
767
|
+
if kinds:
|
|
768
|
+
run = sorted(kinds)
|
|
769
|
+
elif fields or issue_hint is None:
|
|
770
|
+
run = sorted(ISSUE_KINDS)
|
|
771
|
+
else:
|
|
772
|
+
run = []
|
|
773
|
+
for k in run:
|
|
774
|
+
fn = LOCATORS.get(k)
|
|
775
|
+
if fn is None:
|
|
776
|
+
continue
|
|
777
|
+
for h in fn(node_id, meta, content, ctx):
|
|
778
|
+
if fields and h.get("field") not in fields:
|
|
779
|
+
continue
|
|
780
|
+
if h.get("line") is None and h.get("span") is not None:
|
|
781
|
+
h["line"] = _line_of(text, content, h["span"])
|
|
782
|
+
hits.append(h)
|
|
783
|
+
hits.sort(key=lambda h: (h["issue_kind"], str(h.get("field") or ""),
|
|
784
|
+
-1 if h.get("span") is None else h["span"][0],
|
|
785
|
+
str(h.get("peer") or "")))
|
|
786
|
+
return {"hits": hits, "blindspot": blind,
|
|
787
|
+
"load": {"node_id": node_id, "meta": meta,
|
|
788
|
+
"content_len": len(content or "")}}
|
|
789
|
+
|
|
790
|
+
|
|
791
|
+
# 生效条件:给定 node_id(必填)与可选 issue_hint 及 kw 后,直接返回 locate_ex(node_id, issue_hint, **kw) 结果的 "hits" 列表。
|
|
792
|
+
def locate(node_id, issue_hint=None, **kw) -> list:
|
|
793
|
+
"""**D1 契约入口**:`locate(node_id, issue_hint) -> [{field, span, issue_kind, evidence}]`。"""
|
|
794
|
+
return locate_ex(node_id, issue_hint, **kw)["hits"]
|
|
795
|
+
|
|
796
|
+
|
|
797
|
+
# 生效条件:给定 hits 与 key 后,返回 hits 中每个 h.get(key) 字符串化取值到出现次数的字典(键升序);hits 为 None/空时返回空 dict。
|
|
798
|
+
def _counts(hits, key) -> dict:
|
|
799
|
+
out = {}
|
|
800
|
+
for h in hits or []:
|
|
801
|
+
k = h.get(key)
|
|
802
|
+
out[str(k)] = out.get(str(k), 0) + 1
|
|
803
|
+
return dict(sorted(out.items()))
|
|
804
|
+
|
|
805
|
+
|
|
806
|
+
# 生效条件:items 显式给出时不读盘;items 为 None 时 root 为 None 抛 ValueError,否则按 node_ids 逐节点 load_node 组装 items(读不到的记入 missing);随后对每个 item 调 locate_ex 汇总 hits 或 clean,并返回含 nodes/hits/by_kind/by_field/by_rule/clean/missing 的 dict。
|
|
807
|
+
def locate_many(node_ids=None, *, root=None, index=None, items=None,
|
|
808
|
+
issue_hint=None, now=None) -> dict:
|
|
809
|
+
"""批量定位:**同一批内**互为对照(`dup` / `template_flow` 的组 = 本批)。
|
|
810
|
+
|
|
811
|
+
`items` 显式给出 `[{node_id, content, meta?, text?, fm?, hash?}]` 则不再读盘。
|
|
812
|
+
返回 `{nodes, hits, by_kind, by_field, clean, missing}`;`clean` 是**无命中**的节点、
|
|
813
|
+
`missing` 是**读不到**的节点——两者分开报(「没问题」≠「没看」)。
|
|
814
|
+
"""
|
|
815
|
+
missing = []
|
|
816
|
+
if items is None:
|
|
817
|
+
if root is None:
|
|
818
|
+
raise ValueError("locate_many 需要 root(读盘)或 items(显式给出)")
|
|
819
|
+
items = []
|
|
820
|
+
for nid in node_ids or []:
|
|
821
|
+
nd = load_node(nid, root, index=index)
|
|
822
|
+
if nd is None:
|
|
823
|
+
missing.append(nid)
|
|
824
|
+
continue
|
|
825
|
+
items.append({"node_id": nid, "content": nd["content"] or "",
|
|
826
|
+
"text": nd["text"], "fm": nd["fm"], "meta": nd["meta"],
|
|
827
|
+
"path": nd.get("path"),
|
|
828
|
+
"hash": nd["meta"].get("content_hash")})
|
|
829
|
+
peers = build_peers(items)
|
|
830
|
+
now = time.time() if now is None else float(now)
|
|
831
|
+
hits, clean = [], []
|
|
832
|
+
for it in items:
|
|
833
|
+
nid = it["node_id"]
|
|
834
|
+
ex = locate_ex(nid, issue_hint, meta=it.get("meta") or {},
|
|
835
|
+
content=it.get("content") or "", fm=it.get("fm"),
|
|
836
|
+
text=it.get("text"), peers=peers.get(nid) or [], now=now,
|
|
837
|
+
root=root, path=it.get("path"))
|
|
838
|
+
if ex["hits"]:
|
|
839
|
+
hits.extend(ex["hits"])
|
|
840
|
+
else:
|
|
841
|
+
clean.append(nid)
|
|
842
|
+
hits.sort(key=lambda h: (str(h.get("node_id")), h["issue_kind"],
|
|
843
|
+
str(h.get("field") or ""),
|
|
844
|
+
-1 if h.get("span") is None else h["span"][0]))
|
|
845
|
+
return {"nodes": len(items), "hits": hits, "by_kind": _counts(hits, "issue_kind"),
|
|
846
|
+
"by_field": _counts(hits, "field"), "by_rule": _counts(hits, "rule"),
|
|
847
|
+
"clean": clean, "missing": missing}
|
|
848
|
+
|
|
849
|
+
|
|
850
|
+
# 生效条件:从 pkg.get("entries") or [] 取条目并跳过 e.get("node_id") 为空者,root 非 None 时逐个 load_node 取正文,取不到时回落 e.get("excerpt") or "",组装 items 调 locate_many 后返回其结果并附加 bundle_id 与 entries 数量;pkg 为 None 时 entries 为空列表。
|
|
851
|
+
def locate_package(pkg, *, root=None, index=None, issue_hint=None, now=None) -> dict:
|
|
852
|
+
"""M1 包 → 定位汇总(组的作用域 = 本包,与 M1 机械层同口径)。
|
|
853
|
+
|
|
854
|
+
正文优先读盘(准确);文件不可读时退回包内 `excerpt`(可能截断——宁可降级也不
|
|
855
|
+
假装读到了全文,`missing` 会如实记下读不到的节点)。
|
|
856
|
+
"""
|
|
857
|
+
entries = (pkg or {}).get("entries") or []
|
|
858
|
+
items = []
|
|
859
|
+
for e in entries:
|
|
860
|
+
nid = e.get("node_id")
|
|
861
|
+
if not nid:
|
|
862
|
+
continue
|
|
863
|
+
nd = load_node(nid, root, index=index) if root else None
|
|
864
|
+
body = (nd or {}).get("content")
|
|
865
|
+
items.append({"node_id": nid,
|
|
866
|
+
"content": body if body is not None else (e.get("excerpt") or ""),
|
|
867
|
+
"text": (nd or {}).get("text"), "fm": (nd or {}).get("fm"),
|
|
868
|
+
"meta": (nd or {}).get("meta") or e,
|
|
869
|
+
"path": (nd or {}).get("path") or e.get("path"),
|
|
870
|
+
"hash": ((nd or {}).get("meta") or {}).get("content_hash")
|
|
871
|
+
or e.get("content_hash")})
|
|
872
|
+
out = locate_many(items=items, issue_hint=issue_hint, now=now)
|
|
873
|
+
out["bundle_id"] = (pkg or {}).get("bundle_id")
|
|
874
|
+
out["entries"] = len(entries)
|
|
875
|
+
return out
|
|
876
|
+
|
|
877
|
+
|
|
878
|
+
# 生效条件:给定 hits 后返回 total=len(hits or [])、去重 node_id 数、排序后的 node_ids、以及 by_kind/by_field/by_rule 计数;hits 为 None/空时 total=0、nodes=0、node_ids=[]。
|
|
879
|
+
def summary(hits) -> dict:
|
|
880
|
+
"""命中汇总(审计留痕用)。"""
|
|
881
|
+
nodes = sorted({str(h.get("node_id")) for h in hits or []})
|
|
882
|
+
return {"total": len(hits or []), "nodes": len(nodes), "node_ids": nodes,
|
|
883
|
+
"by_kind": _counts(hits, "issue_kind"), "by_field": _counts(hits, "field"),
|
|
884
|
+
"by_rule": _counts(hits, "rule")}
|
|
885
|
+
|
|
886
|
+
|
|
887
|
+
# 生效条件:给定 hits 后逐条生成 Markdown 行并返回表头加各行:field/line/snippet 取 h.get(...) or "—"(假值回落 "—"),span 为 None 时显示 "—" 否则 "起-止",evidence 取 (h.get("evidence") or "").replace("|","\\|")(假值回落空串)。
|
|
888
|
+
def markdown_table(hits) -> str:
|
|
889
|
+
"""人工核对清单:每行一条命中,末列留空供核对者填判定(D1 验收抽样用)。"""
|
|
890
|
+
head = ("| # | node_id | issue_kind | field | span | line | 片段 | evidence | 人工判定 |\n"
|
|
891
|
+
"|---|---|---|---|---|---|---|---|---|\n")
|
|
892
|
+
rows = []
|
|
893
|
+
for i, h in enumerate(hits or [], 1):
|
|
894
|
+
sp = "—" if h.get("span") is None else "%d-%d" % (h["span"][0], h["span"][1])
|
|
895
|
+
rows.append("| %d | %s | %s | %s | %s | %s | %s | %s | |"
|
|
896
|
+
% (i, h.get("node_id"), h.get("issue_kind"), h.get("field") or "—",
|
|
897
|
+
sp, h.get("line") or "—",
|
|
898
|
+
(h.get("snippet") or "—").replace("|", "\\|"),
|
|
899
|
+
(h.get("evidence") or "").replace("|", "\\|")))
|
|
900
|
+
return head + "\n".join(rows)
|
|
901
|
+
|
|
902
|
+
|
|
903
|
+
# 生效条件:解析 argv(缺省 sys.argv)后,--root 为假值(含默认 os.environ.get("MDCG_ROOT") 为 None 或空串)时打印提示并返回 2;--node 追加列表为空时打印提示并返回 2;否则以 --root 与 --node 调 locate_many,并按 --json 或 --markdown 输出后返回 0,两者皆无则逐行打印命中与汇总后返回 0。
|
|
904
|
+
def main(argv=None) -> int:
|
|
905
|
+
ap = argparse.ArgumentParser(prog="python -m md_cg.mreview.locate",
|
|
906
|
+
description="记忆评审 M3 · D1 字段级定位")
|
|
907
|
+
ap.add_argument("--root", default=os.environ.get("MDCG_ROOT"),
|
|
908
|
+
help="真源根(缺省读环境变量 MDCG_ROOT)")
|
|
909
|
+
ap.add_argument("--node", action="append", default=[],
|
|
910
|
+
help="节点 id(可多次;同一批内互为对照)")
|
|
911
|
+
ap.add_argument("--kind", action="append", default=[], help="只定位该类问题")
|
|
912
|
+
ap.add_argument("--field", action="append", default=[], help="只保留该字段的命中")
|
|
913
|
+
ap.add_argument("--json", action="store_true", help="输出 JSON(缺省人读表)")
|
|
914
|
+
ap.add_argument("--markdown", action="store_true", help="输出人工核对清单")
|
|
915
|
+
a = ap.parse_args(argv)
|
|
916
|
+
if not a.root:
|
|
917
|
+
print("需要 --root 或环境变量 MDCG_ROOT", file=sys.stderr)
|
|
918
|
+
return 2
|
|
919
|
+
if not a.node:
|
|
920
|
+
print("需要 --node <id>(可多次)", file=sys.stderr)
|
|
921
|
+
return 2
|
|
922
|
+
rep = locate_many(a.node, root=a.root,
|
|
923
|
+
issue_hint=(list(a.kind) + list(a.field)) or None)
|
|
924
|
+
if a.json:
|
|
925
|
+
print(json.dumps(rep, ensure_ascii=False, indent=2))
|
|
926
|
+
return 0
|
|
927
|
+
if a.markdown:
|
|
928
|
+
print(markdown_table(rep["hits"]))
|
|
929
|
+
return 0
|
|
930
|
+
for h in rep["hits"]:
|
|
931
|
+
print("%-14s %-20s %-14s %s" % (h["node_id"], h["field"], h["issue_kind"],
|
|
932
|
+
h["evidence"]))
|
|
933
|
+
print("-- 命中 %d 条|干净 %d 条|缺项 %d 条"
|
|
934
|
+
% (len(rep["hits"]), len(rep["clean"]), len(rep["missing"])))
|
|
935
|
+
return 0
|
|
936
|
+
|
|
937
|
+
|
|
938
|
+
if __name__ == "__main__":
|
|
939
|
+
sys.exit(main())
|