@furongjun1999/dsh-memory 0.4.11 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +552 -465
- package/codebuddy/CODEBUDDY.md +11 -3
- package/codebuddy/README.md +92 -90
- package/codebuddy/mcp.json +27 -27
- package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
- package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
- package/docs/README.md +143 -111
- package/docs/discipline/harnesses.yaml +244 -226
- package/docs/discipline/templates/full.md.tmpl +61 -61
- package/docs/discipline/templates/rules.mdc.tmpl +68 -0
- package/docs/discipline/templates/skill.md.tmpl +23 -23
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
- package/docs/eval/bench_lingshu_self/bench_self.py +140 -0
- package/docs/eval/bench_lingshu_self/self_bench_result.json +404 -0
- package/docs/eval/bench_lingshu_self//347/201/265/346/236/242/350/207/252/345/272/223/347/253/257/345/210/260/347/253/257/346/243/200/347/264/242/345/256/236/346/265/213_v1.0.md +38 -0
- package/docs/eval//344/270/215/345/217/257/351/235/240/346/200/247/350/220/275/345/234/260_P0_v1.0.md +185 -0
- package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
- package/docs/eval//346/225/205/351/232/234/346/263/250/345/205/245/345/256/236/346/265/213_v1.0.md +422 -0
- package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
- package/docs/eval//347/253/257/345/210/260/347/253/257LoCoMoQA/345/220/214/345/217/243/345/276/204/345/257/271/347/205/247_v1.0.md +100 -0
- package/docs/eval//347/253/257/345/210/260/347/253/257/345/271/262/346/211/260/346/261/240/350/257/204/346/265/213_/347/241/256/345/256/232/346/200/247/350/243/201/345/206/263vsLLM_judge_v1.1.md +197 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v1.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v10.md +210 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v11.md +227 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v12.md +203 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v13.md +233 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v14.md +191 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v15.md +213 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v16.md +214 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v17.md +199 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v2.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v3.md +152 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v4.md +128 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v5.md +114 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v6.md +192 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v7.md +187 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v8.md +207 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v9.md +203 -0
- package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
- package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
- package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
- package/docs/hive//345/256/211/345/205/250/345/256/241/350/256/241/345/256/236/351/224/232_v0.1.md +202 -0
- package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
- package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
- package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
- package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +519 -503
- package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
- package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
- package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
- package/docs/images/lingshu-moonlight-covenant-poster-preview.jpg +0 -0
- package/docs/images/lingshu-moonlight-covenant-poster.png +0 -0
- package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
- package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
- package/docs/mdcg/lingshu_tutorial.html +14449 -14449
- package/docs/mdcg/release_v0.4.11.md +49 -0
- package/docs/mdcg/release_v0.4.5.md +55 -55
- package/docs/mdcg/tool_table_v0.3.0.md +117 -117
- package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
- package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
- package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
- package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
- package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
- package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
- package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
- package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
- package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
- package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
- package/docs/theory//344/270/215/345/217/257/351/235/240/345/256/232/347/220/206/344/270/216/345/244/261/346/225/210/344/274/230/345/205/210/346/241/206/346/236/266_v0.3.md +365 -0
- package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
- package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
- package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
- package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
- package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
- package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
- package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
- package/docs//344/270/215/345/217/257/351/235/240/346/200/247/347/220/206/350/256/272_v0.1.md +343 -0
- package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
- package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
- package/dsh/README.md +82 -82
- package/dsh/cordis.yml.example +139 -139
- package/dsh/update-lingshu.bat +11 -11
- package/lib/bridge.d.ts +9 -0
- package/lib/bridge.js +35 -0
- package/lib/hooks.js +36 -2
- package/lib/index.js +7 -1
- package/lib/lib/roleplay_web.js +116 -29
- package/lib/lib/token_store.d.ts +7 -1
- package/lib/lib/token_store.js +12 -3
- package/md_cg/__init__.py +7 -7
- package/md_cg/audit.py +379 -368
- package/md_cg/autonomy.py +287 -287
- package/md_cg/backfill.py +1328 -1327
- package/md_cg/backfill_bigdomain.py +34 -34
- package/md_cg/backfill_bucket_zh.py +35 -0
- package/md_cg/bench6_arms.py +410 -410
- package/md_cg/bench6_common.py +230 -230
- package/md_cg/bench6_competitors.py +212 -212
- package/md_cg/bench_axis_domain.py +257 -257
- package/md_cg/bench_blind_comp.py +308 -308
- package/md_cg/bench_e2e_judge.py +532 -0
- package/md_cg/bench_e2e_locomo_qa.py +368 -0
- package/md_cg/bench_e2e_qa.py +256 -0
- package/md_cg/bench_en_atoms_public.py +230 -230
- package/md_cg/bench_governance.py +348 -348
- package/md_cg/bench_lme_zh.py +410 -410
- package/md_cg/bench_locomo.py +121 -121
- package/md_cg/bench_locomo_zh.py +450 -450
- package/md_cg/bench_locomo_zh_public.py +147 -147
- package/md_cg/bench_longmem.py +112 -112
- package/md_cg/bench_membench.py +632 -632
- package/md_cg/bench_p0.py +149 -149
- package/md_cg/bench_progressive.py +287 -287
- package/md_cg/bench_role_views.py +238 -238
- package/md_cg/bench_task_ab.py +243 -243
- package/md_cg/bench_task_ab_llm.py +408 -408
- package/md_cg/bench_unified_en.py +204 -204
- package/md_cg/bench_zh_mad.py +601 -601
- package/md_cg/blindspot_tickets.py +123 -123
- package/md_cg/branches.py +301 -285
- package/md_cg/build_postings.py +73 -73
- package/md_cg/ccgc.py +1006 -948
- package/md_cg/census.py +132 -132
- package/md_cg/chain.py +315 -300
- package/md_cg/codeindex.py +531 -531
- package/md_cg/coldverify.py +292 -292
- package/md_cg/comment_gate.py +337 -337
- package/md_cg/cond_compose.py +190 -190
- package/md_cg/cond_facts.py +154 -154
- package/md_cg/cond_template.json +106 -106
- package/md_cg/condition_anchor.py +142 -142
- package/md_cg/conformance.py +726 -726
- package/md_cg/consistency.py +717 -717
- package/md_cg/consolidate.py +1537 -1439
- package/md_cg/corpus.py +110 -110
- package/md_cg/crosscheck.py +1098 -1097
- package/md_cg/crypto.py +3 -1
- package/md_cg/d_meta.py +310 -310
- package/md_cg/datapath.py +78 -18
- package/md_cg/docindex.py +473 -473
- package/md_cg/eval_common.py +575 -575
- package/md_cg/evidence.py +4 -2
- package/md_cg/evolution.py +477 -477
- package/md_cg/export.py +222 -220
- package/md_cg/forgetting.py +581 -581
- package/md_cg/fsutil.py +377 -329
- package/md_cg/hotcache.py +48 -7
- package/md_cg/hyperedge.py +251 -251
- package/md_cg/identity.py +390 -390
- package/md_cg/insight.py +500 -500
- package/md_cg/interop.py +338 -0
- package/md_cg/judgment_manifest.py +177 -0
- package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
- package/md_cg/lexicon/build_standard_en.py +171 -171
- package/md_cg/lexicon/expand_en_zh.py +211 -211
- package/md_cg/lifecycle.py +272 -272
- package/md_cg/linkref.py +280 -280
- package/md_cg/links.py +140 -107
- package/md_cg/mcp_server.py +403 -55
- package/md_cg/md_whitebox.py +345 -345
- package/md_cg/mdcg.py +783 -233
- package/md_cg/mdcos.py +500 -70
- package/md_cg/metacognition.py +591 -591
- package/md_cg/migrate.py +119 -119
- package/md_cg/migrate_aeis.py +221 -221
- package/md_cg/migrate_roleplay.py +293 -293
- package/md_cg/migrate_wisdom_graph.py +360 -360
- package/md_cg/mreview/__init__.py +25 -25
- package/md_cg/mreview/__main__.py +110 -110
- package/md_cg/mreview/bundle.py +178 -178
- package/md_cg/mreview/candidates.py +262 -262
- package/md_cg/mreview/govern.py +694 -693
- package/md_cg/mreview/locate.py +939 -939
- package/md_cg/mreview/pipeline.py +728 -728
- package/md_cg/mreview/rules/duplication.json +21 -21
- package/md_cg/mreview/rules/field_coverage.json +54 -54
- package/md_cg/mreview/rules/source_license.json +21 -21
- package/md_cg/mreview/rules/template_flow.json +21 -21
- package/md_cg/mreview/ruleset.py +252 -252
- package/md_cg/nodefile.py +575 -575
- package/md_cg/pooling.py +484 -472
- package/md_cg/postings.py +300 -298
- package/md_cg/predict.py +1100 -1100
- package/md_cg/progressive.py +123 -123
- package/md_cg/protect.py +272 -272
- package/md_cg/protocol/md_cg_gate.proto +33 -33
- package/md_cg/protocol.py +372 -372
- package/md_cg/provenance.py +582 -582
- package/md_cg/reach.py +453 -453
- package/md_cg/readcache.py +143 -0
- package/md_cg/reconcile.py +228 -0
- package/md_cg/refindex.py +833 -833
- package/md_cg/refine.py +604 -604
- package/md_cg/review_cli.py +170 -0
- package/md_cg/roleviews.py +89 -89
- package/md_cg/routing.py +393 -365
- package/md_cg/run_tests.py +211 -0
- package/md_cg/scrub.py +13 -3
- package/md_cg/security.py +128 -18
- package/md_cg/self_state.py +1029 -1029
- package/md_cg/selfreport.py +152 -151
- package/md_cg/semantic/__init__.py +10 -10
- package/md_cg/semantic/canonical.py +122 -122
- package/md_cg/semantic/en_normalizer.py +364 -364
- package/md_cg/semantic/en_zh_map.json +28694 -0
- package/md_cg/semantic/export_en_zh_map.py +64 -0
- package/md_cg/semantic/unify.py +45 -0
- package/md_cg/semantic/zh_en_atoms.py +139 -139
- package/md_cg/signer.py +7 -4
- package/md_cg/sources.py +816 -582
- package/md_cg/statushdr.py +179 -179
- package/md_cg/stg.py +54 -37
- package/md_cg/subgraph.py +729 -729
- package/md_cg/sustain.py +35 -5
- package/md_cg/tasks.py +470 -470
- package/md_cg/test_access_hints.py +147 -0
- package/md_cg/test_action_derive.py +203 -203
- package/md_cg/test_audit_rotate.py +270 -270
- package/md_cg/test_autonomy.py +143 -143
- package/md_cg/test_bench_governance.py +102 -102
- package/md_cg/test_blindspot_tickets.py +166 -166
- package/md_cg/test_branch_discard_tombstone.py +136 -0
- package/md_cg/test_branches.py +13 -3
- package/md_cg/test_ccg_perturb.py +184 -184
- package/md_cg/test_ccgc.py +433 -433
- package/md_cg/test_census_prune.py +81 -81
- package/md_cg/test_chain_read_isolate.py +168 -0
- package/md_cg/test_cond_compose_anchors.py +76 -76
- package/md_cg/test_cond_match.py +165 -165
- package/md_cg/test_condition_anchor.py +81 -81
- package/md_cg/test_d_meta.py +412 -412
- package/md_cg/test_datapath_device_name.py +203 -0
- package/md_cg/test_datapath_root.py +199 -199
- package/md_cg/test_emit_negtail_cache.py +156 -0
- package/md_cg/test_en_pipeline.py +22 -2
- package/md_cg/test_gain_gate.py +212 -212
- package/md_cg/test_govern_directread.py +421 -0
- package/md_cg/test_health_scale.py +173 -173
- package/md_cg/test_hive_ingest.py +285 -0
- package/md_cg/test_hot_cold.py +215 -215
- package/md_cg/test_hyperedge.py +245 -245
- package/md_cg/test_i26_empty_first_write.py +116 -0
- package/md_cg/test_i27_e041_identity.py +128 -0
- package/md_cg/test_i28_hotcache_prodpath.py +122 -0
- package/md_cg/test_i32_hotcache_env_key.py +218 -0
- package/md_cg/test_identity_attribution.py +96 -15
- package/md_cg/test_index_durability.py +17 -3
- package/md_cg/test_interop.py +95 -0
- package/md_cg/test_interop_judgment.py +228 -0
- package/md_cg/test_issue39_utf8_stdio.py +273 -0
- package/md_cg/test_lifecycle.py +309 -309
- package/md_cg/test_linkref.py +306 -306
- package/md_cg/test_links_concurrent_write.py +188 -0
- package/md_cg/test_lock.py +43 -43
- package/md_cg/test_md_access_parity.py +255 -255
- package/md_cg/test_md_writepath.py +345 -345
- package/md_cg/test_mdstore_search_parity.py +160 -0
- package/md_cg/test_merge_upsert.py +168 -0
- package/md_cg/test_mr_m2.py +587 -587
- package/md_cg/test_mr_m3.py +710 -710
- package/md_cg/test_mr_m4.py +485 -485
- package/md_cg/test_n123_derive_expiry_chain.py +205 -0
- package/md_cg/test_n130_verify_falsified_protect.py +185 -0
- package/md_cg/test_n131_merge_gate.py +205 -0
- package/md_cg/test_p0.py +250 -250
- package/md_cg/test_p1.py +316 -316
- package/md_cg/test_p10_identity.py +173 -173
- package/md_cg/test_p11_consistency.py +233 -233
- package/md_cg/test_p12_metacognition.py +212 -212
- package/md_cg/test_p13_encryption.py +241 -241
- package/md_cg/test_p14_sustain.py +249 -249
- package/md_cg/test_p15_scrub.py +280 -280
- package/md_cg/test_p16_self_state.py +301 -301
- package/md_cg/test_p17_predict.py +354 -354
- package/md_cg/test_p18_whitebox.py +171 -171
- package/md_cg/test_p19_migrate_roleplay.py +149 -149
- package/md_cg/test_p1x_ref_root.py +160 -0
- package/md_cg/test_p20_evolution.py +315 -315
- package/md_cg/test_p21_tokens.py +293 -270
- package/md_cg/test_p22_theory.py +175 -175
- package/md_cg/test_p23_links.py +311 -311
- package/md_cg/test_p24_evidence.py +227 -227
- package/md_cg/test_p25_weights.py +156 -156
- package/md_cg/test_p26_refindex.py +416 -416
- package/md_cg/test_p27_docindex.py +16 -7
- package/md_cg/test_p28_refcheck.py +305 -305
- package/md_cg/test_p29_session_ingest_export.py +354 -333
- package/md_cg/test_p2_mcp.py +3 -0
- package/md_cg/test_p3.py +11 -2
- package/md_cg/test_p30_maintain.py +330 -330
- package/md_cg/test_p31_insight.py +534 -534
- package/md_cg/test_p32_backfill.py +7 -1
- package/md_cg/test_p33_ccg_wiring.py +293 -293
- package/md_cg/test_p34_crosscheck.py +331 -331
- package/md_cg/test_p35_conditioned_claim.py +252 -252
- package/md_cg/test_p36_kp_align.py +230 -230
- package/md_cg/test_p37_condition_space.py +248 -248
- package/md_cg/test_p38_concurrent_flush.py +102 -0
- package/md_cg/test_p38_contextualize.py +273 -273
- package/md_cg/test_p39_verify_flow.py +153 -0
- package/md_cg/test_p39_vision_evidence.py +369 -369
- package/md_cg/test_p40_refine_worklist.py +241 -241
- package/md_cg/test_p41_evolve_patrol.py +224 -224
- package/md_cg/test_p42_provenance.py +269 -269
- package/md_cg/test_p43_pooling.py +412 -398
- package/md_cg/test_p44_md_whitebox.py +231 -231
- package/md_cg/test_p45_session_identity.py +219 -219
- package/md_cg/test_p46_unit_scope.py +272 -272
- package/md_cg/test_p47_session_view.py +316 -0
- package/md_cg/test_p4_fuzzy.py +223 -223
- package/md_cg/test_p5_semantic.py +226 -226
- package/md_cg/test_p6_consolidate.py +440 -387
- package/md_cg/test_p7_goals_recent.py +202 -202
- package/md_cg/test_p8_subgraph_chain.py +200 -200
- package/md_cg/test_p9_forget_protect.py +231 -231
- package/md_cg/test_predict_beta.py +135 -135
- package/md_cg/test_preflight_failclosed.py +100 -100
- package/md_cg/test_progressive.py +146 -146
- package/md_cg/test_propose_tail_index.py +157 -0
- package/md_cg/test_protocol.py +243 -243
- package/md_cg/test_reach.py +378 -378
- package/md_cg/test_reach_keys.py +201 -201
- package/md_cg/test_read_clip.py +141 -141
- package/md_cg/test_read_scope_b27.py +277 -0
- package/md_cg/test_readcache_default_on.py +168 -0
- package/md_cg/test_readcache_precise_inval.py +270 -0
- package/md_cg/test_readcache_prodpath.py +203 -0
- package/md_cg/test_reconcile_v0.py +294 -0
- package/md_cg/test_retr_gates_prodpath.py +140 -0
- package/md_cg/test_retr_s1.py +344 -340
- package/md_cg/test_retr_s1b.py +276 -209
- package/md_cg/test_retr_s3.py +194 -194
- package/md_cg/test_retr_s4.py +163 -163
- package/md_cg/test_retr_s5.py +200 -200
- package/md_cg/test_retr_s6.py +157 -157
- package/md_cg/test_retr_s7.py +392 -384
- package/md_cg/test_retr_s8_time.py +369 -316
- package/md_cg/test_retr_s9_edges.py +286 -286
- package/md_cg/test_retr_s9_entity_ctx.py +9 -3
- package/md_cg/test_retr_score_once.py +208 -0
- package/md_cg/test_review_conformance.py +367 -367
- package/md_cg/test_review_onepass.py +170 -0
- package/md_cg/test_role_views.py +354 -354
- package/md_cg/test_rrf_graph_seed_cache.py +154 -0
- package/md_cg/test_security_audit.py +155 -0
- package/md_cg/test_security_audit_b26.py +161 -0
- package/md_cg/test_security_audit_v21.py +250 -0
- package/md_cg/test_sem_noise.py +242 -242
- package/md_cg/test_semantic_canonical.py +16 -2
- package/md_cg/test_session_isolation.py +168 -0
- package/md_cg/test_snapshot_autoclose.py +187 -0
- package/md_cg/test_subproc_encoding.py +192 -192
- package/md_cg/test_sustain_mutual.py +153 -153
- package/md_cg/test_tail_watermark_race.py +208 -0
- package/md_cg/test_tasks.py +409 -409
- package/md_cg/test_tenant_env_override_warn.py +139 -0
- package/md_cg/test_tenant_registry_corrupt_warn.py +151 -0
- package/md_cg/test_tool_face.py +189 -189
- package/md_cg/test_transfer.py +180 -180
- package/md_cg/test_trust.py +361 -361
- package/md_cg/test_twophase.py +286 -286
- package/md_cg/test_v14_fixes.py +38 -20
- package/md_cg/test_validity_filter.py +280 -280
- package/md_cg/test_verify_answer.py +138 -138
- package/md_cg/test_verify_dirty_reconcile.py +157 -0
- package/md_cg/test_wisdom_md_store.py +292 -292
- package/md_cg/test_writelimit.py +197 -197
- package/md_cg/test_writepipe.py +214 -214
- package/md_cg/theory.py +6 -3
- package/md_cg/tokens.py +85 -14
- package/md_cg/tool_face.py +260 -260
- package/md_cg/trust.py +986 -950
- package/md_cg/twophase.py +231 -231
- package/md_cg/units.py +668 -667
- package/md_cg/vision_evidence.py +667 -666
- package/md_cg/weights.py +624 -624
- package/md_cg/whitebox.py +527 -527
- package/md_cg/whitebox_kb/__init__.py +37 -37
- package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
- package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
- package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
- package/md_cg/whitebox_kb/engine.py +310 -310
- package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
- package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
- package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
- package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
- package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
- package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
- package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
- package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
- package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
- package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
- package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
- package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
- package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
- package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
- package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
- package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
- package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
- package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
- package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
- package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
- package/md_cg/whitebox_kb/wisdom/multilang_ir.py +128 -124
- package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
- package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
- package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
- package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
- package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
- package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
- package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
- package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
- package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
- package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
- package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
- package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
- package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
- package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
- package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
- package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
- package/md_cg/writelimit.py +356 -356
- package/md_cg/writepipe.py +20 -8
- package/package.json +101 -96
- package/skills/plugin.json +54 -54
- package/skills/skills/designer-perspective/SKILL.md +158 -158
- package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
- package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
- package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
- package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
- package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
- package/skills/skills/designer-perspective/scripts/designer.py +545 -545
- package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
- package/skills/skills/designer-perspective/tests/selftest.py +61 -61
- package/skills/skills/lingshu-browser/SKILL.md +60 -60
- package/skills/skills/lingshu-compiler/SKILL.md +56 -56
- package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
- package/skills/skills/lingshu-graph/SKILL.md +63 -63
- package/skills/skills/lingshu-net/SKILL.md +48 -48
- package/skills/skills/lingshu-os/SKILL.md +64 -64
- package/skills/skills/lingshu-pylang/SKILL.md +71 -71
- package/src/bridge.ts +33 -0
- package/src/hooks.ts +38 -2
- package/src/index.ts +526 -518
- package/src/lib/datapath.ts +326 -326
- package/src/lib/mdcg_client.ts +413 -413
- package/src/lib/mutual.ts +428 -428
- package/src/lib/prompt_safety.ts +62 -62
- package/src/lib/python_path.ts +71 -71
- package/src/lib/roleplay_web.ts +116 -29
- package/src/lib/token_store.ts +13 -3
- package/src/tools.ts +212 -212
- package/zcode/AGENTS.md +11 -3
- package/zcode/README.md +41 -41
- /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
package/md_cg/conformance.py
CHANGED
|
@@ -1,727 +1,727 @@
|
|
|
1
|
-
# -*- coding: utf-8 -*-
|
|
2
|
-
"""md_cg · 数据健康不变量断言集(Pi⑤)+ G1 类型空间正交性审计 + G3 unanalyzed 显式占位
|
|
3
|
-
|
|
4
|
-
把 2026-09-15 的一次性审计改写为**周期可复跑**的断言集(混层比例阈值 / 重复度阈值 /
|
|
5
|
-
字段覆盖率下限 / 闸门四态分布),挂 sustain 周期巡检;超阈值**只告警,不自动改数据**。
|
|
6
|
-
|
|
7
|
-
三条纪律:
|
|
8
|
-
1. **不变量 fail-closed**:类型空间封闭性 / 边键规范 / 边目标可解析 / 索引↔盘一致 /
|
|
9
|
-
重复度不劣化——任一 FAIL = 库结构失真,须人工处置。
|
|
10
|
-
2. **健康指标只告警**:混层比 / role 覆盖率 / evidence 覆盖率 / 分析覆盖率 /
|
|
11
|
-
闸门四态 / 触达率——低于靶值 WARN,不阻断、不改数据。
|
|
12
|
-
3. **缺数据源 → BLINDSPOT**(如实上报),绝不冒充 PASS。
|
|
13
|
-
|
|
14
|
-
G1(Ghidra 交接 §5.4):输出各类型空间的「声明值 / 实测值 / 未声明已用 / 声明未用 /
|
|
15
|
-
跨空间重叠 / 方向口径分歧」——类型空间不封闭是写入侧漂移的前兆。
|
|
16
|
-
|
|
17
|
-
G3(同 §5.4):unanalyzed 采用**派生断言**(不新增字段、零写入),判据
|
|
18
|
-
`evidence_count==0 ∧ verification_basis 空 ∧ lifecycle_state 空`。
|
|
19
|
-
不做字段级占位的理由:生命周期是单向降级轴,与分析覆盖正交,塞同一字段破坏正交性。
|
|
20
|
-
|
|
21
|
-
用法:
|
|
22
|
-
python -m md_cg.conformance [--root R] [--json out.json] [--baseline b.json]
|
|
23
|
-
[--no-path-check] [--strict]
|
|
24
|
-
"""
|
|
25
|
-
from __future__ import annotations
|
|
26
|
-
|
|
27
|
-
import argparse
|
|
28
|
-
import json
|
|
29
|
-
import os
|
|
30
|
-
import sys
|
|
31
|
-
import time
|
|
32
|
-
from collections import Counter, deque
|
|
33
|
-
|
|
34
|
-
REPORT_VERSION = 1
|
|
35
|
-
INDEX_FILE = "_index.json"
|
|
36
|
-
ACCESS_LOG = "_access.log"
|
|
37
|
-
DECISION_LOG = os.path.join("hippocampus", "decisions.jsonl")
|
|
38
|
-
INBOX_LOG = os.path.join("hippocampus", "inbox.jsonl")
|
|
39
|
-
DEFAULT_ROOT = os.path.join(
|
|
40
|
-
os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "data", "mdcg")
|
|
41
|
-
#: _audit.jsonl 只读尾部窗口行数(append-only 全史可数十 MB,周期巡检不扫全史)
|
|
42
|
-
AUDIT_TAIL = 20000
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
# 生效条件:无入参,datapath.mdcg_root() 返回真值时返回该值,导入或调用抛异常、或返回假值(如空串)时返回模块常量 DEFAULT_ROOT;
|
|
46
|
-
def _default_root() -> str:
|
|
47
|
-
"""根解析沿用本仓约定:env `MDCG_ROOT` > `paths.json`(用户级,旧包内兼容读)
|
|
48
|
-
> 用户级状态根 `data/mdcg`。
|
|
49
|
-
|
|
50
|
-
走 `datapath.mdcg_root()`(兜底纪律:与其它工具同源解析,不另立一套)。
|
|
51
|
-
"""
|
|
52
|
-
try:
|
|
53
|
-
from .datapath import mdcg_root
|
|
54
|
-
got = mdcg_root()
|
|
55
|
-
if got:
|
|
56
|
-
return got
|
|
57
|
-
except Exception: # noqa: BLE001
|
|
58
|
-
pass
|
|
59
|
-
return DEFAULT_ROOT
|
|
60
|
-
|
|
61
|
-
# 阈值口径 = 9-15 审计基线(2026-09-16 复跑实测:混层 6468/11076=58.4% / role 2.13% /
|
|
62
|
-
# verification_basis 60.3% / evidence 0.10% / dir_mismatch 0 / 闸门 142 条) + M4 治理靶值。
|
|
63
|
-
# 比率型阈值只表达「离靶多远」,不阻断任何写入。
|
|
64
|
-
# 注:触达率随 _access.log 滚动追加而单调增长,9-15 快照 7.0% → 09-16 实测 8.7%,
|
|
65
|
-
# 差异来自时间而非口径;故**不纳入基线不劣化比对**(日志轮转会误报)。
|
|
66
|
-
THRESHOLDS = {
|
|
67
|
-
"stratum_ratio_max": 0.60,
|
|
68
|
-
"stratum_target": 0.30,
|
|
69
|
-
"role_coverage_min": 0.90,
|
|
70
|
-
"evidence_coverage_min": 0.50,
|
|
71
|
-
"analysis_coverage_min": 0.90,
|
|
72
|
-
"reach_ratio_min": 0.30,
|
|
73
|
-
"gate_sample_min": 20,
|
|
74
|
-
# 样本下限:低于此值比率型指标不可判(如仓内自建语料),一律 BLINDSPOT 不冒充 PASS
|
|
75
|
-
"min_nodes": 200,
|
|
76
|
-
}
|
|
77
|
-
|
|
78
|
-
#: 单一内容指纹的最大同组节点数(超过即「同模板批量写入」体征,须人工治理)
|
|
79
|
-
DUP_GROUP_MAX = 200
|
|
80
|
-
|
|
81
|
-
# 写侧规范边键。取证更正(2026-09-16):subgraph.children_index:122 / parents_index:162
|
|
82
|
-
# 已含 `or e.get("type")` 容错,**读侧不忽略** legacy `type`(`subgraph` 内 normalize_edge
|
|
83
|
-
# 的 docstring 仍写「会被静默忽略」,属文档滞后于代码,已登记 DOC_CODE_DRIFT)。
|
|
84
|
-
# 仍记为规范项:chain.edge_rel 与 subgraph 双兼容是本库私约,非 md_cg 读侧(Rust 引擎 /
|
|
85
|
-
# 外部消费者)可能只读 relation_type。
|
|
86
|
-
CANONICAL_EDGE_KEYS = ("relation_type", "relation")
|
|
87
|
-
LAYER_DIRS = ("knowledge", "contextual", "self", "structural", "anchor",
|
|
88
|
-
"rejected", "unresolved", "goals", "data", "trash")
|
|
89
|
-
|
|
90
|
-
# G1 静态登记:已取证的方向口径分歧(事实,非推断)
|
|
91
|
-
KNOWN_DIRECTION_CONFLICTS = [
|
|
92
|
-
{"literal": "hierarchical",
|
|
93
|
-
"conflict": "md_cg subgraph 视其为「本节点是子、target 是父」"
|
|
94
|
-
"(children_index:128);白箱源库语义为 source 是父 —— 同名反义,"
|
|
95
|
-
"直接迁移整树倒置",
|
|
96
|
-
"workaround": "migrate_wisdom_graph.SRC_REL_MAP 迁移时改写为 contains(语义等价)"},
|
|
97
|
-
]
|
|
98
|
-
# G1 静态登记:已核对**无**方向分歧的项(防止「看起来像冲突」被反复误报)
|
|
99
|
-
VERIFIED_CONSISTENT = [
|
|
100
|
-
{"literal": "part_of / parent_of / contains",
|
|
101
|
-
"verified": "children_index:128-135 与 parents_index:171-178 逐字对称、方向取反,自洽"},
|
|
102
|
-
]
|
|
103
|
-
# G1 已取证的**文档滞后于代码**实例(陈述与实现不符,非缺陷但会误导写侧)
|
|
104
|
-
DOC_CODE_DRIFT = [
|
|
105
|
-
{"where": "subgraph.normalize_edge docstring:96",
|
|
106
|
-
"claims": "「早期迁移器写的 type 会被静默忽略(subgraph 不兼容)」",
|
|
107
|
-
"fact": "children_index:122 / parents_index:162 已含 `or e.get(\"type\")` 容错 —— "
|
|
108
|
-
"读侧兼容,陈述已过期"},
|
|
109
|
-
]
|
|
110
|
-
|
|
111
|
-
# G1 静态登记:新增类型的散落点(回答「新增类型要不要改引擎」)
|
|
112
|
-
NEW_TYPE_TOUCHPOINTS = {
|
|
113
|
-
"layer": ["mdcg.LAYERS", "mdcg.BUCKETED_LAYERS", "mdcg.NEG_MEMORY_MARKS",
|
|
114
|
-
"nodefile(fm 契约)", "routing(bucket)", "MCP 工具面描述"],
|
|
115
|
-
"edge_type": ["chain.EDGE_WEIGHTS", "subgraph(父子方向分支)",
|
|
116
|
-
"migrate_wisdom_graph.SRC_REL_MAP"],
|
|
117
|
-
}
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
# 生效条件:root 下 INDEX_FILE 可读且 JSON 顶层为 dict 并含 nodes 字典时返回该 nodes;不可读抛 OSError、JSON 非法抛 JSONDecodeError、nodes 非 dict 抛 ValueError;
|
|
121
|
-
def load_index(root: str) -> dict:
|
|
122
|
-
"""读索引快照(唯一必需数据源);损坏即抛——断言集不建立在猜测上。"""
|
|
123
|
-
with open(os.path.join(root, INDEX_FILE), encoding="utf-8") as f:
|
|
124
|
-
raw = json.load(f)
|
|
125
|
-
nodes = raw.get("nodes") if isinstance(raw, dict) else None
|
|
126
|
-
if not isinstance(nodes, dict):
|
|
127
|
-
raise ValueError(f"索引格式异常(缺 nodes 字典): {root}/{INDEX_FILE}")
|
|
128
|
-
return nodes
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
# 生效条件:path 不存在(os.path.exists 为假)时返回 [];存在时逐行解析,空行与 JSONDecodeError 行被跳过,返回可解析记录的列表(可能为 []);
|
|
132
|
-
def _read_jsonl(path: str) -> list:
|
|
133
|
-
if not os.path.exists(path):
|
|
134
|
-
return []
|
|
135
|
-
out = []
|
|
136
|
-
with open(path, encoding="utf-8", errors="replace") as f:
|
|
137
|
-
for line in f:
|
|
138
|
-
line = line.strip()
|
|
139
|
-
if not line:
|
|
140
|
-
continue
|
|
141
|
-
try:
|
|
142
|
-
out.append(json.loads(line))
|
|
143
|
-
except json.JSONDecodeError:
|
|
144
|
-
continue
|
|
145
|
-
return out
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
# 生效条件:rec 的 tags 缺失或为假值时按空列表处理,仅对含 ":" 的 tag 取首个冒号前的前缀构成集合并返回;无此类 tag 时返回空集合;
|
|
149
|
-
def _tag_prefixes(rec: dict) -> set:
|
|
150
|
-
return {str(t).split(":", 1)[0] for t in (rec.get("tags") or []) if ":" in str(t)}
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
# 生效条件:r 的 verification_basis 为 list/tuple 时返回其真值元素的 str 列表;否则该值真值时返回 [str(b)],缺失或为假值(None/空串/空容器)时返回 [];
|
|
154
|
-
def _basis_of(r: dict):
|
|
155
|
-
b = r.get("verification_basis")
|
|
156
|
-
if isinstance(b, (list, tuple)):
|
|
157
|
-
return [str(x) for x in b if x]
|
|
158
|
-
return [str(b)] if b else []
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
# 生效条件:无入参,返回 layer(取自导入的模块常量 LAYERS 的排序值)+ edge_type/derived_relation/verification_basis/lifecycle_state 五键声明值映射,后四键对应模块导入或属性取值失败、属性值为假值(空容器/None)时该键 declared=None;
|
|
162
|
-
def enum_spaces() -> dict:
|
|
163
|
-
"""各类型空间的声明值(模块真源取不到 → declared=None,报告标 BLINDSPOT)。"""
|
|
164
|
-
from .mdcg import LAYERS
|
|
165
|
-
spaces = {"layer": {"declared": sorted(LAYERS), "source": "mdcg.LAYERS"}}
|
|
166
|
-
|
|
167
|
-
# 生效条件:以 md_cg.<mod> 导入并 getattr(attr)(导入或取值抛异常则 val=None),把 {"declared": sorted(val) if val else None, "source": "<mod>.<attr>"} 写入片段外闭包变量 spaces 的 key 键(val 为假值时 declared=None),无返回值;
|
|
168
|
-
def _grab(mod: str, attr: str, key: str):
|
|
169
|
-
try:
|
|
170
|
-
m = __import__(f"md_cg.{mod}", fromlist=[attr])
|
|
171
|
-
val = getattr(m, attr, None)
|
|
172
|
-
except Exception: # noqa: BLE001
|
|
173
|
-
val = None
|
|
174
|
-
spaces[key] = {"declared": sorted(val) if val else None,
|
|
175
|
-
"source": f"{mod}.{attr}"}
|
|
176
|
-
|
|
177
|
-
_grab("chain", "EDGE_WEIGHTS", "edge_type")
|
|
178
|
-
_grab("provenance", "RELATIONS", "derived_relation")
|
|
179
|
-
_grab("nodefile", "VERIFICATION_BASIS", "verification_basis")
|
|
180
|
-
_grab("lifecycle", "STATES", "lifecycle_state")
|
|
181
|
-
return spaces
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
# ---------------------------- 指标 ----------------------------
|
|
185
|
-
|
|
186
|
-
# 生效条件:对 nodes 各记录按 str(content_hash) 计数,content_hash 缺失被折成 "None",返回出现次数 >1 且哈希不为 "None"/"" 的组数、涉及节点数与前 5 大组大小;
|
|
187
|
-
def _dup_metrics(nodes: dict) -> dict:
|
|
188
|
-
ch = Counter(str(r.get("content_hash")) for r in nodes.values())
|
|
189
|
-
groups = {h: c for h, c in ch.items() if c > 1 and h not in ("None", "")}
|
|
190
|
-
return {"groups": len(groups), "nodes": sum(groups.values()),
|
|
191
|
-
"largest": sorted(groups.values(), reverse=True)[:5]}
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
# 生效条件:对 nodes 中每个节点的 "edges" 真值且为 list/tuple 的边列表,仅遍历其中 dict 边;依据 CANONICAL_EDGE_KEYS 与 "type" 统计键组合、关系类型、总边数、非规范键样本(最多 5)和悬空边,返回汇总字典。
|
|
195
|
-
def _edge_metrics(nodes: dict) -> dict:
|
|
196
|
-
types, key_mix = Counter(), Counter()
|
|
197
|
-
total = dangling = with_edges = non_canonical = 0
|
|
198
|
-
samples = []
|
|
199
|
-
for nid, r in nodes.items():
|
|
200
|
-
es = r.get("edges") or []
|
|
201
|
-
if not isinstance(es, (list, tuple)) or not es:
|
|
202
|
-
continue
|
|
203
|
-
with_edges += 1
|
|
204
|
-
for e in es:
|
|
205
|
-
if not isinstance(e, dict):
|
|
206
|
-
continue
|
|
207
|
-
total += 1
|
|
208
|
-
keys = [k for k in CANONICAL_EDGE_KEYS + ("type",) if e.get(k)]
|
|
209
|
-
types[str(e.get("relation_type") or e.get("relation") or e.get("type"))] += 1
|
|
210
|
-
key_mix["+".join(keys) if keys else "<无关系键>"] += 1
|
|
211
|
-
if not any(k in CANONICAL_EDGE_KEYS for k in keys):
|
|
212
|
-
non_canonical += 1
|
|
213
|
-
if len(samples) < 5:
|
|
214
|
-
samples.append({"node": nid, "edge": dict(e)})
|
|
215
|
-
tgt = e.get("target") or e.get("target_id")
|
|
216
|
-
if tgt is None or str(tgt) not in nodes:
|
|
217
|
-
dangling += 1
|
|
218
|
-
return {"nodes_with_edges": with_edges, "total": total,
|
|
219
|
-
"types": dict(types.most_common(20)), "key_mix": dict(key_mix.most_common(10)),
|
|
220
|
-
"non_canonical": non_canonical, "dangling": dangling, "samples": samples}
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
# 生效条件:遍历 nodes 的值作为记录,path 经 str(r.get("path") or "") 为空即 missing++ 并 continue,不以记录含 layer/path 为前置;仅对非空 path 执行 layer 与 head 检查(layer 非空且 head != layer 且 head in LAYER_DIRS 时 layer_mismatch++),且 check_exists 为 True 时才对非空 path 以 os.path.join(root, rel) 判断不存在并计入 missing,最终返回 missing/layer_mismatch/samples 统计;。
|
|
224
|
-
def _path_metrics(root: str, nodes: dict, check_exists: bool) -> dict:
|
|
225
|
-
missing = mismatch = 0
|
|
226
|
-
samples = []
|
|
227
|
-
for r in nodes.values():
|
|
228
|
-
layer, p = str(r.get("layer") or ""), str(r.get("path") or "")
|
|
229
|
-
if not p:
|
|
230
|
-
missing += 1
|
|
231
|
-
if len(samples) < 5:
|
|
232
|
-
samples.append({"id": r.get("id"), "why": "path 缺失"})
|
|
233
|
-
continue
|
|
234
|
-
rel = p.replace("\\", "/")
|
|
235
|
-
head = rel.split("/", 1)[0]
|
|
236
|
-
if layer and head != layer and head in LAYER_DIRS:
|
|
237
|
-
mismatch += 1
|
|
238
|
-
if len(samples) < 5:
|
|
239
|
-
samples.append({"id": r.get("id"), "path": p, "layer": layer})
|
|
240
|
-
if check_exists and not os.path.exists(os.path.join(root, rel)):
|
|
241
|
-
missing += 1
|
|
242
|
-
if len(samples) < 5:
|
|
243
|
-
samples.append({"id": r.get("id"), "path": p, "why": "文件不存在"})
|
|
244
|
-
return {"missing": missing, "layer_mismatch": mismatch, "samples": samples}
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
# 生效条件:root 下 DECISION_LOG/INBOX_LOG 不存在时按空列表计,返回 decisions(decision 或 status 回落 "?" 的前 10 项)、decisions_total、inbox_pending=max(0, 收件数−决策数)、以及 _audit.jsonl 末尾 AUDIT_TAIL 行内 op 计数的前 6 项与尾窗元信息;
|
|
248
|
-
def _gate_metrics(root: str) -> dict:
|
|
249
|
-
dec = _read_jsonl(os.path.join(root, DECISION_LOG))
|
|
250
|
-
inbox = _read_jsonl(os.path.join(root, INBOX_LOG))
|
|
251
|
-
ops = Counter()
|
|
252
|
-
# _audit.jsonl 是 append-only 且可达数十 MB(真源 15.9MB / 7.4 万行)——
|
|
253
|
-
# 只读尾部窗口(近 AUDIT_TAIL 行)取 op 分布,避免周期巡检在此处 O(全史)。
|
|
254
|
-
audit = os.path.join(root, "_audit.jsonl")
|
|
255
|
-
tail = []
|
|
256
|
-
if os.path.exists(audit):
|
|
257
|
-
with open(audit, encoding="utf-8", errors="replace") as f:
|
|
258
|
-
tail = list(deque(f, maxlen=AUDIT_TAIL))
|
|
259
|
-
for line in tail:
|
|
260
|
-
if not line.strip():
|
|
261
|
-
continue
|
|
262
|
-
try:
|
|
263
|
-
ops[str(json.loads(line).get("op"))] += 1
|
|
264
|
-
except json.JSONDecodeError:
|
|
265
|
-
continue
|
|
266
|
-
return {"decisions": dict(Counter(str(r.get("decision") or r.get("status") or "?")
|
|
267
|
-
for r in dec).most_common(10)),
|
|
268
|
-
"decisions_total": len(dec),
|
|
269
|
-
"inbox_pending": max(0, len(inbox) - len(dec)),
|
|
270
|
-
"audit_ops": dict(ops.most_common(6)), "audit_tail_lines": len(tail),
|
|
271
|
-
"audit_tail_window": AUDIT_TAIL}
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
# 生效条件:对 root 与 nodes,若 root/ACCESS_LOG 路径不存在则返回 {"distinct": None, "ratio": None, "lines": None};否则逐行解析 JSON,当 "ids" 为 list 或 tuple 时把其中真值元素 str 后加入 seen,且当 "id"/"node_id"/"nid" 任一真值时把其 str 加入 seen,返回 distinct 为 seen 与 nodes 键交集数量、ratio 为 distinct/max(len(nodes),1)、lines 为非空行数。
|
|
275
|
-
def _reach_metrics(root: str, nodes: dict) -> dict:
|
|
276
|
-
p = os.path.join(root, ACCESS_LOG)
|
|
277
|
-
if not os.path.exists(p):
|
|
278
|
-
return {"distinct": None, "ratio": None, "lines": None}
|
|
279
|
-
seen, lines = set(), 0
|
|
280
|
-
with open(p, encoding="utf-8", errors="replace") as f:
|
|
281
|
-
for line in f:
|
|
282
|
-
line = line.strip()
|
|
283
|
-
if not line:
|
|
284
|
-
continue
|
|
285
|
-
lines += 1
|
|
286
|
-
try:
|
|
287
|
-
rec = json.loads(line)
|
|
288
|
-
except json.JSONDecodeError:
|
|
289
|
-
continue
|
|
290
|
-
# 实际落盘形态:{"t":..., "ids":[<nid>,...], "tier":...}(_access.log 逐次追加)
|
|
291
|
-
got = rec.get("ids")
|
|
292
|
-
if isinstance(got, (list, tuple)):
|
|
293
|
-
seen.update(str(x) for x in got if x)
|
|
294
|
-
nid = rec.get("id") or rec.get("node_id") or rec.get("nid")
|
|
295
|
-
if nid:
|
|
296
|
-
seen.add(str(nid))
|
|
297
|
-
hit = len(seen & set(nodes))
|
|
298
|
-
return {"distinct": hit, "ratio": hit / max(len(nodes), 1), "lines": lines}
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
# 生效条件:对 nodes,筛出 layer 字符串为 "knowledge" 的 kn;mixed 为 kn 中标签前缀含 doc 或 code 的数量;返回全库 nodes 数与 knowledge 层数、mixed、mixed_ratio=mixed/max(len(kn),1),以及全库口径 role_ratio/basis_ratio/evidence_ratio/state_ratio 和 knowledge 层口径 role_ratio_kn/basis_ratio_kn/evidence_ratio_kn。
|
|
302
|
-
def _coverage_metrics(nodes: dict) -> dict:
|
|
303
|
-
"""混层口径 + 字段覆盖率(9-15 审计基线口径,逐项可复现)。
|
|
304
|
-
|
|
305
|
-
**分母口径取证(2026-09-16,关键)**:审计的「role 99.99% 为空」「evidence_count
|
|
306
|
-
仅 6 条」是 **knowledge 层**分母;全库分母会摊薄成「role 2.13% / evidence 12 条」,
|
|
307
|
-
把一个 99.99% 的空缺伪装成「还行」。故两类分母并存上报,**健康判据一律取
|
|
308
|
-
knowledge 层口径**(`*_ratio_kn`),全库口径只作对照片段。
|
|
309
|
-
"""
|
|
310
|
-
kn = [r for r in nodes.values() if str(r.get("layer")) == "knowledge"]
|
|
311
|
-
mixed = sum(1 for r in kn if {"doc", "code"} & _tag_prefixes(r))
|
|
312
|
-
n, nk = len(nodes), max(len(kn), 1)
|
|
313
|
-
return {"nodes": n, "knowledge": len(kn), "mixed_layer": mixed,
|
|
314
|
-
"mixed_ratio": mixed / nk,
|
|
315
|
-
"role_ratio": sum(1 for r in nodes.values() if r.get("role")) / max(n, 1),
|
|
316
|
-
"basis_ratio": sum(1 for r in nodes.values() if _basis_of(r)) / max(n, 1),
|
|
317
|
-
"evidence_ratio": sum(1 for r in nodes.values()
|
|
318
|
-
if _as_int(r.get("evidence_count")) > 0) / max(n, 1),
|
|
319
|
-
"state_ratio": sum(1 for r in nodes.values()
|
|
320
|
-
if r.get("lifecycle_state")) / max(n, 1),
|
|
321
|
-
# —— knowledge 层口径(= 审计原口径,判据用这组)——
|
|
322
|
-
"role_ratio_kn": sum(1 for r in kn if r.get("role")) / nk,
|
|
323
|
-
"basis_ratio_kn": sum(1 for r in kn if _basis_of(r)) / nk,
|
|
324
|
-
"evidence_ratio_kn": sum(1 for r in kn
|
|
325
|
-
if _as_int(r.get("evidence_count")) > 0) / nk}
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
# 生效条件:int(v) 转换成功时返回该整数;抛 TypeError/ValueError(如 None、非数字串)时返回 0;
|
|
329
|
-
def _as_int(v) -> int:
|
|
330
|
-
try:
|
|
331
|
-
return int(v)
|
|
332
|
-
except (TypeError, ValueError):
|
|
333
|
-
return 0
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
# ---------------------------- G3 · unanalyzed 派生集合 ----------------------------
|
|
337
|
-
|
|
338
|
-
# 生效条件:仅当记录 _as_int(evidence_count)==0 且 verification_basis 为空且 lifecycle_state 为假值时计入 ids,返回 count、ratio=count/max(len(nodes),1) 与前 5 个 id 样本;
|
|
339
|
-
def unanalyzed(nodes: dict) -> dict:
|
|
340
|
-
"""G3:unanalyzed **派生**判据(不新增字段、零写入)。
|
|
341
|
-
|
|
342
|
-
判据 = `evidence_count==0 ∧ verification_basis 空 ∧ lifecycle_state 空`
|
|
343
|
-
—— 三者皆空 = 「无一维分析痕迹」,与「分析结论为负」不同(后者会留 basis)。
|
|
344
|
-
|
|
345
|
-
不做字段级占位的理由(Ghidra 交接 §5.4):
|
|
346
|
-
* 生命周期是**单向降级轴**(active→converged→demoted→archived),与分析覆盖正交,
|
|
347
|
-
塞进同一字段即破坏正交性;
|
|
348
|
-
* **逃逸口已存在**——`verification_basis` 声明集里有 `other`(真源 273 条在用),
|
|
349
|
-
「分析过但无法归类」有明确归宿,故「basis 为空」不再与「分析结论为空」混淆。
|
|
350
|
-
"""
|
|
351
|
-
ids = [nid for nid, r in nodes.items()
|
|
352
|
-
if _as_int(r.get("evidence_count")) == 0
|
|
353
|
-
and not _basis_of(r)
|
|
354
|
-
and not r.get("lifecycle_state")]
|
|
355
|
-
return {"count": len(ids), "ratio": len(ids) / max(len(nodes), 1),
|
|
356
|
-
"sample": ids[:5]}
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
# ---------------------------- G1 · 类型空间正交性审计 ----------------------------
|
|
360
|
-
|
|
361
|
-
# 生效条件:调用 _usage(nodes, edges) 时 edges 形参未被源码引用,对 nodes.values() 中每个节点 r,仅当 r.get("layer")、r.get("role")、r.get("derived_relation")、r.get("lifecycle_state") 为真值时分别以 str 值计入对应 Counter,_basis_of(r) 与 _tag_prefixes(r) 展开的元素分别计入 verification_basis 与 tag_prefix,再遍历各 r 的 r.get("edges") or [] 中 isinstance(e, dict) 的项按 edge_rel(e) 计数,最终 edge_type 仅保留键为真值的计数,返回 used 字典;
|
|
362
|
-
def _usage(nodes: dict, edges: dict) -> dict:
|
|
363
|
-
"""各类型空间的**实测**取值(与 enum_spaces 的声明值对照)。"""
|
|
364
|
-
from .chain import edge_rel
|
|
365
|
-
used = {
|
|
366
|
-
"layer": Counter(str(r.get("layer")) for r in nodes.values() if r.get("layer")),
|
|
367
|
-
"role": Counter(str(r.get("role")) for r in nodes.values() if r.get("role")),
|
|
368
|
-
"verification_basis": Counter(b for r in nodes.values() for b in _basis_of(r)),
|
|
369
|
-
"derived_relation": Counter(str(r.get("derived_relation"))
|
|
370
|
-
for r in nodes.values() if r.get("derived_relation")),
|
|
371
|
-
"lifecycle_state": Counter(str(r.get("lifecycle_state"))
|
|
372
|
-
for r in nodes.values() if r.get("lifecycle_state")),
|
|
373
|
-
"tag_prefix": Counter(p for r in nodes.values() for p in _tag_prefixes(r)),
|
|
374
|
-
}
|
|
375
|
-
rels = Counter()
|
|
376
|
-
for r in nodes.values():
|
|
377
|
-
for e in (r.get("edges") or []):
|
|
378
|
-
if isinstance(e, dict):
|
|
379
|
-
rels[edge_rel(e)] += 1
|
|
380
|
-
used["edge_type"] = Counter({k: v for k, v in rels.items() if k})
|
|
381
|
-
return used
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
# 生效条件:以 enum_spaces 的声明集与 _usage 的实测集逐空间对照(closed 仅在 declared 非 None 且无未声明已用时为 True),role 与 tag_prefix 两空间的 declared/closed 恒为 None,另附 _cross_space_overlap(字面量出现在 ≥2 空间或 layer∩tag_prefix)及 KNOWN_DIRECTION_CONFLICTS/VERIFIED_CONSISTENT/DOC_CODE_DRIFT/NEW_TYPE_TOUCHPOINTS 四个常量;
|
|
385
|
-
def _g1_audit(nodes: dict, edges: dict) -> dict:
|
|
386
|
-
"""G1:声明值 / 实测值 / 未声明已用 / 声明未用 / 跨空间重叠 / 方向口径分歧。"""
|
|
387
|
-
spaces = enum_spaces()
|
|
388
|
-
used = _usage(nodes, edges)
|
|
389
|
-
out = {}
|
|
390
|
-
for name, spec in spaces.items():
|
|
391
|
-
dec = set(spec.get("declared") or [])
|
|
392
|
-
u = set(used.get(name, {}).keys())
|
|
393
|
-
out[name] = {
|
|
394
|
-
"source": spec.get("source"),
|
|
395
|
-
"declared": sorted(dec),
|
|
396
|
-
"used": dict(used.get(name, {}).most_common(30)),
|
|
397
|
-
"undeclared_used": sorted(u - dec),
|
|
398
|
-
"declared_unused": sorted(dec - u),
|
|
399
|
-
"closed": (spec.get("declared") is not None and not (u - dec)),
|
|
400
|
-
}
|
|
401
|
-
for name in ("role", "tag_prefix"):
|
|
402
|
-
out[name] = {"source": None, "declared": None,
|
|
403
|
-
"used": dict(used.get(name, {}).most_common(30)),
|
|
404
|
-
"undeclared_used": sorted(used.get(name, {})),
|
|
405
|
-
"declared_unused": [], "closed": None}
|
|
406
|
-
# 跨空间重叠:同一字面量出现在 ≥2 个空间 → 解析歧义风险
|
|
407
|
-
owners = Counter()
|
|
408
|
-
for name, spec in out.items():
|
|
409
|
-
for lit in (spec.get("declared") or []):
|
|
410
|
-
owners[str(lit)] += 1
|
|
411
|
-
out["_cross_space_overlap"] = sorted(l for l, c in owners.items() if c > 1) + \
|
|
412
|
-
sorted(set(used.get("layer", {})) & set(used.get("tag_prefix", {})))
|
|
413
|
-
out["_direction_conflicts"] = KNOWN_DIRECTION_CONFLICTS
|
|
414
|
-
out["_verified_consistent"] = VERIFIED_CONSISTENT
|
|
415
|
-
out["_doc_code_drift"] = DOC_CODE_DRIFT
|
|
416
|
-
out["_new_type_touchpoints"] = NEW_TYPE_TOUCHPOINTS
|
|
417
|
-
return out
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
# ---------------------------- 断言集 ----------------------------
|
|
421
|
-
|
|
422
|
-
# 生效条件:四参齐备时返回 {"id": cid, "level": level, "ok": bool(ok), "detail": detail},ok 经 bool() 归一为布尔;
|
|
423
|
-
def _ck(cid: str, level: str, ok: bool, detail: str) -> dict:
|
|
424
|
-
return {"id": cid, "level": level, "ok": bool(ok), "detail": detail}
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
# 生效条件:调用 check(root, check_paths, baseline, strict) 时,root 原样传入 load_index 得 nodes,check_paths 原样传入 _path_metrics 并在 index.path.present 检查消息中当其为假时追加“(--no-path-check 未查盘)”,baseline 为真时追加 _regression 基线不劣化检查、为假(None 或空 dict)时跳过,n=len(nodes) 小于 THRESHOLDS["min_nodes"] 时 small 为真且使 _ratio 各项及 reach.ratio、stratum.mixed_ratio 在 val 为 None 或 small 为真时记 BLINDSPOT、否则记 WARN,最终 verdict 为 FAIL(存在 level="FAIL" 且 ok 为假)、否则 WARN(存在 level="WARN" 且 ok 为假,或 strict 为真且存在 level="BLINDSPOT")、否则 PASS,返回含 REPORT_VERSION、t、elapsed_s、root 绝对路径、verdict、nodes、checks、counts、coverage、dup、edges、paths、gate、reach、unanalyzed、g1、protocol(=protocol.audit() 静态对账,恒追加四条 FAIL 级断言:op.declared 无未登记分支 / verb.implemented live 动词均有实现分支 / verb.reserved_clean 预留动词未被静默实现 / shape.declared 形状声明完备)、thresholds 的 dict;
|
|
428
|
-
def check(root: str, *, check_paths: bool = True,
|
|
429
|
-
baseline: dict = None, strict: bool = False) -> dict:
|
|
430
|
-
"""跑一遍全部断言,返回报告 dict(零写入:不修任何数据、不落任何文件)。"""
|
|
431
|
-
t0 = time.time()
|
|
432
|
-
nodes = load_index(root)
|
|
433
|
-
n = len(nodes)
|
|
434
|
-
small = n < THRESHOLDS["min_nodes"]
|
|
435
|
-
cov = _coverage_metrics(nodes)
|
|
436
|
-
edges = _edge_metrics(nodes)
|
|
437
|
-
paths = _path_metrics(root, nodes, check_paths)
|
|
438
|
-
gate = _gate_metrics(root)
|
|
439
|
-
reach = _reach_metrics(root, nodes)
|
|
440
|
-
un = unanalyzed(nodes)
|
|
441
|
-
g1 = _g1_audit(nodes, edges)
|
|
442
|
-
dup = _dup_metrics(nodes)
|
|
443
|
-
checks = []
|
|
444
|
-
T = THRESHOLDS
|
|
445
|
-
|
|
446
|
-
# ---- 不变量(fail-closed)----
|
|
447
|
-
for space, use_key in (("layer", "layer"), ("verification_basis", "verification_basis"),
|
|
448
|
-
("derived_relation", "derived_relation"),
|
|
449
|
-
("lifecycle_state", "lifecycle_state")):
|
|
450
|
-
spec = g1[space]
|
|
451
|
-
undecl = spec["undeclared_used"]
|
|
452
|
-
checks.append(_ck(f"enum.{space}.closed", "FAIL", not undecl,
|
|
453
|
-
f"声明源={spec['source']};未声明已用={undecl or '无'}"))
|
|
454
|
-
checks.append(_ck("edge.rel.declared", "WARN",
|
|
455
|
-
not g1["edge_type"]["undeclared_used"],
|
|
456
|
-
f"未声明边类型={g1['edge_type']['undeclared_used'] or '无'}"
|
|
457
|
-
"(读侧退化为 chain.DEFAULT_EDGE_WEIGHT=0.50,不炸但权重失真)"))
|
|
458
|
-
checks.append(_ck("edge.key.canonical", "WARN", edges["non_canonical"] == 0,
|
|
459
|
-
f"非规范键边 {edges['non_canonical']}/{edges['total']}"
|
|
460
|
-
f"(写侧规范={CANONICAL_EDGE_KEYS};md_cg 读侧兼容 type,"
|
|
461
|
-
"非 md_cg 读侧不兼容——存量债,判据=不劣化)"))
|
|
462
|
-
checks.append(_ck("edge.target.resolvable", "FAIL", edges["dangling"] == 0,
|
|
463
|
-
f"悬空边 {edges['dangling']}/{edges['total']}(图断裂)"))
|
|
464
|
-
checks.append(_ck("index.path.present", "FAIL", paths["missing"] == 0,
|
|
465
|
-
f"path 缺失/文件不存在 {paths['missing']}"
|
|
466
|
-
+ ("" if check_paths else "(--no-path-check 未查盘)")))
|
|
467
|
-
checks.append(_ck("index.layer.dir.match", "FAIL", paths["layer_mismatch"] == 0,
|
|
468
|
-
f"层-目录不一致 {paths['layer_mismatch']}(9-15 基线=0)"))
|
|
469
|
-
top_dup = dup["largest"][0] if dup["largest"] else 0
|
|
470
|
-
checks.append(_ck("dup.content.no_blowup", "FAIL", top_dup <= DUP_GROUP_MAX,
|
|
471
|
-
f"内容指纹重复组 {dup['groups']} / 涉及节点 {dup['nodes']}"
|
|
472
|
-
f" / 最大组 {dup['largest'][:3]}(最大组上限 {DUP_GROUP_MAX})"))
|
|
473
|
-
|
|
474
|
-
# ---- 健康指标(只告警)----
|
|
475
|
-
# 生效条件:val 为 None 或片段外闭包布尔 small 为真(small、n、T 均未在本片段定义)时追加 level="BLINDSPOT" 项;否则追加 level="WARN"、ok=(val>=low) 的项;两分支均无返回值;
|
|
476
|
-
def _ratio(cid, val, low, name):
|
|
477
|
-
if small or val is None:
|
|
478
|
-
checks.append(_ck(cid, "BLINDSPOT", True,
|
|
479
|
-
f"{name} 不可判(样本 {n} < {T['min_nodes']} / 数据源缺失)"))
|
|
480
|
-
return
|
|
481
|
-
checks.append(_ck(cid, "WARN", val >= low, f"{name}={val * 100:.1f}%(下限 {low:.0%})"))
|
|
482
|
-
|
|
483
|
-
mr = None if small else cov["mixed_ratio"]
|
|
484
|
-
checks.append(_ck("stratum.mixed_ratio",
|
|
485
|
-
"BLINDSPOT" if mr is None else "WARN",
|
|
486
|
-
mr is None or mr <= T["stratum_ratio_max"],
|
|
487
|
-
"混层比(knowledge 中 doc∪code 标签)=N/A(样本不足)" if mr is None else
|
|
488
|
-
f"混层比(knowledge 中 doc∪code 标签)={mr * 100:.1f}%"
|
|
489
|
-
f"(上限 {T['stratum_ratio_max']:.0%},治理靶 {T['stratum_target']:.0%})"))
|
|
490
|
-
_ratio("stratum.role_coverage", cov["role_ratio_kn"], T["role_coverage_min"],
|
|
491
|
-
"role 覆盖率[knowledge]")
|
|
492
|
-
_ratio("stratum.evidence_coverage", cov["evidence_ratio_kn"],
|
|
493
|
-
T["evidence_coverage_min"], "evidence 覆盖率[knowledge]")
|
|
494
|
-
_ratio("analysis.coverage", 1 - un["ratio"], T["analysis_coverage_min"],
|
|
495
|
-
f"分析覆盖率(非 unanalyzed;unanalyzed={un['count']})")
|
|
496
|
-
_ratio("reach.ratio", None if small else reach["ratio"], T["reach_ratio_min"], "触达率")
|
|
497
|
-
checks.append(_ck("gate.sample", "WARN",
|
|
498
|
-
gate["decisions_total"] >= T["gate_sample_min"],
|
|
499
|
-
f"闸门裁决样本 {gate['decisions_total']} 条 {gate['decisions']}"
|
|
500
|
-
f"(下限 {T['gate_sample_min']})"))
|
|
501
|
-
|
|
502
|
-
# ---- 记忆动词协议 v1 静态对账(声明 ↔ MCP 面实现,纯源码事实、不连库)----
|
|
503
|
-
# 与 G1 的分工:G1 对账**数据值域**,这里对账**动词面**——两者都是
|
|
504
|
-
# 「声明了没做 / 做了没说」的前置红灯,属结构性不变量(FAIL 级,fail-closed)。
|
|
505
|
-
from . import protocol as _proto
|
|
506
|
-
pa = _proto.audit()
|
|
507
|
-
checks.append(_ck("protocol.op.extension_surface", "WARN", True,
|
|
508
|
-
f"协议 v{pa['protocol_version']} 冻结面={pa['declared']}"
|
|
509
|
-
f"(live {len(pa['live'])} / reserved {len(pa['reserved'])});"
|
|
510
|
-
f"MCP 面另有 {len(pa['extension_ops'])} 个扩展 op 不属协议面"
|
|
511
|
-
f"(文档一致性由 cogmap 门禁承担)"))
|
|
512
|
-
checks.append(_ck("protocol.verb.implemented", "FAIL",
|
|
513
|
-
not pa["missing_impl"],
|
|
514
|
-
f"声明为 live 的动词 {pa['live']} 均有实现分支"
|
|
515
|
-
if not pa["missing_impl"] else
|
|
516
|
-
f"声明为 live 却无实现分支:{pa['missing_impl']}"))
|
|
517
|
-
checks.append(_ck("protocol.verb.reserved_clean", "FAIL",
|
|
518
|
-
not pa["reserved_leaked"],
|
|
519
|
-
f"reserved 动词 {pa['reserved']} 未出现实现分支"
|
|
520
|
-
if not pa["reserved_leaked"] else
|
|
521
|
-
f"reserved 动词被静默实现(须先改 status=live):"
|
|
522
|
-
f"{pa['reserved_leaked']}"))
|
|
523
|
-
checks.append(_ck("protocol.shape.declared", "FAIL",
|
|
524
|
-
not pa["shape_errors"],
|
|
525
|
-
f"{len(pa['declared'])} 个动词形状声明完备"
|
|
526
|
-
if not pa["shape_errors"] else
|
|
527
|
-
f"形状声明不完整:{pa['shape_errors']}"))
|
|
528
|
-
|
|
529
|
-
# ---- 基线不劣化(有 baseline 时才可判)----
|
|
530
|
-
if baseline:
|
|
531
|
-
checks += _regression({"mixed_ratio": cov["mixed_ratio"],
|
|
532
|
-
"non_canonical": edges["non_canonical"],
|
|
533
|
-
"dangling": edges["dangling"],
|
|
534
|
-
"dup_groups": dup["groups"],
|
|
535
|
-
"unanalyzed": un["count"]}, baseline)
|
|
536
|
-
|
|
537
|
-
fails = [c for c in checks if c["level"] == "FAIL" and not c["ok"]]
|
|
538
|
-
warns = [c for c in checks if c["level"] == "WARN" and not c["ok"]]
|
|
539
|
-
blind = [c for c in checks if c["level"] == "BLINDSPOT"]
|
|
540
|
-
v = "FAIL" if fails else ("WARN" if (warns or (strict and blind)) else "PASS")
|
|
541
|
-
return {"report_version": REPORT_VERSION, "t": t0, "elapsed_s": round(time.time() - t0, 2),
|
|
542
|
-
"root": os.path.abspath(root), "verdict": v,
|
|
543
|
-
"nodes": n, "checks": checks,
|
|
544
|
-
"counts": {"fail": len(fails), "warn": len(warns), "blindspot": len(blind),
|
|
545
|
-
"total": len(checks)},
|
|
546
|
-
"coverage": cov, "dup": dup, "edges": edges, "paths": paths,
|
|
547
|
-
"gate": gate, "reach": reach, "unanalyzed": un, "g1": g1,
|
|
548
|
-
"protocol": pa, "thresholds": T}
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
# 生效条件:对 cur 各键,baseline(或其 metrics)缺失、该键在基线为 None 或 cur 值为 None 时跳过;否则以 val > b+1e-9 判劣化,产出 level 恒为 "FAIL"、ok=not worse 的检查项;
|
|
552
|
-
def _regression(cur: dict, baseline: dict) -> list:
|
|
553
|
-
"""与基线比对:单调量只准不变或改善(重复度/悬空/非规范键/混层/unanalyzed)。
|
|
554
|
-
|
|
555
|
-
只收**单调有害量**(越大越坏);触达率等随运行时间自然增长的量不进比对,
|
|
556
|
-
否则日志轮转即误报。
|
|
557
|
-
"""
|
|
558
|
-
base = (baseline or {}).get("metrics") or {}
|
|
559
|
-
out = []
|
|
560
|
-
for key, val in cur.items():
|
|
561
|
-
b = base.get(key)
|
|
562
|
-
if b is None or val is None:
|
|
563
|
-
continue
|
|
564
|
-
worse = val > b + 1e-9
|
|
565
|
-
out.append(_ck(f"baseline.{key}", "FAIL", not worse,
|
|
566
|
-
f"现值 {val:.4f} vs 基线 {b:.4f}"
|
|
567
|
-
+ ("(劣化)" if worse else "(未劣化)")))
|
|
568
|
-
return out
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
# 生效条件:rep 含 coverage/edges/dup/unanalyzed/gate/reach 子字典与 nodes 键时(全部按下标取值,缺键即抛 KeyError)返回扁平指标映射;
|
|
572
|
-
def metrics_of(rep: dict) -> dict:
|
|
573
|
-
"""抽成可比对的扁平指标(--baseline 的写入面)。"""
|
|
574
|
-
return {"mixed_ratio": rep["coverage"]["mixed_ratio"],
|
|
575
|
-
"non_canonical": rep["edges"]["non_canonical"],
|
|
576
|
-
"dangling": rep["edges"]["dangling"],
|
|
577
|
-
"dup_groups": rep["dup"]["groups"],
|
|
578
|
-
"unanalyzed": rep["unanalyzed"]["count"],
|
|
579
|
-
"nodes": rep["nodes"],
|
|
580
|
-
"role_ratio": rep["coverage"]["role_ratio"],
|
|
581
|
-
"basis_ratio": rep["coverage"]["basis_ratio"],
|
|
582
|
-
"evidence_ratio": rep["coverage"]["evidence_ratio"],
|
|
583
|
-
"role_ratio_kn": rep["coverage"]["role_ratio_kn"],
|
|
584
|
-
"evidence_ratio_kn": rep["coverage"]["evidence_ratio_kn"],
|
|
585
|
-
"reach_ratio": rep["reach"]["ratio"],
|
|
586
|
-
"gate_decisions": rep["gate"]["decisions_total"]}
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
# ---------------------------- 报告 ----------------------------
|
|
590
|
-
|
|
591
|
-
# 生效条件:rep 各键齐备(按下标取值,任一键缺失抛 KeyError)时拼接为多行文本,reach.ratio 为 None 时渲染 "N/A",g1 中以 "_" 开头的键在遍历中被跳过;
|
|
592
|
-
def render(rep: dict) -> str:
|
|
593
|
-
L = []
|
|
594
|
-
a = L.append
|
|
595
|
-
a(f"== conformance v{rep['report_version']} · {rep['root']}")
|
|
596
|
-
a(f" verdict: {rep['verdict']} nodes={rep['nodes']} "
|
|
597
|
-
f"checks={rep['counts']['total']} (fail {rep['counts']['fail']} / "
|
|
598
|
-
f"warn {rep['counts']['warn']} / blindspot {rep['counts']['blindspot']})")
|
|
599
|
-
a("\n-- 断言集(FAIL=fail-closed 不变量 / WARN=健康指标只告警)--")
|
|
600
|
-
for c in rep["checks"]:
|
|
601
|
-
mark = {"FAIL": "!!", "WARN": " ?", "BLINDSPOT": " ~"}.get(c["level"], " .")
|
|
602
|
-
ok = "ok " if c["ok"] else "NO "
|
|
603
|
-
a(f" [{mark}] {ok}{c['id']}: {c['detail']}")
|
|
604
|
-
c = rep["coverage"]
|
|
605
|
-
a(f"\n-- 口径复核(9-15 基线)--")
|
|
606
|
-
a(f" nodes={c['nodes']} knowledge={c['knowledge']} 混层={c['mixed_layer']}"
|
|
607
|
-
f" ({c['mixed_ratio'] * 100:.1f}%)")
|
|
608
|
-
a(f" dir_mismatch={rep['paths']['layer_mismatch']} path_missing={rep['paths']['missing']}")
|
|
609
|
-
a(f" 覆盖率[knowledge 层·判据口径] role={c['role_ratio_kn'] * 100:.3f}%"
|
|
610
|
-
f" basis={c['basis_ratio_kn'] * 100:.1f}% evidence={c['evidence_ratio_kn'] * 100:.3f}%")
|
|
611
|
-
a(f" 覆盖率为对照[全库口径] role={c['role_ratio'] * 100:.2f}%"
|
|
612
|
-
f" basis={c['basis_ratio'] * 100:.1f}% evidence={c['evidence_ratio'] * 100:.2f}%"
|
|
613
|
-
f" state={c['state_ratio'] * 100:.1f}%")
|
|
614
|
-
a(f" 重复 content_hash 组={rep['dup']['groups']} 节点={rep['dup']['nodes']}"
|
|
615
|
-
f" 最大组={rep['dup']['largest'][:3]}")
|
|
616
|
-
a(f" 边 total={rep['edges']['total']} 非规范键={rep['edges']['non_canonical']}"
|
|
617
|
-
f" 悬空={rep['edges']['dangling']} 类型={rep['edges']['types']}")
|
|
618
|
-
r = rep["reach"]
|
|
619
|
-
a(f" 触达 distinct={r['distinct']} ratio="
|
|
620
|
-
+ ("N/A" if r['ratio'] is None else f"{r['ratio'] * 100:.1f}%") + f" log_lines={r['lines']}")
|
|
621
|
-
a(f" 闸门 decisions={rep['gate']['decisions_total']} {rep['gate']['decisions']}"
|
|
622
|
-
f" inbox_pending={rep['gate']['inbox_pending']}")
|
|
623
|
-
a(f" 审计 op(尾窗 {rep['gate']['audit_tail_lines']}"
|
|
624
|
-
f"/{rep['gate']['audit_tail_window']} 行)={rep['gate']['audit_ops']}")
|
|
625
|
-
a(f" G3 unanalyzed 派生={rep['unanalyzed']['count']}"
|
|
626
|
-
f" ({rep['unanalyzed']['ratio'] * 100:.1f}%)")
|
|
627
|
-
pr = rep.get("protocol") or {}
|
|
628
|
-
if pr:
|
|
629
|
-
a("\n-- 记忆动词协议 v1 静态对账 --")
|
|
630
|
-
a(f" {pr['doc']} frozen={pr['frozen']} ok={pr['ok']}")
|
|
631
|
-
a(f" 声明={pr['declared']}")
|
|
632
|
-
a(f" live={pr['live']} reserved={pr['reserved']}(未实现即如实登记,不冒充)")
|
|
633
|
-
a(f" MCP 面 op 分支={len(pr['actual'])} 个,其中扩展能力面"
|
|
634
|
-
f"(不属协议 v1){len(pr['extension_ops'])} 个")
|
|
635
|
-
if pr["errors"]:
|
|
636
|
-
for e in pr["errors"]:
|
|
637
|
-
a(f" !! {e}")
|
|
638
|
-
a("\n-- G1 类型空间正交性 --")
|
|
639
|
-
for name, spec in rep["g1"].items():
|
|
640
|
-
if name.startswith("_"):
|
|
641
|
-
continue
|
|
642
|
-
a(f" [{name}] 源={spec['source']} 封闭={spec['closed']}")
|
|
643
|
-
a(f" 实测={spec['used']}")
|
|
644
|
-
if spec["undeclared_used"]:
|
|
645
|
-
a(f" ★未声明已用={spec['undeclared_used']}")
|
|
646
|
-
if spec["declared_unused"]:
|
|
647
|
-
a(f" 声明未用={spec['declared_unused']}")
|
|
648
|
-
a(f" 跨空间重叠={rep['g1']['_cross_space_overlap'] or '无'}")
|
|
649
|
-
a(f" 方向口径分歧={[d['literal'] for d in rep['g1']['_direction_conflicts']]}")
|
|
650
|
-
a(f" 已核对无分歧={[d['literal'] for d in rep['g1']['_verified_consistent']]}")
|
|
651
|
-
a(f" 文档滞后于代码={[d['where'] for d in rep['g1']['_doc_code_drift']]}")
|
|
652
|
-
a(f" 新增类型触点={rep['g1']['_new_type_touchpoints']}")
|
|
653
|
-
return "\n".join(L)
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
# 生效条件:传入对象可写属性时执行赋值 sustain_module.conformance_summary = report_summary,无返回值(None);
|
|
657
|
-
def register_sustain(sustain_module) -> None: # pragma: no cover
|
|
658
|
-
"""挂点说明(供 sustain 侧最小侵入调用)。
|
|
659
|
-
|
|
660
|
-
周期巡检**不加新 tick**:`_tick_tidy()` 已是「contextual 存量治理」的
|
|
661
|
-
读侧巡检,本断言集复用同一节奏(tidy_interval),由 `report_summary(cg)`
|
|
662
|
-
只取结论不落盘——避免为只读检查新增后台周期与 env 开关。
|
|
663
|
-
"""
|
|
664
|
-
sustain_module.conformance_summary = report_summary
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
# 生效条件:root 依次取 cg_or_root.root 属性、cg_or_root 本身、_default_root()(前者为假值即回落,故 0/空串会继续回落);check 抛异常时返回 ok=False、verdict="BLINDSPOT"、error,否则返回 ok=(verdict!="FAIL") 与 fail/warn/blindspot/failed_ids/t;
|
|
668
|
-
def report_summary(cg_or_root) -> dict:
|
|
669
|
-
"""给常驻循环用的**轻量**结论(不渲染全文、不写盘):verdict + 计数。"""
|
|
670
|
-
root = getattr(cg_or_root, "root", None) or cg_or_root or _default_root()
|
|
671
|
-
try:
|
|
672
|
-
rep = check(root, check_paths=False)
|
|
673
|
-
except Exception as e: # noqa: BLE001
|
|
674
|
-
return {"ok": False, "verdict": "BLINDSPOT", "error": f"{type(e).__name__}: {e}"}
|
|
675
|
-
return {"ok": rep["verdict"] != "FAIL", "verdict": rep["verdict"],
|
|
676
|
-
"fail": rep["counts"]["fail"], "warn": rep["counts"]["warn"],
|
|
677
|
-
"blindspot": rep["counts"]["blindspot"],
|
|
678
|
-
"failed_ids": [c["id"] for c in rep["checks"]
|
|
679
|
-
if c["level"] == "FAIL" and not c["ok"]],
|
|
680
|
-
"t": rep["t"]}
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
# 生效条件:argv 为 None 时 argparse 从 sys.argv 取值,--root 缺失或为假值回落 _default_root();check 抛 OSError/ValueError(如索引不可读)时打印 BLINDSPOT 并返回 2,否则按 --write-baseline/--json 写盘后返回 0(verdict≠"FAIL")或 1("FAIL");
|
|
684
|
-
def main(argv=None) -> int:
|
|
685
|
-
p = argparse.ArgumentParser(prog="python -m md_cg.conformance",
|
|
686
|
-
description="数据健康不变量断言集(只读,零写入)")
|
|
687
|
-
p.add_argument("--root", default=None,
|
|
688
|
-
help="认知图根(默认 MDCG_ROOT > paths.json > 用户级状态根 data/mdcg)")
|
|
689
|
-
p.add_argument("--json", dest="json_out", default=None)
|
|
690
|
-
p.add_argument("--baseline", default=None, help="基线报告 json(比对不劣化)")
|
|
691
|
-
p.add_argument("--write-baseline", default=None, help="把本次指标写成新基线")
|
|
692
|
-
p.add_argument("--no-path-check", action="store_true", help="不查盘上文件是否存在")
|
|
693
|
-
p.add_argument("--strict", action="store_true", help="BLINDSPOT 也视为非 PASS")
|
|
694
|
-
a = p.parse_args(argv)
|
|
695
|
-
|
|
696
|
-
root = a.root or _default_root()
|
|
697
|
-
baseline = None
|
|
698
|
-
if a.baseline:
|
|
699
|
-
with open(a.baseline, encoding="utf-8") as f:
|
|
700
|
-
baseline = json.load(f)
|
|
701
|
-
try:
|
|
702
|
-
rep = check(root, check_paths=not a.no_path_check,
|
|
703
|
-
baseline=baseline, strict=a.strict)
|
|
704
|
-
except (OSError, ValueError) as e:
|
|
705
|
-
# 数据源不可读 = BLINDSPOT(如实上报),绝不打印「通过」
|
|
706
|
-
print(f"== conformance v{REPORT_VERSION} · {os.path.abspath(root)}")
|
|
707
|
-
print(f" verdict: BLINDSPOT —— 索引不可读: {type(e).__name__}: {e}")
|
|
708
|
-
print(f" 缺失维度: 数据源({INDEX_FILE});修复后重跑,勿以本结果为「通过」。")
|
|
709
|
-
return 2
|
|
710
|
-
print(render(rep))
|
|
711
|
-
if a.write_baseline:
|
|
712
|
-
with open(a.write_baseline, "w", encoding="utf-8") as f:
|
|
713
|
-
json.dump({"report_version": REPORT_VERSION, "t": rep["t"],
|
|
714
|
-
"root": rep["root"], "metrics": metrics_of(rep)},
|
|
715
|
-
f, ensure_ascii=False, indent=2)
|
|
716
|
-
print(f"\n 基线已写入: {a.write_baseline}")
|
|
717
|
-
if a.json_out:
|
|
718
|
-
with open(a.json_out, "w", encoding="utf-8") as f:
|
|
719
|
-
json.dump(rep, f, ensure_ascii=False, indent=2, default=str)
|
|
720
|
-
print(f" 报告已写入: {a.json_out}")
|
|
721
|
-
return 0 if rep["verdict"] != "FAIL" else 1
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
if __name__ == "__main__":
|
|
725
|
-
if hasattr(sys.stdout, "reconfigure"):
|
|
726
|
-
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""md_cg · 数据健康不变量断言集(Pi⑤)+ G1 类型空间正交性审计 + G3 unanalyzed 显式占位
|
|
3
|
+
|
|
4
|
+
把 2026-09-15 的一次性审计改写为**周期可复跑**的断言集(混层比例阈值 / 重复度阈值 /
|
|
5
|
+
字段覆盖率下限 / 闸门四态分布),挂 sustain 周期巡检;超阈值**只告警,不自动改数据**。
|
|
6
|
+
|
|
7
|
+
三条纪律:
|
|
8
|
+
1. **不变量 fail-closed**:类型空间封闭性 / 边键规范 / 边目标可解析 / 索引↔盘一致 /
|
|
9
|
+
重复度不劣化——任一 FAIL = 库结构失真,须人工处置。
|
|
10
|
+
2. **健康指标只告警**:混层比 / role 覆盖率 / evidence 覆盖率 / 分析覆盖率 /
|
|
11
|
+
闸门四态 / 触达率——低于靶值 WARN,不阻断、不改数据。
|
|
12
|
+
3. **缺数据源 → BLINDSPOT**(如实上报),绝不冒充 PASS。
|
|
13
|
+
|
|
14
|
+
G1(Ghidra 交接 §5.4):输出各类型空间的「声明值 / 实测值 / 未声明已用 / 声明未用 /
|
|
15
|
+
跨空间重叠 / 方向口径分歧」——类型空间不封闭是写入侧漂移的前兆。
|
|
16
|
+
|
|
17
|
+
G3(同 §5.4):unanalyzed 采用**派生断言**(不新增字段、零写入),判据
|
|
18
|
+
`evidence_count==0 ∧ verification_basis 空 ∧ lifecycle_state 空`。
|
|
19
|
+
不做字段级占位的理由:生命周期是单向降级轴,与分析覆盖正交,塞同一字段破坏正交性。
|
|
20
|
+
|
|
21
|
+
用法:
|
|
22
|
+
python -m md_cg.conformance [--root R] [--json out.json] [--baseline b.json]
|
|
23
|
+
[--no-path-check] [--strict]
|
|
24
|
+
"""
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import argparse
|
|
28
|
+
import json
|
|
29
|
+
import os
|
|
30
|
+
import sys
|
|
31
|
+
import time
|
|
32
|
+
from collections import Counter, deque
|
|
33
|
+
|
|
34
|
+
REPORT_VERSION = 1
|
|
35
|
+
INDEX_FILE = "_index.json"
|
|
36
|
+
ACCESS_LOG = "_access.log"
|
|
37
|
+
DECISION_LOG = os.path.join("hippocampus", "decisions.jsonl")
|
|
38
|
+
INBOX_LOG = os.path.join("hippocampus", "inbox.jsonl")
|
|
39
|
+
DEFAULT_ROOT = os.path.join(
|
|
40
|
+
os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "data", "mdcg")
|
|
41
|
+
#: _audit.jsonl 只读尾部窗口行数(append-only 全史可数十 MB,周期巡检不扫全史)
|
|
42
|
+
AUDIT_TAIL = 20000
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
# 生效条件:无入参,datapath.mdcg_root() 返回真值时返回该值,导入或调用抛异常、或返回假值(如空串)时返回模块常量 DEFAULT_ROOT;
|
|
46
|
+
def _default_root() -> str:
|
|
47
|
+
"""根解析沿用本仓约定:env `MDCG_ROOT` > `paths.json`(用户级,旧包内兼容读)
|
|
48
|
+
> 用户级状态根 `data/mdcg`。
|
|
49
|
+
|
|
50
|
+
走 `datapath.mdcg_root()`(兜底纪律:与其它工具同源解析,不另立一套)。
|
|
51
|
+
"""
|
|
52
|
+
try:
|
|
53
|
+
from .datapath import mdcg_root
|
|
54
|
+
got = mdcg_root()
|
|
55
|
+
if got:
|
|
56
|
+
return got
|
|
57
|
+
except Exception: # noqa: BLE001
|
|
58
|
+
pass
|
|
59
|
+
return DEFAULT_ROOT
|
|
60
|
+
|
|
61
|
+
# 阈值口径 = 9-15 审计基线(2026-09-16 复跑实测:混层 6468/11076=58.4% / role 2.13% /
|
|
62
|
+
# verification_basis 60.3% / evidence 0.10% / dir_mismatch 0 / 闸门 142 条) + M4 治理靶值。
|
|
63
|
+
# 比率型阈值只表达「离靶多远」,不阻断任何写入。
|
|
64
|
+
# 注:触达率随 _access.log 滚动追加而单调增长,9-15 快照 7.0% → 09-16 实测 8.7%,
|
|
65
|
+
# 差异来自时间而非口径;故**不纳入基线不劣化比对**(日志轮转会误报)。
|
|
66
|
+
THRESHOLDS = {
|
|
67
|
+
"stratum_ratio_max": 0.60,
|
|
68
|
+
"stratum_target": 0.30,
|
|
69
|
+
"role_coverage_min": 0.90,
|
|
70
|
+
"evidence_coverage_min": 0.50,
|
|
71
|
+
"analysis_coverage_min": 0.90,
|
|
72
|
+
"reach_ratio_min": 0.30,
|
|
73
|
+
"gate_sample_min": 20,
|
|
74
|
+
# 样本下限:低于此值比率型指标不可判(如仓内自建语料),一律 BLINDSPOT 不冒充 PASS
|
|
75
|
+
"min_nodes": 200,
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
#: 单一内容指纹的最大同组节点数(超过即「同模板批量写入」体征,须人工治理)
|
|
79
|
+
DUP_GROUP_MAX = 200
|
|
80
|
+
|
|
81
|
+
# 写侧规范边键。取证更正(2026-09-16):subgraph.children_index:122 / parents_index:162
|
|
82
|
+
# 已含 `or e.get("type")` 容错,**读侧不忽略** legacy `type`(`subgraph` 内 normalize_edge
|
|
83
|
+
# 的 docstring 仍写「会被静默忽略」,属文档滞后于代码,已登记 DOC_CODE_DRIFT)。
|
|
84
|
+
# 仍记为规范项:chain.edge_rel 与 subgraph 双兼容是本库私约,非 md_cg 读侧(Rust 引擎 /
|
|
85
|
+
# 外部消费者)可能只读 relation_type。
|
|
86
|
+
CANONICAL_EDGE_KEYS = ("relation_type", "relation")
|
|
87
|
+
LAYER_DIRS = ("knowledge", "contextual", "self", "structural", "anchor",
|
|
88
|
+
"rejected", "unresolved", "goals", "data", "trash")
|
|
89
|
+
|
|
90
|
+
# G1 静态登记:已取证的方向口径分歧(事实,非推断)
|
|
91
|
+
KNOWN_DIRECTION_CONFLICTS = [
|
|
92
|
+
{"literal": "hierarchical",
|
|
93
|
+
"conflict": "md_cg subgraph 视其为「本节点是子、target 是父」"
|
|
94
|
+
"(children_index:128);白箱源库语义为 source 是父 —— 同名反义,"
|
|
95
|
+
"直接迁移整树倒置",
|
|
96
|
+
"workaround": "migrate_wisdom_graph.SRC_REL_MAP 迁移时改写为 contains(语义等价)"},
|
|
97
|
+
]
|
|
98
|
+
# G1 静态登记:已核对**无**方向分歧的项(防止「看起来像冲突」被反复误报)
|
|
99
|
+
VERIFIED_CONSISTENT = [
|
|
100
|
+
{"literal": "part_of / parent_of / contains",
|
|
101
|
+
"verified": "children_index:128-135 与 parents_index:171-178 逐字对称、方向取反,自洽"},
|
|
102
|
+
]
|
|
103
|
+
# G1 已取证的**文档滞后于代码**实例(陈述与实现不符,非缺陷但会误导写侧)
|
|
104
|
+
DOC_CODE_DRIFT = [
|
|
105
|
+
{"where": "subgraph.normalize_edge docstring:96",
|
|
106
|
+
"claims": "「早期迁移器写的 type 会被静默忽略(subgraph 不兼容)」",
|
|
107
|
+
"fact": "children_index:122 / parents_index:162 已含 `or e.get(\"type\")` 容错 —— "
|
|
108
|
+
"读侧兼容,陈述已过期"},
|
|
109
|
+
]
|
|
110
|
+
|
|
111
|
+
# G1 静态登记:新增类型的散落点(回答「新增类型要不要改引擎」)
|
|
112
|
+
NEW_TYPE_TOUCHPOINTS = {
|
|
113
|
+
"layer": ["mdcg.LAYERS", "mdcg.BUCKETED_LAYERS", "mdcg.NEG_MEMORY_MARKS",
|
|
114
|
+
"nodefile(fm 契约)", "routing(bucket)", "MCP 工具面描述"],
|
|
115
|
+
"edge_type": ["chain.EDGE_WEIGHTS", "subgraph(父子方向分支)",
|
|
116
|
+
"migrate_wisdom_graph.SRC_REL_MAP"],
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
# 生效条件:root 下 INDEX_FILE 可读且 JSON 顶层为 dict 并含 nodes 字典时返回该 nodes;不可读抛 OSError、JSON 非法抛 JSONDecodeError、nodes 非 dict 抛 ValueError;
|
|
121
|
+
def load_index(root: str) -> dict:
|
|
122
|
+
"""读索引快照(唯一必需数据源);损坏即抛——断言集不建立在猜测上。"""
|
|
123
|
+
with open(os.path.join(root, INDEX_FILE), encoding="utf-8") as f:
|
|
124
|
+
raw = json.load(f)
|
|
125
|
+
nodes = raw.get("nodes") if isinstance(raw, dict) else None
|
|
126
|
+
if not isinstance(nodes, dict):
|
|
127
|
+
raise ValueError(f"索引格式异常(缺 nodes 字典): {root}/{INDEX_FILE}")
|
|
128
|
+
return nodes
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
# 生效条件:path 不存在(os.path.exists 为假)时返回 [];存在时逐行解析,空行与 JSONDecodeError 行被跳过,返回可解析记录的列表(可能为 []);
|
|
132
|
+
def _read_jsonl(path: str) -> list:
|
|
133
|
+
if not os.path.exists(path):
|
|
134
|
+
return []
|
|
135
|
+
out = []
|
|
136
|
+
with open(path, encoding="utf-8", errors="replace") as f:
|
|
137
|
+
for line in f:
|
|
138
|
+
line = line.strip()
|
|
139
|
+
if not line:
|
|
140
|
+
continue
|
|
141
|
+
try:
|
|
142
|
+
out.append(json.loads(line))
|
|
143
|
+
except json.JSONDecodeError:
|
|
144
|
+
continue
|
|
145
|
+
return out
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
# 生效条件:rec 的 tags 缺失或为假值时按空列表处理,仅对含 ":" 的 tag 取首个冒号前的前缀构成集合并返回;无此类 tag 时返回空集合;
|
|
149
|
+
def _tag_prefixes(rec: dict) -> set:
|
|
150
|
+
return {str(t).split(":", 1)[0] for t in (rec.get("tags") or []) if ":" in str(t)}
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
# 生效条件:r 的 verification_basis 为 list/tuple 时返回其真值元素的 str 列表;否则该值真值时返回 [str(b)],缺失或为假值(None/空串/空容器)时返回 [];
|
|
154
|
+
def _basis_of(r: dict):
|
|
155
|
+
b = r.get("verification_basis")
|
|
156
|
+
if isinstance(b, (list, tuple)):
|
|
157
|
+
return [str(x) for x in b if x]
|
|
158
|
+
return [str(b)] if b else []
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
# 生效条件:无入参,返回 layer(取自导入的模块常量 LAYERS 的排序值)+ edge_type/derived_relation/verification_basis/lifecycle_state 五键声明值映射,后四键对应模块导入或属性取值失败、属性值为假值(空容器/None)时该键 declared=None;
|
|
162
|
+
def enum_spaces() -> dict:
|
|
163
|
+
"""各类型空间的声明值(模块真源取不到 → declared=None,报告标 BLINDSPOT)。"""
|
|
164
|
+
from .mdcg import LAYERS
|
|
165
|
+
spaces = {"layer": {"declared": sorted(LAYERS), "source": "mdcg.LAYERS"}}
|
|
166
|
+
|
|
167
|
+
# 生效条件:以 md_cg.<mod> 导入并 getattr(attr)(导入或取值抛异常则 val=None),把 {"declared": sorted(val) if val else None, "source": "<mod>.<attr>"} 写入片段外闭包变量 spaces 的 key 键(val 为假值时 declared=None),无返回值;
|
|
168
|
+
def _grab(mod: str, attr: str, key: str):
|
|
169
|
+
try:
|
|
170
|
+
m = __import__(f"md_cg.{mod}", fromlist=[attr])
|
|
171
|
+
val = getattr(m, attr, None)
|
|
172
|
+
except Exception: # noqa: BLE001
|
|
173
|
+
val = None
|
|
174
|
+
spaces[key] = {"declared": sorted(val) if val else None,
|
|
175
|
+
"source": f"{mod}.{attr}"}
|
|
176
|
+
|
|
177
|
+
_grab("chain", "EDGE_WEIGHTS", "edge_type")
|
|
178
|
+
_grab("provenance", "RELATIONS", "derived_relation")
|
|
179
|
+
_grab("nodefile", "VERIFICATION_BASIS", "verification_basis")
|
|
180
|
+
_grab("lifecycle", "STATES", "lifecycle_state")
|
|
181
|
+
return spaces
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
# ---------------------------- 指标 ----------------------------
|
|
185
|
+
|
|
186
|
+
# 生效条件:对 nodes 各记录按 str(content_hash) 计数,content_hash 缺失被折成 "None",返回出现次数 >1 且哈希不为 "None"/"" 的组数、涉及节点数与前 5 大组大小;
|
|
187
|
+
def _dup_metrics(nodes: dict) -> dict:
|
|
188
|
+
ch = Counter(str(r.get("content_hash")) for r in nodes.values())
|
|
189
|
+
groups = {h: c for h, c in ch.items() if c > 1 and h not in ("None", "")}
|
|
190
|
+
return {"groups": len(groups), "nodes": sum(groups.values()),
|
|
191
|
+
"largest": sorted(groups.values(), reverse=True)[:5]}
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
# 生效条件:对 nodes 中每个节点的 "edges" 真值且为 list/tuple 的边列表,仅遍历其中 dict 边;依据 CANONICAL_EDGE_KEYS 与 "type" 统计键组合、关系类型、总边数、非规范键样本(最多 5)和悬空边,返回汇总字典。
|
|
195
|
+
def _edge_metrics(nodes: dict) -> dict:
|
|
196
|
+
types, key_mix = Counter(), Counter()
|
|
197
|
+
total = dangling = with_edges = non_canonical = 0
|
|
198
|
+
samples = []
|
|
199
|
+
for nid, r in nodes.items():
|
|
200
|
+
es = r.get("edges") or []
|
|
201
|
+
if not isinstance(es, (list, tuple)) or not es:
|
|
202
|
+
continue
|
|
203
|
+
with_edges += 1
|
|
204
|
+
for e in es:
|
|
205
|
+
if not isinstance(e, dict):
|
|
206
|
+
continue
|
|
207
|
+
total += 1
|
|
208
|
+
keys = [k for k in CANONICAL_EDGE_KEYS + ("type",) if e.get(k)]
|
|
209
|
+
types[str(e.get("relation_type") or e.get("relation") or e.get("type"))] += 1
|
|
210
|
+
key_mix["+".join(keys) if keys else "<无关系键>"] += 1
|
|
211
|
+
if not any(k in CANONICAL_EDGE_KEYS for k in keys):
|
|
212
|
+
non_canonical += 1
|
|
213
|
+
if len(samples) < 5:
|
|
214
|
+
samples.append({"node": nid, "edge": dict(e)})
|
|
215
|
+
tgt = e.get("target") or e.get("target_id")
|
|
216
|
+
if tgt is None or str(tgt) not in nodes:
|
|
217
|
+
dangling += 1
|
|
218
|
+
return {"nodes_with_edges": with_edges, "total": total,
|
|
219
|
+
"types": dict(types.most_common(20)), "key_mix": dict(key_mix.most_common(10)),
|
|
220
|
+
"non_canonical": non_canonical, "dangling": dangling, "samples": samples}
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
# 生效条件:遍历 nodes 的值作为记录,path 经 str(r.get("path") or "") 为空即 missing++ 并 continue,不以记录含 layer/path 为前置;仅对非空 path 执行 layer 与 head 检查(layer 非空且 head != layer 且 head in LAYER_DIRS 时 layer_mismatch++),且 check_exists 为 True 时才对非空 path 以 os.path.join(root, rel) 判断不存在并计入 missing,最终返回 missing/layer_mismatch/samples 统计;。
|
|
224
|
+
def _path_metrics(root: str, nodes: dict, check_exists: bool) -> dict:
|
|
225
|
+
missing = mismatch = 0
|
|
226
|
+
samples = []
|
|
227
|
+
for r in nodes.values():
|
|
228
|
+
layer, p = str(r.get("layer") or ""), str(r.get("path") or "")
|
|
229
|
+
if not p:
|
|
230
|
+
missing += 1
|
|
231
|
+
if len(samples) < 5:
|
|
232
|
+
samples.append({"id": r.get("id"), "why": "path 缺失"})
|
|
233
|
+
continue
|
|
234
|
+
rel = p.replace("\\", "/")
|
|
235
|
+
head = rel.split("/", 1)[0]
|
|
236
|
+
if layer and head != layer and head in LAYER_DIRS:
|
|
237
|
+
mismatch += 1
|
|
238
|
+
if len(samples) < 5:
|
|
239
|
+
samples.append({"id": r.get("id"), "path": p, "layer": layer})
|
|
240
|
+
if check_exists and not os.path.exists(os.path.join(root, rel)):
|
|
241
|
+
missing += 1
|
|
242
|
+
if len(samples) < 5:
|
|
243
|
+
samples.append({"id": r.get("id"), "path": p, "why": "文件不存在"})
|
|
244
|
+
return {"missing": missing, "layer_mismatch": mismatch, "samples": samples}
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
# 生效条件:root 下 DECISION_LOG/INBOX_LOG 不存在时按空列表计,返回 decisions(decision 或 status 回落 "?" 的前 10 项)、decisions_total、inbox_pending=max(0, 收件数−决策数)、以及 _audit.jsonl 末尾 AUDIT_TAIL 行内 op 计数的前 6 项与尾窗元信息;
|
|
248
|
+
def _gate_metrics(root: str) -> dict:
|
|
249
|
+
dec = _read_jsonl(os.path.join(root, DECISION_LOG))
|
|
250
|
+
inbox = _read_jsonl(os.path.join(root, INBOX_LOG))
|
|
251
|
+
ops = Counter()
|
|
252
|
+
# _audit.jsonl 是 append-only 且可达数十 MB(真源 15.9MB / 7.4 万行)——
|
|
253
|
+
# 只读尾部窗口(近 AUDIT_TAIL 行)取 op 分布,避免周期巡检在此处 O(全史)。
|
|
254
|
+
audit = os.path.join(root, "_audit.jsonl")
|
|
255
|
+
tail = []
|
|
256
|
+
if os.path.exists(audit):
|
|
257
|
+
with open(audit, encoding="utf-8", errors="replace") as f:
|
|
258
|
+
tail = list(deque(f, maxlen=AUDIT_TAIL))
|
|
259
|
+
for line in tail:
|
|
260
|
+
if not line.strip():
|
|
261
|
+
continue
|
|
262
|
+
try:
|
|
263
|
+
ops[str(json.loads(line).get("op"))] += 1
|
|
264
|
+
except json.JSONDecodeError:
|
|
265
|
+
continue
|
|
266
|
+
return {"decisions": dict(Counter(str(r.get("decision") or r.get("status") or "?")
|
|
267
|
+
for r in dec).most_common(10)),
|
|
268
|
+
"decisions_total": len(dec),
|
|
269
|
+
"inbox_pending": max(0, len(inbox) - len(dec)),
|
|
270
|
+
"audit_ops": dict(ops.most_common(6)), "audit_tail_lines": len(tail),
|
|
271
|
+
"audit_tail_window": AUDIT_TAIL}
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
# 生效条件:对 root 与 nodes,若 root/ACCESS_LOG 路径不存在则返回 {"distinct": None, "ratio": None, "lines": None};否则逐行解析 JSON,当 "ids" 为 list 或 tuple 时把其中真值元素 str 后加入 seen,且当 "id"/"node_id"/"nid" 任一真值时把其 str 加入 seen,返回 distinct 为 seen 与 nodes 键交集数量、ratio 为 distinct/max(len(nodes),1)、lines 为非空行数。
|
|
275
|
+
def _reach_metrics(root: str, nodes: dict) -> dict:
|
|
276
|
+
p = os.path.join(root, ACCESS_LOG)
|
|
277
|
+
if not os.path.exists(p):
|
|
278
|
+
return {"distinct": None, "ratio": None, "lines": None}
|
|
279
|
+
seen, lines = set(), 0
|
|
280
|
+
with open(p, encoding="utf-8", errors="replace") as f:
|
|
281
|
+
for line in f:
|
|
282
|
+
line = line.strip()
|
|
283
|
+
if not line:
|
|
284
|
+
continue
|
|
285
|
+
lines += 1
|
|
286
|
+
try:
|
|
287
|
+
rec = json.loads(line)
|
|
288
|
+
except json.JSONDecodeError:
|
|
289
|
+
continue
|
|
290
|
+
# 实际落盘形态:{"t":..., "ids":[<nid>,...], "tier":...}(_access.log 逐次追加)
|
|
291
|
+
got = rec.get("ids")
|
|
292
|
+
if isinstance(got, (list, tuple)):
|
|
293
|
+
seen.update(str(x) for x in got if x)
|
|
294
|
+
nid = rec.get("id") or rec.get("node_id") or rec.get("nid")
|
|
295
|
+
if nid:
|
|
296
|
+
seen.add(str(nid))
|
|
297
|
+
hit = len(seen & set(nodes))
|
|
298
|
+
return {"distinct": hit, "ratio": hit / max(len(nodes), 1), "lines": lines}
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
# 生效条件:对 nodes,筛出 layer 字符串为 "knowledge" 的 kn;mixed 为 kn 中标签前缀含 doc 或 code 的数量;返回全库 nodes 数与 knowledge 层数、mixed、mixed_ratio=mixed/max(len(kn),1),以及全库口径 role_ratio/basis_ratio/evidence_ratio/state_ratio 和 knowledge 层口径 role_ratio_kn/basis_ratio_kn/evidence_ratio_kn。
|
|
302
|
+
def _coverage_metrics(nodes: dict) -> dict:
|
|
303
|
+
"""混层口径 + 字段覆盖率(9-15 审计基线口径,逐项可复现)。
|
|
304
|
+
|
|
305
|
+
**分母口径取证(2026-09-16,关键)**:审计的「role 99.99% 为空」「evidence_count
|
|
306
|
+
仅 6 条」是 **knowledge 层**分母;全库分母会摊薄成「role 2.13% / evidence 12 条」,
|
|
307
|
+
把一个 99.99% 的空缺伪装成「还行」。故两类分母并存上报,**健康判据一律取
|
|
308
|
+
knowledge 层口径**(`*_ratio_kn`),全库口径只作对照片段。
|
|
309
|
+
"""
|
|
310
|
+
kn = [r for r in nodes.values() if str(r.get("layer")) == "knowledge"]
|
|
311
|
+
mixed = sum(1 for r in kn if {"doc", "code"} & _tag_prefixes(r))
|
|
312
|
+
n, nk = len(nodes), max(len(kn), 1)
|
|
313
|
+
return {"nodes": n, "knowledge": len(kn), "mixed_layer": mixed,
|
|
314
|
+
"mixed_ratio": mixed / nk,
|
|
315
|
+
"role_ratio": sum(1 for r in nodes.values() if r.get("role")) / max(n, 1),
|
|
316
|
+
"basis_ratio": sum(1 for r in nodes.values() if _basis_of(r)) / max(n, 1),
|
|
317
|
+
"evidence_ratio": sum(1 for r in nodes.values()
|
|
318
|
+
if _as_int(r.get("evidence_count")) > 0) / max(n, 1),
|
|
319
|
+
"state_ratio": sum(1 for r in nodes.values()
|
|
320
|
+
if r.get("lifecycle_state")) / max(n, 1),
|
|
321
|
+
# —— knowledge 层口径(= 审计原口径,判据用这组)——
|
|
322
|
+
"role_ratio_kn": sum(1 for r in kn if r.get("role")) / nk,
|
|
323
|
+
"basis_ratio_kn": sum(1 for r in kn if _basis_of(r)) / nk,
|
|
324
|
+
"evidence_ratio_kn": sum(1 for r in kn
|
|
325
|
+
if _as_int(r.get("evidence_count")) > 0) / nk}
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
# 生效条件:int(v) 转换成功时返回该整数;抛 TypeError/ValueError(如 None、非数字串)时返回 0;
|
|
329
|
+
def _as_int(v) -> int:
|
|
330
|
+
try:
|
|
331
|
+
return int(v)
|
|
332
|
+
except (TypeError, ValueError):
|
|
333
|
+
return 0
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
# ---------------------------- G3 · unanalyzed 派生集合 ----------------------------
|
|
337
|
+
|
|
338
|
+
# 生效条件:仅当记录 _as_int(evidence_count)==0 且 verification_basis 为空且 lifecycle_state 为假值时计入 ids,返回 count、ratio=count/max(len(nodes),1) 与前 5 个 id 样本;
|
|
339
|
+
def unanalyzed(nodes: dict) -> dict:
|
|
340
|
+
"""G3:unanalyzed **派生**判据(不新增字段、零写入)。
|
|
341
|
+
|
|
342
|
+
判据 = `evidence_count==0 ∧ verification_basis 空 ∧ lifecycle_state 空`
|
|
343
|
+
—— 三者皆空 = 「无一维分析痕迹」,与「分析结论为负」不同(后者会留 basis)。
|
|
344
|
+
|
|
345
|
+
不做字段级占位的理由(Ghidra 交接 §5.4):
|
|
346
|
+
* 生命周期是**单向降级轴**(active→converged→demoted→archived),与分析覆盖正交,
|
|
347
|
+
塞进同一字段即破坏正交性;
|
|
348
|
+
* **逃逸口已存在**——`verification_basis` 声明集里有 `other`(真源 273 条在用),
|
|
349
|
+
「分析过但无法归类」有明确归宿,故「basis 为空」不再与「分析结论为空」混淆。
|
|
350
|
+
"""
|
|
351
|
+
ids = [nid for nid, r in nodes.items()
|
|
352
|
+
if _as_int(r.get("evidence_count")) == 0
|
|
353
|
+
and not _basis_of(r)
|
|
354
|
+
and not r.get("lifecycle_state")]
|
|
355
|
+
return {"count": len(ids), "ratio": len(ids) / max(len(nodes), 1),
|
|
356
|
+
"sample": ids[:5]}
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
# ---------------------------- G1 · 类型空间正交性审计 ----------------------------
|
|
360
|
+
|
|
361
|
+
# 生效条件:调用 _usage(nodes, edges) 时 edges 形参未被源码引用,对 nodes.values() 中每个节点 r,仅当 r.get("layer")、r.get("role")、r.get("derived_relation")、r.get("lifecycle_state") 为真值时分别以 str 值计入对应 Counter,_basis_of(r) 与 _tag_prefixes(r) 展开的元素分别计入 verification_basis 与 tag_prefix,再遍历各 r 的 r.get("edges") or [] 中 isinstance(e, dict) 的项按 edge_rel(e) 计数,最终 edge_type 仅保留键为真值的计数,返回 used 字典;
|
|
362
|
+
def _usage(nodes: dict, edges: dict) -> dict:
|
|
363
|
+
"""各类型空间的**实测**取值(与 enum_spaces 的声明值对照)。"""
|
|
364
|
+
from .chain import edge_rel
|
|
365
|
+
used = {
|
|
366
|
+
"layer": Counter(str(r.get("layer")) for r in nodes.values() if r.get("layer")),
|
|
367
|
+
"role": Counter(str(r.get("role")) for r in nodes.values() if r.get("role")),
|
|
368
|
+
"verification_basis": Counter(b for r in nodes.values() for b in _basis_of(r)),
|
|
369
|
+
"derived_relation": Counter(str(r.get("derived_relation"))
|
|
370
|
+
for r in nodes.values() if r.get("derived_relation")),
|
|
371
|
+
"lifecycle_state": Counter(str(r.get("lifecycle_state"))
|
|
372
|
+
for r in nodes.values() if r.get("lifecycle_state")),
|
|
373
|
+
"tag_prefix": Counter(p for r in nodes.values() for p in _tag_prefixes(r)),
|
|
374
|
+
}
|
|
375
|
+
rels = Counter()
|
|
376
|
+
for r in nodes.values():
|
|
377
|
+
for e in (r.get("edges") or []):
|
|
378
|
+
if isinstance(e, dict):
|
|
379
|
+
rels[edge_rel(e)] += 1
|
|
380
|
+
used["edge_type"] = Counter({k: v for k, v in rels.items() if k})
|
|
381
|
+
return used
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
# 生效条件:以 enum_spaces 的声明集与 _usage 的实测集逐空间对照(closed 仅在 declared 非 None 且无未声明已用时为 True),role 与 tag_prefix 两空间的 declared/closed 恒为 None,另附 _cross_space_overlap(字面量出现在 ≥2 空间或 layer∩tag_prefix)及 KNOWN_DIRECTION_CONFLICTS/VERIFIED_CONSISTENT/DOC_CODE_DRIFT/NEW_TYPE_TOUCHPOINTS 四个常量;
|
|
385
|
+
def _g1_audit(nodes: dict, edges: dict) -> dict:
|
|
386
|
+
"""G1:声明值 / 实测值 / 未声明已用 / 声明未用 / 跨空间重叠 / 方向口径分歧。"""
|
|
387
|
+
spaces = enum_spaces()
|
|
388
|
+
used = _usage(nodes, edges)
|
|
389
|
+
out = {}
|
|
390
|
+
for name, spec in spaces.items():
|
|
391
|
+
dec = set(spec.get("declared") or [])
|
|
392
|
+
u = set(used.get(name, {}).keys())
|
|
393
|
+
out[name] = {
|
|
394
|
+
"source": spec.get("source"),
|
|
395
|
+
"declared": sorted(dec),
|
|
396
|
+
"used": dict(used.get(name, {}).most_common(30)),
|
|
397
|
+
"undeclared_used": sorted(u - dec),
|
|
398
|
+
"declared_unused": sorted(dec - u),
|
|
399
|
+
"closed": (spec.get("declared") is not None and not (u - dec)),
|
|
400
|
+
}
|
|
401
|
+
for name in ("role", "tag_prefix"):
|
|
402
|
+
out[name] = {"source": None, "declared": None,
|
|
403
|
+
"used": dict(used.get(name, {}).most_common(30)),
|
|
404
|
+
"undeclared_used": sorted(used.get(name, {})),
|
|
405
|
+
"declared_unused": [], "closed": None}
|
|
406
|
+
# 跨空间重叠:同一字面量出现在 ≥2 个空间 → 解析歧义风险
|
|
407
|
+
owners = Counter()
|
|
408
|
+
for name, spec in out.items():
|
|
409
|
+
for lit in (spec.get("declared") or []):
|
|
410
|
+
owners[str(lit)] += 1
|
|
411
|
+
out["_cross_space_overlap"] = sorted(l for l, c in owners.items() if c > 1) + \
|
|
412
|
+
sorted(set(used.get("layer", {})) & set(used.get("tag_prefix", {})))
|
|
413
|
+
out["_direction_conflicts"] = KNOWN_DIRECTION_CONFLICTS
|
|
414
|
+
out["_verified_consistent"] = VERIFIED_CONSISTENT
|
|
415
|
+
out["_doc_code_drift"] = DOC_CODE_DRIFT
|
|
416
|
+
out["_new_type_touchpoints"] = NEW_TYPE_TOUCHPOINTS
|
|
417
|
+
return out
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
# ---------------------------- 断言集 ----------------------------
|
|
421
|
+
|
|
422
|
+
# 生效条件:四参齐备时返回 {"id": cid, "level": level, "ok": bool(ok), "detail": detail},ok 经 bool() 归一为布尔;
|
|
423
|
+
def _ck(cid: str, level: str, ok: bool, detail: str) -> dict:
|
|
424
|
+
return {"id": cid, "level": level, "ok": bool(ok), "detail": detail}
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
# 生效条件:调用 check(root, check_paths, baseline, strict) 时,root 原样传入 load_index 得 nodes,check_paths 原样传入 _path_metrics 并在 index.path.present 检查消息中当其为假时追加“(--no-path-check 未查盘)”,baseline 为真时追加 _regression 基线不劣化检查、为假(None 或空 dict)时跳过,n=len(nodes) 小于 THRESHOLDS["min_nodes"] 时 small 为真且使 _ratio 各项及 reach.ratio、stratum.mixed_ratio 在 val 为 None 或 small 为真时记 BLINDSPOT、否则记 WARN,最终 verdict 为 FAIL(存在 level="FAIL" 且 ok 为假)、否则 WARN(存在 level="WARN" 且 ok 为假,或 strict 为真且存在 level="BLINDSPOT")、否则 PASS,返回含 REPORT_VERSION、t、elapsed_s、root 绝对路径、verdict、nodes、checks、counts、coverage、dup、edges、paths、gate、reach、unanalyzed、g1、protocol(=protocol.audit() 静态对账,恒追加四条 FAIL 级断言:op.declared 无未登记分支 / verb.implemented live 动词均有实现分支 / verb.reserved_clean 预留动词未被静默实现 / shape.declared 形状声明完备)、thresholds 的 dict;
|
|
428
|
+
def check(root: str, *, check_paths: bool = True,
|
|
429
|
+
baseline: dict = None, strict: bool = False) -> dict:
|
|
430
|
+
"""跑一遍全部断言,返回报告 dict(零写入:不修任何数据、不落任何文件)。"""
|
|
431
|
+
t0 = time.time()
|
|
432
|
+
nodes = load_index(root)
|
|
433
|
+
n = len(nodes)
|
|
434
|
+
small = n < THRESHOLDS["min_nodes"]
|
|
435
|
+
cov = _coverage_metrics(nodes)
|
|
436
|
+
edges = _edge_metrics(nodes)
|
|
437
|
+
paths = _path_metrics(root, nodes, check_paths)
|
|
438
|
+
gate = _gate_metrics(root)
|
|
439
|
+
reach = _reach_metrics(root, nodes)
|
|
440
|
+
un = unanalyzed(nodes)
|
|
441
|
+
g1 = _g1_audit(nodes, edges)
|
|
442
|
+
dup = _dup_metrics(nodes)
|
|
443
|
+
checks = []
|
|
444
|
+
T = THRESHOLDS
|
|
445
|
+
|
|
446
|
+
# ---- 不变量(fail-closed)----
|
|
447
|
+
for space, use_key in (("layer", "layer"), ("verification_basis", "verification_basis"),
|
|
448
|
+
("derived_relation", "derived_relation"),
|
|
449
|
+
("lifecycle_state", "lifecycle_state")):
|
|
450
|
+
spec = g1[space]
|
|
451
|
+
undecl = spec["undeclared_used"]
|
|
452
|
+
checks.append(_ck(f"enum.{space}.closed", "FAIL", not undecl,
|
|
453
|
+
f"声明源={spec['source']};未声明已用={undecl or '无'}"))
|
|
454
|
+
checks.append(_ck("edge.rel.declared", "WARN",
|
|
455
|
+
not g1["edge_type"]["undeclared_used"],
|
|
456
|
+
f"未声明边类型={g1['edge_type']['undeclared_used'] or '无'}"
|
|
457
|
+
"(读侧退化为 chain.DEFAULT_EDGE_WEIGHT=0.50,不炸但权重失真)"))
|
|
458
|
+
checks.append(_ck("edge.key.canonical", "WARN", edges["non_canonical"] == 0,
|
|
459
|
+
f"非规范键边 {edges['non_canonical']}/{edges['total']}"
|
|
460
|
+
f"(写侧规范={CANONICAL_EDGE_KEYS};md_cg 读侧兼容 type,"
|
|
461
|
+
"非 md_cg 读侧不兼容——存量债,判据=不劣化)"))
|
|
462
|
+
checks.append(_ck("edge.target.resolvable", "FAIL", edges["dangling"] == 0,
|
|
463
|
+
f"悬空边 {edges['dangling']}/{edges['total']}(图断裂)"))
|
|
464
|
+
checks.append(_ck("index.path.present", "FAIL", paths["missing"] == 0,
|
|
465
|
+
f"path 缺失/文件不存在 {paths['missing']}"
|
|
466
|
+
+ ("" if check_paths else "(--no-path-check 未查盘)")))
|
|
467
|
+
checks.append(_ck("index.layer.dir.match", "FAIL", paths["layer_mismatch"] == 0,
|
|
468
|
+
f"层-目录不一致 {paths['layer_mismatch']}(9-15 基线=0)"))
|
|
469
|
+
top_dup = dup["largest"][0] if dup["largest"] else 0
|
|
470
|
+
checks.append(_ck("dup.content.no_blowup", "FAIL", top_dup <= DUP_GROUP_MAX,
|
|
471
|
+
f"内容指纹重复组 {dup['groups']} / 涉及节点 {dup['nodes']}"
|
|
472
|
+
f" / 最大组 {dup['largest'][:3]}(最大组上限 {DUP_GROUP_MAX})"))
|
|
473
|
+
|
|
474
|
+
# ---- 健康指标(只告警)----
|
|
475
|
+
# 生效条件:val 为 None 或片段外闭包布尔 small 为真(small、n、T 均未在本片段定义)时追加 level="BLINDSPOT" 项;否则追加 level="WARN"、ok=(val>=low) 的项;两分支均无返回值;
|
|
476
|
+
def _ratio(cid, val, low, name):
|
|
477
|
+
if small or val is None:
|
|
478
|
+
checks.append(_ck(cid, "BLINDSPOT", True,
|
|
479
|
+
f"{name} 不可判(样本 {n} < {T['min_nodes']} / 数据源缺失)"))
|
|
480
|
+
return
|
|
481
|
+
checks.append(_ck(cid, "WARN", val >= low, f"{name}={val * 100:.1f}%(下限 {low:.0%})"))
|
|
482
|
+
|
|
483
|
+
mr = None if small else cov["mixed_ratio"]
|
|
484
|
+
checks.append(_ck("stratum.mixed_ratio",
|
|
485
|
+
"BLINDSPOT" if mr is None else "WARN",
|
|
486
|
+
mr is None or mr <= T["stratum_ratio_max"],
|
|
487
|
+
"混层比(knowledge 中 doc∪code 标签)=N/A(样本不足)" if mr is None else
|
|
488
|
+
f"混层比(knowledge 中 doc∪code 标签)={mr * 100:.1f}%"
|
|
489
|
+
f"(上限 {T['stratum_ratio_max']:.0%},治理靶 {T['stratum_target']:.0%})"))
|
|
490
|
+
_ratio("stratum.role_coverage", cov["role_ratio_kn"], T["role_coverage_min"],
|
|
491
|
+
"role 覆盖率[knowledge]")
|
|
492
|
+
_ratio("stratum.evidence_coverage", cov["evidence_ratio_kn"],
|
|
493
|
+
T["evidence_coverage_min"], "evidence 覆盖率[knowledge]")
|
|
494
|
+
_ratio("analysis.coverage", 1 - un["ratio"], T["analysis_coverage_min"],
|
|
495
|
+
f"分析覆盖率(非 unanalyzed;unanalyzed={un['count']})")
|
|
496
|
+
_ratio("reach.ratio", None if small else reach["ratio"], T["reach_ratio_min"], "触达率")
|
|
497
|
+
checks.append(_ck("gate.sample", "WARN",
|
|
498
|
+
gate["decisions_total"] >= T["gate_sample_min"],
|
|
499
|
+
f"闸门裁决样本 {gate['decisions_total']} 条 {gate['decisions']}"
|
|
500
|
+
f"(下限 {T['gate_sample_min']})"))
|
|
501
|
+
|
|
502
|
+
# ---- 记忆动词协议 v1 静态对账(声明 ↔ MCP 面实现,纯源码事实、不连库)----
|
|
503
|
+
# 与 G1 的分工:G1 对账**数据值域**,这里对账**动词面**——两者都是
|
|
504
|
+
# 「声明了没做 / 做了没说」的前置红灯,属结构性不变量(FAIL 级,fail-closed)。
|
|
505
|
+
from . import protocol as _proto
|
|
506
|
+
pa = _proto.audit()
|
|
507
|
+
checks.append(_ck("protocol.op.extension_surface", "WARN", True,
|
|
508
|
+
f"协议 v{pa['protocol_version']} 冻结面={pa['declared']}"
|
|
509
|
+
f"(live {len(pa['live'])} / reserved {len(pa['reserved'])});"
|
|
510
|
+
f"MCP 面另有 {len(pa['extension_ops'])} 个扩展 op 不属协议面"
|
|
511
|
+
f"(文档一致性由 cogmap 门禁承担)"))
|
|
512
|
+
checks.append(_ck("protocol.verb.implemented", "FAIL",
|
|
513
|
+
not pa["missing_impl"],
|
|
514
|
+
f"声明为 live 的动词 {pa['live']} 均有实现分支"
|
|
515
|
+
if not pa["missing_impl"] else
|
|
516
|
+
f"声明为 live 却无实现分支:{pa['missing_impl']}"))
|
|
517
|
+
checks.append(_ck("protocol.verb.reserved_clean", "FAIL",
|
|
518
|
+
not pa["reserved_leaked"],
|
|
519
|
+
f"reserved 动词 {pa['reserved']} 未出现实现分支"
|
|
520
|
+
if not pa["reserved_leaked"] else
|
|
521
|
+
f"reserved 动词被静默实现(须先改 status=live):"
|
|
522
|
+
f"{pa['reserved_leaked']}"))
|
|
523
|
+
checks.append(_ck("protocol.shape.declared", "FAIL",
|
|
524
|
+
not pa["shape_errors"],
|
|
525
|
+
f"{len(pa['declared'])} 个动词形状声明完备"
|
|
526
|
+
if not pa["shape_errors"] else
|
|
527
|
+
f"形状声明不完整:{pa['shape_errors']}"))
|
|
528
|
+
|
|
529
|
+
# ---- 基线不劣化(有 baseline 时才可判)----
|
|
530
|
+
if baseline:
|
|
531
|
+
checks += _regression({"mixed_ratio": cov["mixed_ratio"],
|
|
532
|
+
"non_canonical": edges["non_canonical"],
|
|
533
|
+
"dangling": edges["dangling"],
|
|
534
|
+
"dup_groups": dup["groups"],
|
|
535
|
+
"unanalyzed": un["count"]}, baseline)
|
|
536
|
+
|
|
537
|
+
fails = [c for c in checks if c["level"] == "FAIL" and not c["ok"]]
|
|
538
|
+
warns = [c for c in checks if c["level"] == "WARN" and not c["ok"]]
|
|
539
|
+
blind = [c for c in checks if c["level"] == "BLINDSPOT"]
|
|
540
|
+
v = "FAIL" if fails else ("WARN" if (warns or (strict and blind)) else "PASS")
|
|
541
|
+
return {"report_version": REPORT_VERSION, "t": t0, "elapsed_s": round(time.time() - t0, 2),
|
|
542
|
+
"root": os.path.abspath(root), "verdict": v,
|
|
543
|
+
"nodes": n, "checks": checks,
|
|
544
|
+
"counts": {"fail": len(fails), "warn": len(warns), "blindspot": len(blind),
|
|
545
|
+
"total": len(checks)},
|
|
546
|
+
"coverage": cov, "dup": dup, "edges": edges, "paths": paths,
|
|
547
|
+
"gate": gate, "reach": reach, "unanalyzed": un, "g1": g1,
|
|
548
|
+
"protocol": pa, "thresholds": T}
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
# 生效条件:对 cur 各键,baseline(或其 metrics)缺失、该键在基线为 None 或 cur 值为 None 时跳过;否则以 val > b+1e-9 判劣化,产出 level 恒为 "FAIL"、ok=not worse 的检查项;
|
|
552
|
+
def _regression(cur: dict, baseline: dict) -> list:
|
|
553
|
+
"""与基线比对:单调量只准不变或改善(重复度/悬空/非规范键/混层/unanalyzed)。
|
|
554
|
+
|
|
555
|
+
只收**单调有害量**(越大越坏);触达率等随运行时间自然增长的量不进比对,
|
|
556
|
+
否则日志轮转即误报。
|
|
557
|
+
"""
|
|
558
|
+
base = (baseline or {}).get("metrics") or {}
|
|
559
|
+
out = []
|
|
560
|
+
for key, val in cur.items():
|
|
561
|
+
b = base.get(key)
|
|
562
|
+
if b is None or val is None:
|
|
563
|
+
continue
|
|
564
|
+
worse = val > b + 1e-9
|
|
565
|
+
out.append(_ck(f"baseline.{key}", "FAIL", not worse,
|
|
566
|
+
f"现值 {val:.4f} vs 基线 {b:.4f}"
|
|
567
|
+
+ ("(劣化)" if worse else "(未劣化)")))
|
|
568
|
+
return out
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
# 生效条件:rep 含 coverage/edges/dup/unanalyzed/gate/reach 子字典与 nodes 键时(全部按下标取值,缺键即抛 KeyError)返回扁平指标映射;
|
|
572
|
+
def metrics_of(rep: dict) -> dict:
|
|
573
|
+
"""抽成可比对的扁平指标(--baseline 的写入面)。"""
|
|
574
|
+
return {"mixed_ratio": rep["coverage"]["mixed_ratio"],
|
|
575
|
+
"non_canonical": rep["edges"]["non_canonical"],
|
|
576
|
+
"dangling": rep["edges"]["dangling"],
|
|
577
|
+
"dup_groups": rep["dup"]["groups"],
|
|
578
|
+
"unanalyzed": rep["unanalyzed"]["count"],
|
|
579
|
+
"nodes": rep["nodes"],
|
|
580
|
+
"role_ratio": rep["coverage"]["role_ratio"],
|
|
581
|
+
"basis_ratio": rep["coverage"]["basis_ratio"],
|
|
582
|
+
"evidence_ratio": rep["coverage"]["evidence_ratio"],
|
|
583
|
+
"role_ratio_kn": rep["coverage"]["role_ratio_kn"],
|
|
584
|
+
"evidence_ratio_kn": rep["coverage"]["evidence_ratio_kn"],
|
|
585
|
+
"reach_ratio": rep["reach"]["ratio"],
|
|
586
|
+
"gate_decisions": rep["gate"]["decisions_total"]}
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
# ---------------------------- 报告 ----------------------------
|
|
590
|
+
|
|
591
|
+
# 生效条件:rep 各键齐备(按下标取值,任一键缺失抛 KeyError)时拼接为多行文本,reach.ratio 为 None 时渲染 "N/A",g1 中以 "_" 开头的键在遍历中被跳过;
|
|
592
|
+
def render(rep: dict) -> str:
|
|
593
|
+
L = []
|
|
594
|
+
a = L.append
|
|
595
|
+
a(f"== conformance v{rep['report_version']} · {rep['root']}")
|
|
596
|
+
a(f" verdict: {rep['verdict']} nodes={rep['nodes']} "
|
|
597
|
+
f"checks={rep['counts']['total']} (fail {rep['counts']['fail']} / "
|
|
598
|
+
f"warn {rep['counts']['warn']} / blindspot {rep['counts']['blindspot']})")
|
|
599
|
+
a("\n-- 断言集(FAIL=fail-closed 不变量 / WARN=健康指标只告警)--")
|
|
600
|
+
for c in rep["checks"]:
|
|
601
|
+
mark = {"FAIL": "!!", "WARN": " ?", "BLINDSPOT": " ~"}.get(c["level"], " .")
|
|
602
|
+
ok = "ok " if c["ok"] else "NO "
|
|
603
|
+
a(f" [{mark}] {ok}{c['id']}: {c['detail']}")
|
|
604
|
+
c = rep["coverage"]
|
|
605
|
+
a(f"\n-- 口径复核(9-15 基线)--")
|
|
606
|
+
a(f" nodes={c['nodes']} knowledge={c['knowledge']} 混层={c['mixed_layer']}"
|
|
607
|
+
f" ({c['mixed_ratio'] * 100:.1f}%)")
|
|
608
|
+
a(f" dir_mismatch={rep['paths']['layer_mismatch']} path_missing={rep['paths']['missing']}")
|
|
609
|
+
a(f" 覆盖率[knowledge 层·判据口径] role={c['role_ratio_kn'] * 100:.3f}%"
|
|
610
|
+
f" basis={c['basis_ratio_kn'] * 100:.1f}% evidence={c['evidence_ratio_kn'] * 100:.3f}%")
|
|
611
|
+
a(f" 覆盖率为对照[全库口径] role={c['role_ratio'] * 100:.2f}%"
|
|
612
|
+
f" basis={c['basis_ratio'] * 100:.1f}% evidence={c['evidence_ratio'] * 100:.2f}%"
|
|
613
|
+
f" state={c['state_ratio'] * 100:.1f}%")
|
|
614
|
+
a(f" 重复 content_hash 组={rep['dup']['groups']} 节点={rep['dup']['nodes']}"
|
|
615
|
+
f" 最大组={rep['dup']['largest'][:3]}")
|
|
616
|
+
a(f" 边 total={rep['edges']['total']} 非规范键={rep['edges']['non_canonical']}"
|
|
617
|
+
f" 悬空={rep['edges']['dangling']} 类型={rep['edges']['types']}")
|
|
618
|
+
r = rep["reach"]
|
|
619
|
+
a(f" 触达 distinct={r['distinct']} ratio="
|
|
620
|
+
+ ("N/A" if r['ratio'] is None else f"{r['ratio'] * 100:.1f}%") + f" log_lines={r['lines']}")
|
|
621
|
+
a(f" 闸门 decisions={rep['gate']['decisions_total']} {rep['gate']['decisions']}"
|
|
622
|
+
f" inbox_pending={rep['gate']['inbox_pending']}")
|
|
623
|
+
a(f" 审计 op(尾窗 {rep['gate']['audit_tail_lines']}"
|
|
624
|
+
f"/{rep['gate']['audit_tail_window']} 行)={rep['gate']['audit_ops']}")
|
|
625
|
+
a(f" G3 unanalyzed 派生={rep['unanalyzed']['count']}"
|
|
626
|
+
f" ({rep['unanalyzed']['ratio'] * 100:.1f}%)")
|
|
627
|
+
pr = rep.get("protocol") or {}
|
|
628
|
+
if pr:
|
|
629
|
+
a("\n-- 记忆动词协议 v1 静态对账 --")
|
|
630
|
+
a(f" {pr['doc']} frozen={pr['frozen']} ok={pr['ok']}")
|
|
631
|
+
a(f" 声明={pr['declared']}")
|
|
632
|
+
a(f" live={pr['live']} reserved={pr['reserved']}(未实现即如实登记,不冒充)")
|
|
633
|
+
a(f" MCP 面 op 分支={len(pr['actual'])} 个,其中扩展能力面"
|
|
634
|
+
f"(不属协议 v1){len(pr['extension_ops'])} 个")
|
|
635
|
+
if pr["errors"]:
|
|
636
|
+
for e in pr["errors"]:
|
|
637
|
+
a(f" !! {e}")
|
|
638
|
+
a("\n-- G1 类型空间正交性 --")
|
|
639
|
+
for name, spec in rep["g1"].items():
|
|
640
|
+
if name.startswith("_"):
|
|
641
|
+
continue
|
|
642
|
+
a(f" [{name}] 源={spec['source']} 封闭={spec['closed']}")
|
|
643
|
+
a(f" 实测={spec['used']}")
|
|
644
|
+
if spec["undeclared_used"]:
|
|
645
|
+
a(f" ★未声明已用={spec['undeclared_used']}")
|
|
646
|
+
if spec["declared_unused"]:
|
|
647
|
+
a(f" 声明未用={spec['declared_unused']}")
|
|
648
|
+
a(f" 跨空间重叠={rep['g1']['_cross_space_overlap'] or '无'}")
|
|
649
|
+
a(f" 方向口径分歧={[d['literal'] for d in rep['g1']['_direction_conflicts']]}")
|
|
650
|
+
a(f" 已核对无分歧={[d['literal'] for d in rep['g1']['_verified_consistent']]}")
|
|
651
|
+
a(f" 文档滞后于代码={[d['where'] for d in rep['g1']['_doc_code_drift']]}")
|
|
652
|
+
a(f" 新增类型触点={rep['g1']['_new_type_touchpoints']}")
|
|
653
|
+
return "\n".join(L)
|
|
654
|
+
|
|
655
|
+
|
|
656
|
+
# 生效条件:传入对象可写属性时执行赋值 sustain_module.conformance_summary = report_summary,无返回值(None);
|
|
657
|
+
def register_sustain(sustain_module) -> None: # pragma: no cover
|
|
658
|
+
"""挂点说明(供 sustain 侧最小侵入调用)。
|
|
659
|
+
|
|
660
|
+
周期巡检**不加新 tick**:`_tick_tidy()` 已是「contextual 存量治理」的
|
|
661
|
+
读侧巡检,本断言集复用同一节奏(tidy_interval),由 `report_summary(cg)`
|
|
662
|
+
只取结论不落盘——避免为只读检查新增后台周期与 env 开关。
|
|
663
|
+
"""
|
|
664
|
+
sustain_module.conformance_summary = report_summary
|
|
665
|
+
|
|
666
|
+
|
|
667
|
+
# 生效条件:root 依次取 cg_or_root.root 属性、cg_or_root 本身、_default_root()(前者为假值即回落,故 0/空串会继续回落);check 抛异常时返回 ok=False、verdict="BLINDSPOT"、error,否则返回 ok=(verdict!="FAIL") 与 fail/warn/blindspot/failed_ids/t;
|
|
668
|
+
def report_summary(cg_or_root) -> dict:
|
|
669
|
+
"""给常驻循环用的**轻量**结论(不渲染全文、不写盘):verdict + 计数。"""
|
|
670
|
+
root = getattr(cg_or_root, "root", None) or cg_or_root or _default_root()
|
|
671
|
+
try:
|
|
672
|
+
rep = check(root, check_paths=False)
|
|
673
|
+
except Exception as e: # noqa: BLE001
|
|
674
|
+
return {"ok": False, "verdict": "BLINDSPOT", "error": f"{type(e).__name__}: {e}"}
|
|
675
|
+
return {"ok": rep["verdict"] != "FAIL", "verdict": rep["verdict"],
|
|
676
|
+
"fail": rep["counts"]["fail"], "warn": rep["counts"]["warn"],
|
|
677
|
+
"blindspot": rep["counts"]["blindspot"],
|
|
678
|
+
"failed_ids": [c["id"] for c in rep["checks"]
|
|
679
|
+
if c["level"] == "FAIL" and not c["ok"]],
|
|
680
|
+
"t": rep["t"]}
|
|
681
|
+
|
|
682
|
+
|
|
683
|
+
# 生效条件:argv 为 None 时 argparse 从 sys.argv 取值,--root 缺失或为假值回落 _default_root();check 抛 OSError/ValueError(如索引不可读)时打印 BLINDSPOT 并返回 2,否则按 --write-baseline/--json 写盘后返回 0(verdict≠"FAIL")或 1("FAIL");
|
|
684
|
+
def main(argv=None) -> int:
|
|
685
|
+
p = argparse.ArgumentParser(prog="python -m md_cg.conformance",
|
|
686
|
+
description="数据健康不变量断言集(只读,零写入)")
|
|
687
|
+
p.add_argument("--root", default=None,
|
|
688
|
+
help="认知图根(默认 MDCG_ROOT > paths.json > 用户级状态根 data/mdcg)")
|
|
689
|
+
p.add_argument("--json", dest="json_out", default=None)
|
|
690
|
+
p.add_argument("--baseline", default=None, help="基线报告 json(比对不劣化)")
|
|
691
|
+
p.add_argument("--write-baseline", default=None, help="把本次指标写成新基线")
|
|
692
|
+
p.add_argument("--no-path-check", action="store_true", help="不查盘上文件是否存在")
|
|
693
|
+
p.add_argument("--strict", action="store_true", help="BLINDSPOT 也视为非 PASS")
|
|
694
|
+
a = p.parse_args(argv)
|
|
695
|
+
|
|
696
|
+
root = a.root or _default_root()
|
|
697
|
+
baseline = None
|
|
698
|
+
if a.baseline:
|
|
699
|
+
with open(a.baseline, encoding="utf-8") as f:
|
|
700
|
+
baseline = json.load(f)
|
|
701
|
+
try:
|
|
702
|
+
rep = check(root, check_paths=not a.no_path_check,
|
|
703
|
+
baseline=baseline, strict=a.strict)
|
|
704
|
+
except (OSError, ValueError) as e:
|
|
705
|
+
# 数据源不可读 = BLINDSPOT(如实上报),绝不打印「通过」
|
|
706
|
+
print(f"== conformance v{REPORT_VERSION} · {os.path.abspath(root)}")
|
|
707
|
+
print(f" verdict: BLINDSPOT —— 索引不可读: {type(e).__name__}: {e}")
|
|
708
|
+
print(f" 缺失维度: 数据源({INDEX_FILE});修复后重跑,勿以本结果为「通过」。")
|
|
709
|
+
return 2
|
|
710
|
+
print(render(rep))
|
|
711
|
+
if a.write_baseline:
|
|
712
|
+
with open(a.write_baseline, "w", encoding="utf-8") as f:
|
|
713
|
+
json.dump({"report_version": REPORT_VERSION, "t": rep["t"],
|
|
714
|
+
"root": rep["root"], "metrics": metrics_of(rep)},
|
|
715
|
+
f, ensure_ascii=False, indent=2)
|
|
716
|
+
print(f"\n 基线已写入: {a.write_baseline}")
|
|
717
|
+
if a.json_out:
|
|
718
|
+
with open(a.json_out, "w", encoding="utf-8") as f:
|
|
719
|
+
json.dump(rep, f, ensure_ascii=False, indent=2, default=str)
|
|
720
|
+
print(f" 报告已写入: {a.json_out}")
|
|
721
|
+
return 0 if rep["verdict"] != "FAIL" else 1
|
|
722
|
+
|
|
723
|
+
|
|
724
|
+
if __name__ == "__main__":
|
|
725
|
+
if hasattr(sys.stdout, "reconfigure"):
|
|
726
|
+
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
|
727
727
|
sys.exit(main())
|